structflo-ner 0.5.0__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. structflo_ner-0.5.0/README.md → structflo_ner-0.7.0/PKG-INFO +36 -0
  2. structflo_ner-0.5.0/PKG-INFO → structflo_ner-0.7.0/README.md +23 -13
  3. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/pyproject.toml +1 -1
  4. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/__init__.py +1 -1
  5. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/_examples.py +222 -5
  6. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/_mapping.py +4 -0
  7. structflo_ner-0.7.0/structflo/ner/_prompts.py +164 -0
  8. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/extractor.py +69 -2
  9. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/README.md +1 -1
  10. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/_loader.py +7 -2
  11. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/_matcher.py +3 -2
  12. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/tests/test_extractor.py +110 -1
  13. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/tests/test_fast.py +14 -0
  14. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/uv.lock +4 -4
  15. structflo_ner-0.5.0/structflo/ner/_prompts.py +0 -95
  16. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/.github/workflows/ci.yml +0 -0
  17. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/.github/workflows/publish.yml +0 -0
  18. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/.gitignore +0 -0
  19. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/Makefile +0 -0
  20. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/coverage.xml +0 -0
  21. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/images/fast-viz.png +0 -0
  22. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/images/local-gen-pandas.png +0 -0
  23. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/images/local-gen-viz.png +0 -0
  24. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/images/local-tb-viz.png +0 -0
  25. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/images/ner_visualization.gif +0 -0
  26. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/images/struct-flo-ner.png +0 -0
  27. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/notebooks/01_quickstart.ipynb +0 -0
  28. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/notebooks/02_fast_ner.ipynb +0 -0
  29. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/_display.py +0 -0
  30. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/_entities.py +0 -0
  31. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/__init__.py +0 -0
  32. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/_normalize.py +0 -0
  33. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/extractor.py +0 -0
  34. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/accession_number.yml +0 -0
  35. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/compound_name.yml +0 -0
  36. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/disease.yml +0 -0
  37. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/functional_category.yml +0 -0
  38. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/gene_name.yml +0 -0
  39. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/product.yml +0 -0
  40. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/screening_method.yml +0 -0
  41. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/strain.yml +0 -0
  42. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/target.yml +0 -0
  43. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/profiles.py +0 -0
  44. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/tests/__init__.py +0 -0
  45. {structflo_ner-0.5.0 → structflo_ner-0.7.0}/tests/test_entities.py +0 -0
@@ -1,3 +1,16 @@
1
+ Metadata-Version: 2.5
2
+ Name: structflo-ner
3
+ Version: 0.7.0
4
+ Summary: Drug discovery NER wrapper around LangExtract — zero-config entity extraction for chemistry and biology.
5
+ License: Apache-2.0
6
+ Requires-Python: >=3.10
7
+ Requires-Dist: langextract>=1.7.0
8
+ Requires-Dist: pyyaml>=6.0
9
+ Requires-Dist: rapidfuzz>=3.0
10
+ Provides-Extra: dataframe
11
+ Requires-Dist: pandas>=1.5; extra == 'dataframe'
12
+ Description-Content-Type: text/markdown
13
+
1
14
 
2
15
  <h1 align="center">structflo.ner</h1>
3
16
  <p align="center">
@@ -334,6 +347,29 @@ df = result.to_dataframe()
334
347
  result.to_dict()
335
348
  ```
336
349
 
350
+ Each bioactivity carries its measurement in `attributes`:
351
+
352
+ | Attribute | Example | Meaning |
353
+ | --------------- | --------------------------- | ---------------------------------------------------------- |
354
+ | `value` | `>20.0` | number as reported, qualifier kept |
355
+ | `unit` | `µM` | unit as written |
356
+ | `assay_type` | `CC50` | the endpoint (IC50, MIC90, GI50, ED90, ...) |
357
+ | `assay` | `HepG2 MTT` | assay, cell line or read-out the value was measured in |
358
+ | `compound_name` | `8t` | compound the value belongs to |
359
+ | `target` | `InhA` | protein measured against (`TB` profile only) |
360
+ | `strain` | `H37Rv` | organism, strain or virus measured against (`TB` only) |
361
+ | `combination` | `meropenem` | second compound dosed alongside (`TB` profile only) |
362
+
363
+ An attribute the text does not state is absent from `attributes`.
364
+
365
+ A table on one page often names its target only on another. Pass text from
366
+ elsewhere in the document (its title or summary) as `context`; the model uses it
367
+ to fill attributes the page leaves unstated, and extracts nothing from it:
368
+
369
+ ```python
370
+ result = extractor.extract(page_text, context=document_summary)
371
+ ```
372
+
337
373
 
338
374
  ## Notebooks
339
375
 
@@ -1,16 +1,3 @@
1
- Metadata-Version: 2.5
2
- Name: structflo-ner
3
- Version: 0.5.0
4
- Summary: Drug discovery NER wrapper around LangExtract — zero-config entity extraction for chemistry and biology.
5
- License: Apache-2.0
6
- Requires-Python: >=3.10
7
- Requires-Dist: langextract>=1.6.0
8
- Requires-Dist: pyyaml>=6.0
9
- Requires-Dist: rapidfuzz>=3.0
10
- Provides-Extra: dataframe
11
- Requires-Dist: pandas>=1.5; extra == 'dataframe'
12
- Description-Content-Type: text/markdown
13
-
14
1
 
15
2
  <h1 align="center">structflo.ner</h1>
16
3
  <p align="center">
@@ -347,6 +334,29 @@ df = result.to_dataframe()
347
334
  result.to_dict()
348
335
  ```
349
336
 
337
+ Each bioactivity carries its measurement in `attributes`:
338
+
339
+ | Attribute | Example | Meaning |
340
+ | --------------- | --------------------------- | ---------------------------------------------------------- |
341
+ | `value` | `>20.0` | number as reported, qualifier kept |
342
+ | `unit` | `µM` | unit as written |
343
+ | `assay_type` | `CC50` | the endpoint (IC50, MIC90, GI50, ED90, ...) |
344
+ | `assay` | `HepG2 MTT` | assay, cell line or read-out the value was measured in |
345
+ | `compound_name` | `8t` | compound the value belongs to |
346
+ | `target` | `InhA` | protein measured against (`TB` profile only) |
347
+ | `strain` | `H37Rv` | organism, strain or virus measured against (`TB` only) |
348
+ | `combination` | `meropenem` | second compound dosed alongside (`TB` profile only) |
349
+
350
+ An attribute the text does not state is absent from `attributes`.
351
+
352
+ A table on one page often names its target only on another. Pass text from
353
+ elsewhere in the document (its title or summary) as `context`; the model uses it
354
+ to fill attributes the page leaves unstated, and extracts nothing from it:
355
+
356
+ ```python
357
+ result = extractor.extract(page_text, context=document_summary)
358
+ ```
359
+
350
360
 
351
361
  ## Notebooks
352
362
 
@@ -6,7 +6,7 @@ readme = "README.md"
6
6
  requires-python = ">=3.10"
7
7
  license = { text = "Apache-2.0" }
8
8
  dependencies = [
9
- "langextract>=1.6.0",
9
+ "langextract>=1.7.0",
10
10
  "rapidfuzz>=3.0",
11
11
  "PyYAML>=6.0",
12
12
  ]
@@ -63,7 +63,7 @@ from structflo.ner.profiles import (
63
63
  EntityProfile,
64
64
  )
65
65
 
66
- __version__ = "0.5.0"
66
+ __version__ = "0.7.0"
67
67
 
68
68
  __all__ = [
69
69
  # Main classes
@@ -158,8 +158,8 @@ BIOLOGY_EXAMPLES: list[lx.data.ExampleData] = [
158
158
 
159
159
  _BIOACTIVITY_EXAMPLE_1 = lx.data.ExampleData(
160
160
  text=(
161
- "Compound 7 inhibited EGFR with an IC50 of 2.3 nM in a cell-free enzymatic assay "
162
- "and showed an EC50 of 45 nM in A549 (human lung adenocarcinoma) cell proliferation assay. "
161
+ "Compound 7 (CHEMBL5171042) inhibited EGFR with an IC50 of 2.3 nM in a cell-free "
162
+ "enzymatic assay and showed an EC50 of 45 nM in A549 (human lung adenocarcinoma) cell proliferation assay. "
163
163
  "Selectivity over ERBB2 was >100-fold (Ki = 0.8 nM vs 95 nM)."
164
164
  ),
165
165
  extractions=[
@@ -170,7 +170,8 @@ _BIOACTIVITY_EXAMPLE_1 = lx.data.ExampleData(
170
170
  "value": "2.3",
171
171
  "unit": "nM",
172
172
  "assay_type": "IC50",
173
- "compound_name": "Compound 7",
173
+ "compound_name": "CHEMBL5171042",
174
+ "assay": "cell-free enzymatic assay",
174
175
  },
175
176
  ),
176
177
  lx.data.Extraction(
@@ -185,7 +186,8 @@ _BIOACTIVITY_EXAMPLE_1 = lx.data.ExampleData(
185
186
  "value": "45",
186
187
  "unit": "nM",
187
188
  "assay_type": "EC50",
188
- "compound_name": "Compound 7",
189
+ "compound_name": "CHEMBL5171042",
190
+ "assay": "A549 cell proliferation assay",
189
191
  },
190
192
  ),
191
193
  lx.data.Extraction(
@@ -204,7 +206,7 @@ _BIOACTIVITY_EXAMPLE_1 = lx.data.ExampleData(
204
206
  "value": "0.8",
205
207
  "unit": "nM",
206
208
  "assay_type": "Ki",
207
- "compound_name": "Compound 7",
209
+ "compound_name": "CHEMBL5171042",
208
210
  },
209
211
  ),
210
212
  ],
@@ -337,6 +339,7 @@ _FULL_EXAMPLE_1 = lx.data.ExampleData(
337
339
  "unit": "µM",
338
340
  "assay_type": "IC50",
339
341
  "compound_name": "Gefitinib",
342
+ "assay": "cell-free biochemical assay",
340
343
  },
341
344
  ),
342
345
  lx.data.Extraction(
@@ -357,6 +360,7 @@ _FULL_EXAMPLE_1 = lx.data.ExampleData(
357
360
  "unit": "µM",
358
361
  "assay_type": "IC50",
359
362
  "compound_name": "Gefitinib",
363
+ "assay": "A431 cells",
360
364
  },
361
365
  ),
362
366
  lx.data.Extraction(
@@ -449,6 +453,7 @@ _TB_EXAMPLE_1 = lx.data.ExampleData(
449
453
  "assay_type": "MIC",
450
454
  "strain": "H37Rv",
451
455
  "compound_name": "Bedaquiline",
456
+ "assay": "MABA",
452
457
  },
453
458
  ),
454
459
  lx.data.Extraction(
@@ -619,6 +624,8 @@ _TB_EXAMPLE_3 = lx.data.ExampleData(
619
624
  "unit": "nM",
620
625
  "assay_type": "IC50",
621
626
  "compound_name": "Compound 14a",
627
+ "assay": "biochemical assay",
628
+ "target": "InhA",
622
629
  },
623
630
  ),
624
631
  lx.data.Extraction(
@@ -643,6 +650,7 @@ _TB_EXAMPLE_3 = lx.data.ExampleData(
643
650
  "assay_type": "MIC90",
644
651
  "strain": "H37Rv",
645
652
  "compound_name": "Compound 14a",
653
+ "assay": "REMA",
646
654
  },
647
655
  ),
648
656
  lx.data.Extraction(
@@ -658,6 +666,7 @@ _TB_EXAMPLE_3 = lx.data.ExampleData(
658
666
  "unit": "uM",
659
667
  "assay_type": "EC50",
660
668
  "compound_name": "Compound 14a",
669
+ "assay": "THP-1 macrophage infection assay",
661
670
  },
662
671
  ),
663
672
  lx.data.Extraction(
@@ -673,6 +682,7 @@ _TB_EXAMPLE_3 = lx.data.ExampleData(
673
682
  "unit": "uM",
674
683
  "assay_type": "MIC",
675
684
  "compound_name": "Compound 14a",
685
+ "assay": "LORA",
676
686
  },
677
687
  ),
678
688
  lx.data.Extraction(
@@ -688,6 +698,7 @@ _TB_EXAMPLE_3 = lx.data.ExampleData(
688
698
  "unit": "uM",
689
699
  "assay_type": "CC50",
690
700
  "compound_name": "Compound 14a",
701
+ "assay": "HepG2 cells",
691
702
  },
692
703
  ),
693
704
  ],
@@ -843,6 +854,7 @@ _TB_EXAMPLE_5 = lx.data.ExampleData(
843
854
  "unit": "nM",
844
855
  "assay_type": "IC50",
845
856
  "compound_name": "SACC-3060",
857
+ "target": "DprE1",
846
858
  },
847
859
  ),
848
860
  lx.data.Extraction(
@@ -860,12 +872,217 @@ _TB_EXAMPLE_5 = lx.data.ExampleData(
860
872
  ],
861
873
  )
862
874
 
875
+ # Table as Docling emits it: the column header names the assay, the endpoint
876
+ # and unit come from the footnote, and every cell is its own bioactivity. The
877
+ # ID column, not the deck-local row label, names the compound.
878
+ _TB_EXAMPLE_6 = lx.data.ExampleData(
879
+ text=(
880
+ "Table 2. Whole-cell activity\n\n"
881
+ "| Cpd | ID | MABA | THP-1 | Vero |\n"
882
+ "| --- | --- | --- | --- | --- |\n"
883
+ "| 21a | SACC-4412 | 0.12 | 0.9 | >50 |\n"
884
+ "| 21b | CHEMBL5290347 | <0.03 | 2.4* | 18.5 |\n\n"
885
+ "MABA: MIC90 (µM) against M. tuberculosis H37Rv. THP-1: intracellular EC90 (µM). "
886
+ "Vero: CC50 (µM). *Single determination."
887
+ ),
888
+ extractions=[
889
+ lx.data.Extraction(
890
+ extraction_class="assay",
891
+ extraction_text="MABA",
892
+ attributes={"assay_format": "whole-cell", "strain": "H37Rv"},
893
+ ),
894
+ lx.data.Extraction(
895
+ extraction_class="assay",
896
+ extraction_text="THP-1",
897
+ attributes={"cell_line": "THP-1", "assay_format": "intracellular"},
898
+ ),
899
+ lx.data.Extraction(
900
+ extraction_class="assay",
901
+ extraction_text="Vero",
902
+ attributes={"cell_line": "Vero", "assay_format": "cytotoxicity"},
903
+ ),
904
+ lx.data.Extraction(
905
+ extraction_class="compound_name",
906
+ extraction_text="SACC-4412",
907
+ attributes={"synonyms": "21a"},
908
+ ),
909
+ lx.data.Extraction(
910
+ extraction_class="bioactivity",
911
+ extraction_text="0.12",
912
+ attributes={
913
+ "value": "0.12",
914
+ "unit": "µM",
915
+ "assay_type": "MIC90",
916
+ "strain": "H37Rv",
917
+ "compound_name": "SACC-4412",
918
+ "assay": "MABA",
919
+ },
920
+ ),
921
+ lx.data.Extraction(
922
+ extraction_class="bioactivity",
923
+ extraction_text="0.9",
924
+ attributes={
925
+ "value": "0.9",
926
+ "unit": "µM",
927
+ "assay_type": "EC90",
928
+ "compound_name": "SACC-4412",
929
+ "assay": "THP-1",
930
+ },
931
+ ),
932
+ lx.data.Extraction(
933
+ extraction_class="bioactivity",
934
+ extraction_text=">50",
935
+ attributes={
936
+ "value": ">50",
937
+ "unit": "µM",
938
+ "assay_type": "CC50",
939
+ "compound_name": "SACC-4412",
940
+ "assay": "Vero",
941
+ },
942
+ ),
943
+ lx.data.Extraction(
944
+ extraction_class="compound_name",
945
+ extraction_text="CHEMBL5290347",
946
+ attributes={"synonyms": "21b"},
947
+ ),
948
+ lx.data.Extraction(
949
+ extraction_class="bioactivity",
950
+ extraction_text="<0.03",
951
+ attributes={
952
+ "value": "<0.03",
953
+ "unit": "µM",
954
+ "assay_type": "MIC90",
955
+ "strain": "H37Rv",
956
+ "compound_name": "CHEMBL5290347",
957
+ "assay": "MABA",
958
+ },
959
+ ),
960
+ lx.data.Extraction(
961
+ extraction_class="bioactivity",
962
+ extraction_text="2.4",
963
+ attributes={
964
+ "value": "2.4",
965
+ "unit": "µM",
966
+ "assay_type": "EC90",
967
+ "compound_name": "CHEMBL5290347",
968
+ "assay": "THP-1",
969
+ },
970
+ ),
971
+ lx.data.Extraction(
972
+ extraction_class="bioactivity",
973
+ extraction_text="18.5",
974
+ attributes={
975
+ "value": "18.5",
976
+ "unit": "µM",
977
+ "assay_type": "CC50",
978
+ "compound_name": "CHEMBL5290347",
979
+ "assay": "Vero",
980
+ },
981
+ ),
982
+ ],
983
+ )
984
+
985
+ # Strain panel: organism column headers are the strain, a protein column the
986
+ # target, a cytotoxicity column the assay; a column dosed with a second drug
987
+ # names the combination partner. The disease stays a disease entity.
988
+ _TB_EXAMPLE_7 = lx.data.ExampleData(
989
+ text=(
990
+ "Table 3. Antibacterial activity against hospital-acquired pneumonia isolates\n\n"
991
+ "| Cpd | ID | E. coli ATCC 25922 | K. pneumoniae BAA-1705 | A. baumannii ATCC 19606 "
992
+ "| ATCC 19606 + polymyxin B | HepG2 | hERG |\n"
993
+ "| --- | --- | --- | --- | --- | --- | --- | --- |\n"
994
+ "| 9c | CHEMBL5311027 | 0.25 | 1 | >64 | 2 | >100 | 12.5 |\n\n"
995
+ "MIC (µg/mL); polymyxin B at 0.5 µg/mL. HepG2: CC50 (µM). hERG: IC50 (µM)."
996
+ ),
997
+ extractions=[
998
+ lx.data.Extraction(
999
+ extraction_class="disease",
1000
+ extraction_text="hospital-acquired pneumonia",
1001
+ attributes={"therapeutic_area": "infectious disease"},
1002
+ ),
1003
+ lx.data.Extraction(
1004
+ extraction_class="compound_name",
1005
+ extraction_text="CHEMBL5311027",
1006
+ attributes={"synonyms": "9c"},
1007
+ ),
1008
+ lx.data.Extraction(
1009
+ extraction_class="bioactivity",
1010
+ extraction_text="0.25",
1011
+ attributes={
1012
+ "value": "0.25",
1013
+ "unit": "µg/mL",
1014
+ "assay_type": "MIC",
1015
+ "strain": "E. coli ATCC 25922",
1016
+ "compound_name": "CHEMBL5311027",
1017
+ },
1018
+ ),
1019
+ lx.data.Extraction(
1020
+ extraction_class="bioactivity",
1021
+ extraction_text="1",
1022
+ attributes={
1023
+ "value": "1",
1024
+ "unit": "µg/mL",
1025
+ "assay_type": "MIC",
1026
+ "strain": "K. pneumoniae BAA-1705",
1027
+ "compound_name": "CHEMBL5311027",
1028
+ },
1029
+ ),
1030
+ lx.data.Extraction(
1031
+ extraction_class="bioactivity",
1032
+ extraction_text=">64",
1033
+ attributes={
1034
+ "value": ">64",
1035
+ "unit": "µg/mL",
1036
+ "assay_type": "MIC",
1037
+ "strain": "A. baumannii ATCC 19606",
1038
+ "compound_name": "CHEMBL5311027",
1039
+ },
1040
+ ),
1041
+ lx.data.Extraction(
1042
+ extraction_class="bioactivity",
1043
+ extraction_text="2",
1044
+ attributes={
1045
+ "value": "2",
1046
+ "unit": "µg/mL",
1047
+ "assay_type": "MIC",
1048
+ "strain": "ATCC 19606",
1049
+ "combination": "polymyxin B",
1050
+ "compound_name": "CHEMBL5311027",
1051
+ },
1052
+ ),
1053
+ lx.data.Extraction(
1054
+ extraction_class="bioactivity",
1055
+ extraction_text=">100",
1056
+ attributes={
1057
+ "value": ">100",
1058
+ "unit": "µM",
1059
+ "assay_type": "CC50",
1060
+ "compound_name": "CHEMBL5311027",
1061
+ "assay": "HepG2",
1062
+ },
1063
+ ),
1064
+ lx.data.Extraction(
1065
+ extraction_class="bioactivity",
1066
+ extraction_text="12.5",
1067
+ attributes={
1068
+ "value": "12.5",
1069
+ "unit": "µM",
1070
+ "assay_type": "IC50",
1071
+ "compound_name": "CHEMBL5311027",
1072
+ "target": "hERG",
1073
+ },
1074
+ ),
1075
+ ],
1076
+ )
1077
+
863
1078
  TB_EXAMPLES: list[lx.data.ExampleData] = [
864
1079
  _TB_EXAMPLE_1,
865
1080
  _TB_EXAMPLE_2,
866
1081
  _TB_EXAMPLE_3,
867
1082
  _TB_EXAMPLE_4,
868
1083
  _TB_EXAMPLE_5,
1084
+ _TB_EXAMPLE_6,
1085
+ _TB_EXAMPLE_7,
869
1086
  ]
870
1087
 
871
1088
  TB_CHEMISTRY_EXAMPLES: list[lx.data.ExampleData] = [
@@ -29,6 +29,10 @@ def extraction_to_entity(extraction: lx.data.Extraction) -> NEREntity:
29
29
  attributes: dict[str, str] = {}
30
30
  if extraction.attributes:
31
31
  for key, value in extraction.attributes.items():
32
+ # Strict-schema providers (OpenAI) must emit every attribute key, so an
33
+ # unstated one arrives as null; leave it out rather than store "None".
34
+ if value is None:
35
+ continue
32
36
  if isinstance(value, list):
33
37
  attributes[key] = ", ".join(value)
34
38
  else:
@@ -0,0 +1,164 @@
1
+ """Prompt strings for each built-in EntityProfile."""
2
+
3
+ # Shared by every prompt that extracts bioactivity values. The attribute names
4
+ # must match the keys used in the few-shot examples: langextract derives the
5
+ # output schema from those keys, not from this text.
6
+ _BIOACTIVITY_ATTRIBUTES = (
7
+ "For each bioactivity value capture:\n"
8
+ "- value: the number as reported, keeping qualifiers ('>20.0', '<1.0') and dropping "
9
+ "footnote markers ('0.3*' → '0.3').\n"
10
+ "- unit: as written (nM, µM, µg/mL, mg/kg, %, h).\n"
11
+ "- assay_type: the endpoint only, in plain characters (CC₅₀ → CC50). Common endpoints: "
12
+ "IC50, IC90, EC50, EC90, pIC50, Ki, Kd, '% inhibition at <concentration>'; "
13
+ "MIC, MIC50, MIC90, MIC99, MBC, log10 CFU reduction; CC50, TC50, SI (selectivity index); "
14
+ "GI50, TGI, LC50; DC50, Dmax (degraders); ED50, ED90, %TGI, T/C (in vivo); "
15
+ "PRR (parasite reduction ratio), parasite clearance half-life.\n"
16
+ "- compound_name: the compound the value belongs to, by its name or identifier "
17
+ "(e.g. 'bedaquiline', 'BTZ043', 'CHEMBL1234', 'SACC-3000'); a document-local label "
18
+ "('7a', 'Compound 9b') only when nothing else names the compound.\n"
19
+ )
20
+ _ASSAY_ATTRIBUTE = (
21
+ "- assay: the assay, cell line or read-out the value was measured in, e.g. 'MABA', "
22
+ "'LORA', 'HepG2 MTT', 'THP-1 macrophage infection assay', 'asexual blood-stage', "
23
+ "'liver-stage', 'gametocyte', 'NCI-60', 'hERG', 'P. berghei 4-day suppressive test'. "
24
+ "null when no assay is stated.\n"
25
+ )
26
+ # TB gives each kind of context its own slot: what the value was measured in
27
+ # (assay), against (target, strain), and with (combination).
28
+ _TB_SLOT_ATTRIBUTES = (
29
+ "- assay: the assay system the value was measured in: a cell line, format or read-out "
30
+ "('MABA', 'LORA', 'HepG2 MTT', 'THP-1 macrophage infection assay', 'asexual blood-stage', "
31
+ "'AlphaScreen', 'FP', 'NCI-60', 'P. berghei 4-day suppressive test'). null when none is "
32
+ "stated.\n"
33
+ "- target: the protein the value was measured against: an enzyme, receptor, channel or "
34
+ "other molecular target, as written ('InhA', 'DprE1', 'hERG', 'KasA', 'PDE4B', "
35
+ "'BACE1'). null for whole-cell, organism and cytotoxicity values unless one is named.\n"
36
+ "- strain: the organism the value was measured against, as written: species, strain, "
37
+ "isolate or virus (H37Rv, M. tuberculosis, E. coli ATCC 25922, A. baumannii ATCC 19606, "
38
+ "Pf3D7, Dd2, S. mansoni, HIV-1, norovirus). An organism always goes here, never in assay or target, even "
39
+ "when it heads a table column or is all that names the assay. null when not stated.\n"
40
+ "- combination: a second compound dosed together with the one measured (a combination "
41
+ "or potentiation partner), as written ('avibactam', 'polymyxin B 0.5 µg/mL'). Not a "
42
+ "reference compound the value is relative to ('% of isoproterenol') or an agonist, substrate "
43
+ "or tracer the assay uses. null for a single agent.\n"
44
+ )
45
+ _BIOACTIVITY_TABLES = (
46
+ "In a data table, extract every cell value as its own bioactivity. An ID or registry "
47
+ "column, when present, supplies the compound_name and the row label is only its synonym; "
48
+ "otherwise the row label is the compound_name. A column header fills the attribute for "
49
+ "what it names (an assay or cell line, a protein target, an organism or strain), and the "
50
+ "endpoint and unit often come from the caption or a footnote ('CC50 in µM')."
51
+ )
52
+
53
+ CHEMISTRY_PROMPT = (
54
+ "Extract chemical entities from the text. "
55
+ "Include: compound names (generic names, IUPAC names, code names like 'Compound 5b', "
56
+ "brand names, and abbreviations), SMILES strings (the exact SMILES notation as written), "
57
+ "CAS registry numbers (e.g. '50-78-2'), and molecular formulas (e.g. 'C9H8O4'). "
58
+ "Do not infer or generate SMILES — only extract them if explicitly present in the text."
59
+ )
60
+
61
+ BIOLOGY_PROMPT = (
62
+ "Extract biological target entities from the text. "
63
+ "Include: protein targets (e.g. 'EGFR', 'CDK4/6', 'PD-L1'), gene names (e.g. 'KRAS', 'TP53', "
64
+ "'BRCA1'), receptor names, enzyme names, and pathway names. "
65
+ "For each target, capture the gene symbol if mentioned alongside a protein name. "
66
+ "Capture the organism context if specified (e.g. 'human', 'mouse')."
67
+ )
68
+
69
+ BIOACTIVITY_PROMPT = (
70
+ "Extract bioactivity measurements and assay data from the text: potency, selectivity, "
71
+ "cytotoxicity and efficacy values.\n"
72
+ + _BIOACTIVITY_ATTRIBUTES
73
+ + _ASSAY_ATTRIBUTE
74
+ + _BIOACTIVITY_TABLES
75
+ + "\nAlso extract assay descriptions: cell lines used (e.g. 'HeLa', 'A549'), assay formats "
76
+ "(e.g. 'cell viability', 'binding assay', 'enzymatic assay'), and organisms."
77
+ )
78
+
79
+ DISEASE_PROMPT = (
80
+ "Extract disease names and clinical indications from the text. "
81
+ "Include: cancer types (e.g. 'non-small cell lung cancer', 'AML', 'NSCLC'), "
82
+ "non-oncology diseases (e.g. 'type 2 diabetes', 'rheumatoid arthritis'), "
83
+ "and therapeutic areas (e.g. 'oncology', 'CNS'). "
84
+ "Capture both full names and abbreviations. "
85
+ "For each disease, note the therapeutic area if discernible from context."
86
+ )
87
+
88
+ FULL_PROMPT = (
89
+ "Extract all drug discovery entities from the text. "
90
+ "This includes:\n"
91
+ "- Chemical entities: compound names (generic, IUPAC, code names, brand names), "
92
+ "SMILES strings (only if explicitly written), CAS numbers, molecular formulas.\n"
93
+ "- Biological targets: protein names, gene names, receptor names, enzyme names, pathways.\n"
94
+ "- Bioactivity data: potency, selectivity, cytotoxicity and efficacy values "
95
+ "(attributes below).\n"
96
+ "- Assay information: cell lines, assay formats, experimental organisms.\n"
97
+ "- Diseases and indications: cancer types, disease names, therapeutic areas.\n"
98
+ "- Mechanisms of action: binding modes, inhibition types, selectivity descriptions.\n"
99
+ + _BIOACTIVITY_ATTRIBUTES
100
+ + _ASSAY_ATTRIBUTE
101
+ + _BIOACTIVITY_TABLES
102
+ + "\nExtract only what is explicitly stated; do not infer or generate values."
103
+ )
104
+
105
+ # ── Tuberculosis early drug discovery prompts ──────────────────────────
106
+
107
+ TB_PROMPT = (
108
+ "Extract drug discovery entities from this tuberculosis research text.\n\n"
109
+ + _BIOACTIVITY_ATTRIBUTES
110
+ + _TB_SLOT_ATTRIBUTES
111
+ + _BIOACTIVITY_TABLES
112
+ + "\n\n"
113
+ "DISAMBIGUATION RULES:\n"
114
+ "- Mycobacterial proteins (e.g. ClpC1, DprE1, InhA, AtpE, MmpL3, QcrB) "
115
+ "are biological targets, NOT compounds.\n"
116
+ "- Rv locus tags (Rv3790, Rv1484), UniProt IDs (P9WPS1), and PDB codes "
117
+ "are accession_number, not target or gene_name.\n"
118
+ "- Compound registry and programme identifiers (CHEMBL4521987, ZINC000012345678, "
119
+ "DB00945, SACC-3060, TBDA-01187, GSK3036656) are compound_name; only Rv locus "
120
+ "tags, UniProt IDs, PDB codes and RefSeq IDs are accession_number.\n"
121
+ "- A compound with both a document-local label (a number with an optional letter: 7a, "
122
+ "12, 'Compound 9b') and a longer identifier (registry ID; programme, partner or vendor "
123
+ "code: CHEMBL…, SACC-…, a ChemBridge number) is extracted once, as the identifier, with "
124
+ "the label in synonyms; its bioactivities use the identifier as compound_name. A label "
125
+ "is the compound_name only when the document gives nothing else.\n"
126
+ "- Enzyme descriptions like 'enoyl-ACP reductase' are product, not target.\n"
127
+ "- 'cell wall', 'lipid metabolism' are functional_category, not mechanism_of_action.\n"
128
+ "- 'fragment screening', 'biochemical assay' are screening_method, not assay.\n"
129
+ "- A disease or infection is a disease entity wherever it is named, in prose, a heading "
130
+ "or a table cell (tuberculosis, HIV infection, malaria, schistosomiasis, glioblastoma in 'U87 (glioblastoma)'), "
131
+ "even when its organism also fills a bioactivity's strain.\n"
132
+ "- Use target for proteins in a drug-targeting context, gene_name for loci, "
133
+ "protein_name for non-drug-target proteins.\n\n"
134
+ "Extract only what is explicitly stated; do not infer or generate values."
135
+ )
136
+
137
+ TB_CHEMISTRY_PROMPT = (
138
+ "Extract chemical entities from this tuberculosis drug discovery text. "
139
+ "Include compound names, SMILES (only if explicitly written), CAS numbers, "
140
+ "and molecular formulas. "
141
+ "Mycobacterial proteins (ClpC1, DprE1, InhA, AtpE, etc.) are NOT compounds. "
142
+ "Extract only what is explicitly stated; do not infer or generate values."
143
+ )
144
+
145
+ TB_BIOLOGY_PROMPT = (
146
+ "Extract biological entities from this tuberculosis research text. "
147
+ "Use target for proteins in a drug-targeting context, gene_name for loci, "
148
+ "protein_name for non-drug-target proteins. "
149
+ "Rv locus tags and UniProt IDs are accession_number. "
150
+ "Enzyme descriptions (e.g. 'enoyl-ACP reductase') are product. "
151
+ "Protein functional categories (e.g. 'cell wall', 'lipid metabolism') "
152
+ "are functional_category. "
153
+ "Extract only what is explicitly stated; do not infer or generate values."
154
+ )
155
+
156
+ # Framing for NERExtractor.extract(context=...), rendered by langextract ahead of
157
+ # the examples.
158
+ DOCUMENT_CONTEXT = (
159
+ "Document context, from elsewhere in the same document. It is not the text to extract "
160
+ "from: take no entity and no value from it, and do not pair a value in the text with a "
161
+ "number or claim made here. Use it only to fill a bioactivity attribute the text leaves "
162
+ "unstated, such as the target or organism of a table, when the context names one for "
163
+ "the whole document.\n"
164
+ )
@@ -3,15 +3,43 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import logging
6
+ import re
6
7
 
7
8
  import langextract as lx
8
9
 
10
+ from structflo.ner import _prompts
9
11
  from structflo.ner._entities import NERResult
10
12
  from structflo.ner._mapping import annotated_doc_to_result
11
13
  from structflo.ner.profiles import FULL, EntityProfile
12
14
 
13
15
  logger = logging.getLogger(__name__)
14
16
 
17
+ # A table cell holding only a value: '0.94', '>32', '2.4*', '0.62µM'.
18
+ _VALUE_CELL = re.compile(r"^[<>≤≥~=]*\s*\d[\d,]*\.?\d*\s*\*?\s*(%|[nµu]M|µg/mL|mg/kg)?$")
19
+
20
+
21
+ def _bioactivity_count(doc: lx.data.AnnotatedDocument) -> int:
22
+ return sum(1 for ext in doc.extractions if ext.extraction_class == "bioactivity")
23
+
24
+
25
+ def _table_values_dropped(text: str, doc: lx.data.AnnotatedDocument) -> bool:
26
+ """True when fewer than half of a page's table values came back as bioactivities.
27
+
28
+ Models now and then return a complete, valid answer that lists a table's
29
+ compounds but leaves out its values. On the ChEMBL-gold decks this fires on
30
+ ~2% of pages and catches ~60% of such dropouts.
31
+ """
32
+ # ponytail: markdown tables only; values in prose and qualitative results
33
+ # are not checked, so those dropouts still pass silently.
34
+ cells = sum(
35
+ 1
36
+ for line in text.splitlines()
37
+ if line.lstrip().startswith("|")
38
+ for cell in line.strip().strip("|").split("|")
39
+ if _VALUE_CELL.match(cell.strip())
40
+ )
41
+ return cells >= 5 and _bioactivity_count(doc) < 0.5 * cells
42
+
15
43
 
16
44
  class NERExtractor:
17
45
  """Extract drug discovery entities from text with zero configuration.
@@ -90,12 +118,17 @@ class NERExtractor:
90
118
  self,
91
119
  text: str | list[str],
92
120
  profile: EntityProfile | None = None,
121
+ context: str | None = None,
93
122
  ) -> NERResult | list[NERResult]:
94
123
  """Extract drug discovery entities from text.
95
124
 
96
125
  Args:
97
126
  text: Input text (or list of texts) to process.
98
127
  profile: Override the default profile for this call only.
128
+ context: Text from elsewhere in the same document (its title or
129
+ summary). The model reads it to fill attributes the text leaves
130
+ unstated, such as the target of a table whose target is named
131
+ only on another page; no entities are extracted from it.
99
132
 
100
133
  Returns:
101
134
  A :class:`NERResult` for a single string input, or a list of
@@ -107,7 +140,13 @@ class NERExtractor:
107
140
 
108
141
  results = []
109
142
  for single_text in texts:
110
- doc = self._run_extraction(single_text, active_profile)
143
+ doc = self._run_extraction(single_text, active_profile, context)
144
+ if "bioactivity" in active_profile.entity_classes and _table_values_dropped(
145
+ single_text, doc
146
+ ):
147
+ logger.warning("Table values missing from extraction; retrying once.")
148
+ retry = self._run_extraction(single_text, active_profile, context)
149
+ doc = max(doc, retry, key=_bioactivity_count)
111
150
  results.append(annotated_doc_to_result(doc, single_text))
112
151
 
113
152
  return results if is_batch else results[0]
@@ -198,6 +237,7 @@ class NERExtractor:
198
237
  self,
199
238
  text: str,
200
239
  profile: EntityProfile,
240
+ context: str | None = None,
201
241
  ) -> lx.data.AnnotatedDocument:
202
242
  """Call lx.extract and return the AnnotatedDocument."""
203
243
  examples = self._build_examples(profile)
@@ -206,6 +246,8 @@ class NERExtractor:
206
246
  kwargs: dict = dict(self._langextract_kwargs)
207
247
  kwargs.setdefault("use_schema_constraints", True)
208
248
  kwargs.setdefault("show_progress", False)
249
+ if context:
250
+ kwargs["additional_context"] = _prompts.DOCUMENT_CONTEXT + context
209
251
 
210
252
  if self._provider is not None:
211
253
  # Explicit provider → deterministic routing via ModelConfig.
@@ -236,4 +278,29 @@ class NERExtractor:
236
278
 
237
279
  # Post-process: drop extractions with hallucinated entity classes
238
280
  allowed = set(profile.entity_classes)
239
- return self._filter_extractions(doc, allowed)
281
+ doc = self._filter_extractions(doc, allowed)
282
+ if context:
283
+ doc = self._drop_context_copies(doc, context)
284
+ return doc
285
+
286
+ @staticmethod
287
+ def _drop_context_copies(
288
+ doc: lx.data.AnnotatedDocument,
289
+ context: str,
290
+ ) -> lx.data.AnnotatedDocument:
291
+ """Drop extractions the model took from the context instead of the text.
292
+
293
+ The prompt forbids it, but models still extract from the context now and
294
+ then. Such an extraction does not align to the text, and its text occurs
295
+ in the context but not in the text (an unaligned one that is in the text
296
+ is a missed alignment, not a copy).
297
+ """
298
+ ctx, text = context.casefold(), doc.text.casefold()
299
+ kept = [
300
+ ext
301
+ for ext in doc.extractions
302
+ if ext.char_interval is not None
303
+ or ext.extraction_text.casefold() not in ctx
304
+ or ext.extraction_text.casefold() in text
305
+ ]
306
+ return lx.data.AnnotatedDocument(text=doc.text, extractions=kept)
@@ -66,7 +66,7 @@ so a deployment's own gazetteer decides which ones apply:
66
66
  | Seed | Auto-derived Pattern | Matches |
67
67
  |---|---|---|
68
68
  | `Rv0005` | `Rv\d{4}[c]?` | All Rv locus tags |
69
- | `MT0005` | `MT\w+` | Mycobrowser IDs |
69
+ | `MT0005` | `MT[A-Z]{0,2}\d{2,}\w*` | Mycobrowser IDs (not MTT, MTX) |
70
70
  | `P9WGR1` | `[OPQ][0-9][A-Z0-9]{3}[0-9]` | UniProt accessions |
71
71
  | `4TZK` | `(?=[A-Z0-9]{0,3}[A-Z])[1-9][A-Z0-9]{3}` | PDB codes |
72
72
  | `WP_003407354` | `WP_\d+` | NCBI RefSeq proteins |
@@ -89,8 +89,13 @@ _COMPOUND_ID_PATTERNS: list[IdPattern] = [
89
89
  _ACCESSION_PATTERNS: list[tuple[re.Pattern[str], re.Pattern[str], str]] = [
90
90
  # Rv locus tags: Rv0005, Rv3854c
91
91
  (re.compile(r"^Rv\d{4}[c]?$"), re.compile(r"\bRv\d{4}[c]?\b"), "Rv locus tag"),
92
- # Mycobrowser MT IDs: MT0005, MTCI00.01
93
- (re.compile(r"^MT\w+$"), re.compile(r"\bMT\w+\b"), "Mycobrowser ID"),
92
+ # Mycobrowser MT IDs: MT0005, MT18B_0001, MTB000001. The digit run keeps
93
+ # MTT (viability assay), MTX and MTD from matching.
94
+ (
95
+ re.compile(r"^MT[A-Z]{0,2}\d{2,}\w*$"),
96
+ re.compile(r"\bMT[A-Z]{0,2}\d{2,}\w*\b"),
97
+ "Mycobrowser ID",
98
+ ),
94
99
  # UniProt accessions: P9WGR1, O53617
95
100
  (
96
101
  re.compile(r"^[OPQ][0-9][A-Z0-9]{3}[0-9]$"),
@@ -275,8 +275,9 @@ def _build_position_map(original: str, normalized: str) -> list[int]:
275
275
 
276
276
  # Advance original pointer to find the matching character
277
277
  while orig_idx < len(original):
278
- orig_lower = original[orig_idx].lower()
279
- if orig_lower == norm_char:
278
+ # normalize() the char too: a newline became " " and an en dash "-",
279
+ # and comparing raw chars would desync every position after them.
280
+ if (normalize(original[orig_idx]) or " ") == norm_char:
280
281
  position_map.append(orig_idx)
281
282
  orig_idx += 1
282
283
  break
@@ -7,6 +7,7 @@ from unittest.mock import MagicMock
7
7
  import langextract as lx
8
8
 
9
9
  from structflo.ner import (
10
+ BIOACTIVITY,
10
11
  BIOLOGY,
11
12
  CHEMISTRY,
12
13
  FULL,
@@ -121,6 +122,50 @@ class TestNERExtractorExtract:
121
122
  result = extractor.extract("My source text")
122
123
  assert result.source_text == "My source text"
123
124
 
125
+ _TABLE = "| cpd | MIC | CC50 |\n| --- | --- | --- |\n" + "".join(
126
+ f"| C{i} | {i}.5 | >50 |\n" for i in range(5)
127
+ )
128
+
129
+ def _bio_doc(self, n: int) -> lx.data.AnnotatedDocument:
130
+ return _make_annotated_doc(
131
+ [
132
+ lx.data.Extraction(extraction_class="bioactivity", extraction_text=f"{i}.5")
133
+ for i in range(n)
134
+ ]
135
+ )
136
+
137
+ def test_retries_once_when_table_values_are_missing(self):
138
+ extractor = NERExtractor(profile=TB)
139
+ extractor._run_extraction = MagicMock(side_effect=[self._bio_doc(0), self._bio_doc(10)])
140
+ result = extractor.extract(self._TABLE)
141
+ assert extractor._run_extraction.call_count == 2
142
+ assert len(result.bioactivities) == 10
143
+
144
+ def test_no_retry_when_table_values_came_back(self):
145
+ extractor = NERExtractor(profile=TB)
146
+ extractor._run_extraction = MagicMock(return_value=self._bio_doc(10))
147
+ extractor.extract(self._TABLE)
148
+ assert extractor._run_extraction.call_count == 1
149
+
150
+ def test_no_retry_for_profiles_without_bioactivity(self):
151
+ extractor = NERExtractor(profile=CHEMISTRY)
152
+ extractor._run_extraction = MagicMock(return_value=self._bio_doc(0))
153
+ extractor.extract(self._TABLE)
154
+ assert extractor._run_extraction.call_count == 1
155
+
156
+ def test_null_attributes_are_absent_not_none_strings(self):
157
+ # OpenAI strict mode returns every schema key, null when unstated.
158
+ extractor = self._extractor_with_mock(
159
+ [
160
+ lx.data.Extraction(
161
+ extraction_class="bioactivity",
162
+ extraction_text="IC50 = 3 nM",
163
+ attributes={"value": "3", "strain": None, "assay": None},
164
+ )
165
+ ]
166
+ )
167
+ assert extractor.extract("text").bioactivities[0].attributes == {"value": "3"}
168
+
124
169
 
125
170
  class TestProviderRouting:
126
171
  """Explicit provider selection routes deterministically via ModelConfig,
@@ -180,6 +225,24 @@ class TestProviderRouting:
180
225
  assert pk["base_url"] == "http://ollama:11434"
181
226
  assert pk["num_ctx"] == 8192
182
227
 
228
+ def test_context_reaches_langextract_framed(self, monkeypatch):
229
+ from structflo.ner import extractor as extractor_mod
230
+
231
+ captured = {}
232
+
233
+ def fake_extract(**kwargs):
234
+ captured.update(kwargs)
235
+ return _make_annotated_doc([])
236
+
237
+ monkeypatch.setattr(extractor_mod.lx, "extract", fake_extract)
238
+ NERExtractor().extract("IC50s table", context="Deck on InhA inhibitors.")
239
+ ctx = captured["additional_context"]
240
+ assert ctx.startswith("Document context") and ctx.endswith("Deck on InhA inhibitors.")
241
+
242
+ def test_no_context_passes_no_additional_context(self, monkeypatch):
243
+ captured = self._run_and_capture(monkeypatch)
244
+ assert "additional_context" not in captured
245
+
183
246
  def test_cloud_provider_kwargs_carry_api_key(self, monkeypatch):
184
247
  captured = self._run_and_capture(
185
248
  monkeypatch, provider="openai", model_id="gpt-4o", api_key="sk-test"
@@ -234,6 +297,27 @@ class TestFilterExtractions:
234
297
  assert len(filtered.extractions) == 0
235
298
 
236
299
 
300
+ class TestDropContextCopies:
301
+ def test_drops_only_unaligned_extractions_found_in_context(self):
302
+ aligned = lx.data.Extraction(
303
+ extraction_class="target",
304
+ extraction_text="InhA",
305
+ char_interval=lx.data.CharInterval(start_pos=0, end_pos=4),
306
+ )
307
+ copied = lx.data.Extraction(extraction_class="compound_name", extraction_text="Isoniazid")
308
+ unaligned = lx.data.Extraction(extraction_class="compound_name", extraction_text="7a")
309
+ # In the context too, but also on the page: a missed alignment, not a copy.
310
+ on_page = lx.data.Extraction(extraction_class="disease", extraction_text="malaria")
311
+ doc = lx.data.AnnotatedDocument(
312
+ text="InhA IC50s for 7a; malaria panel",
313
+ extractions=[aligned, copied, unaligned, on_page],
314
+ )
315
+ kept = NERExtractor._drop_context_copies(
316
+ doc, "InhA inhibitors that match isoniazid, and a malaria screen"
317
+ )
318
+ assert kept.extractions == [aligned, unaligned, on_page]
319
+
320
+
237
321
  class TestBuildExamples:
238
322
  def test_extra_examples_appended(self):
239
323
  extra = lx.data.ExampleData(text="extra", extractions=[])
@@ -284,8 +368,33 @@ class TestTBProfile:
284
368
  for cls in TB_BIOLOGY.entity_classes:
285
369
  assert cls in TB.entity_classes
286
370
 
371
+ def test_bioactivity_schema_has_assay(self):
372
+ """langextract builds the output schema from example attribute keys, so an
373
+ `assay` slot exists only if some bioactivity example carries one."""
374
+ for profile in (TB, BIOACTIVITY, FULL):
375
+ keys = {
376
+ k
377
+ for ex in profile.examples
378
+ for e in ex.extractions
379
+ if e.extraction_class == "bioactivity"
380
+ for k in (e.attributes or {})
381
+ }
382
+ assert {"value", "unit", "assay_type", "compound_name", "assay"} <= keys, profile.name
383
+
384
+ def test_tb_bioactivity_schema_has_context_slots(self):
385
+ """TB gives what a value was measured against (target, strain) and with
386
+ (combination) their own slots, apart from assay."""
387
+ keys = {
388
+ k
389
+ for ex in TB.examples
390
+ for e in ex.extractions
391
+ if e.extraction_class == "bioactivity"
392
+ for k in (e.attributes or {})
393
+ }
394
+ assert {"target", "strain", "combination"} <= keys
395
+
287
396
  def test_tb_examples_count(self):
288
- assert len(TB.examples) == 5
397
+ assert len(TB.examples) == 7
289
398
  assert len(TB_CHEMISTRY.examples) == 2
290
399
  assert len(TB_BIOLOGY.examples) == 2
291
400
 
@@ -180,6 +180,13 @@ class TestDeriveIdPatterns:
180
180
  assert not pdb.regex.search("in 2019 the trial")
181
181
  assert not pdb.regex.search("n = 1234")
182
182
 
183
+ def test_mycobrowser_pattern_needs_locus_digits(self):
184
+ """MTT (viability assay), MTX and MTD are not Mycobrowser loci."""
185
+ mt = next(p for p in _accessions(["MT18B_0001"]) if p.description == "Mycobrowser ID")
186
+ for locus in ["MT0005", "MT18B_0001", "MTB000001"]:
187
+ assert mt.regex.fullmatch(locus), locus
188
+ assert not mt.regex.search("HepG2 MTT, MTX and MTD read-outs")
189
+
183
190
  def test_cas_check_digit_rejects_dates(self):
184
191
  cas = next(p for p in derive_id_patterns({}) if p.description == "CAS number")
185
192
  assert cas.validate("50-78-2") # aspirin
@@ -246,6 +253,13 @@ class TestGazetteerMatcher:
246
253
  m = inha_matches[0]
247
254
  assert text[m.char_start : m.char_end] == "InhA"
248
255
 
256
+ def test_char_offsets_survive_newlines(self):
257
+ """normalize() folds a newline into a space; offsets after it must not drift."""
258
+ matcher = self._simple_matcher(fuzzy_threshold=0)
259
+ text = "Targets:\nINHA and\n\nDPRE1."
260
+ found = {m.canonical: text[m.char_start : m.char_end] for m in matcher.match(text)}
261
+ assert found == {"InhA": "INHA", "DprE1": "DPRE1"}
262
+
249
263
  def test_regex_accession_matching(self):
250
264
  matcher = GazetteerMatcher(
251
265
  gazetteers={"target": ["InhA"]},
@@ -917,7 +917,7 @@ wheels = [
917
917
 
918
918
  [[package]]
919
919
  name = "langextract"
920
- version = "1.6.0"
920
+ version = "1.7.0"
921
921
  source = { registry = "https://pypi.org/simple" }
922
922
  dependencies = [
923
923
  { name = "absl-py" },
@@ -940,9 +940,9 @@ dependencies = [
940
940
  { name = "tqdm" },
941
941
  { name = "typing-extensions" },
942
942
  ]
943
- sdist = { url = "https://files.pythonhosted.org/packages/49/c2/30216c49a53419bf3c124cb3ea6daba95a617a62e5ea4964b9fb11784f1e/langextract-1.6.0.tar.gz", hash = "sha256:ac6412152b173bfb0d6b4e08bf6ace7319b1e734c4f9c6f33d39f0c5915e640d", size = 142637, upload-time = "2026-07-02T06:24:00.632Z" }
943
+ sdist = { url = "https://files.pythonhosted.org/packages/e3/e7/6cf877cdd3d5d3a479c0688ff2b63c665c4319ae35ba00734c9fef36a1d2/langextract-1.7.0.tar.gz", hash = "sha256:cb6dc515583b0fdc0f9eff9dce61b2a3c7c84b0668add4dd4e7be364f4c188ba", size = 147514, upload-time = "2026-09-13T18:58:06.376Z" }
944
944
  wheels = [
945
- { url = "https://files.pythonhosted.org/packages/d7/e0/3475ffd1c7e2a4a49f3d8c186c27869a53c710335742964707dbc499590e/langextract-1.6.0-py3-none-any.whl", hash = "sha256:ffa2d584c29ea25c73a2723cec756f21a76799e8ab76a1d9d92841e27504b057", size = 150147, upload-time = "2026-07-02T06:23:59.25Z" },
945
+ { url = "https://files.pythonhosted.org/packages/ae/ec/2883a93872701045971ea77e30434d72556a557f188df575b154d179854c/langextract-1.7.0-py3-none-any.whl", hash = "sha256:908f8ec696ab13578cdc3862efcf1a4fda929ae426410d99e86977d50e52136d", size = 154297, upload-time = "2026-09-13T18:58:04.896Z" },
946
946
  ]
947
947
 
948
948
  [[package]]
@@ -2170,7 +2170,7 @@ dev = [
2170
2170
 
2171
2171
  [package.metadata]
2172
2172
  requires-dist = [
2173
- { name = "langextract", specifier = ">=1.6.0" },
2173
+ { name = "langextract", specifier = ">=1.7.0" },
2174
2174
  { name = "pandas", marker = "extra == 'dataframe'", specifier = ">=1.5" },
2175
2175
  { name = "pyyaml", specifier = ">=6.0" },
2176
2176
  { name = "rapidfuzz", specifier = ">=3.0" },
@@ -1,95 +0,0 @@
1
- """Prompt strings for each built-in EntityProfile."""
2
-
3
- CHEMISTRY_PROMPT = (
4
- "Extract chemical entities from the text. "
5
- "Include: compound names (generic names, IUPAC names, code names like 'Compound 5b', "
6
- "brand names, and abbreviations), SMILES strings (the exact SMILES notation as written), "
7
- "CAS registry numbers (e.g. '50-78-2'), and molecular formulas (e.g. 'C9H8O4'). "
8
- "Do not infer or generate SMILES — only extract them if explicitly present in the text."
9
- )
10
-
11
- BIOLOGY_PROMPT = (
12
- "Extract biological target entities from the text. "
13
- "Include: protein targets (e.g. 'EGFR', 'CDK4/6', 'PD-L1'), gene names (e.g. 'KRAS', 'TP53', "
14
- "'BRCA1'), receptor names, enzyme names, and pathway names. "
15
- "For each target, capture the gene symbol if mentioned alongside a protein name. "
16
- "Capture the organism context if specified (e.g. 'human', 'mouse')."
17
- )
18
-
19
- BIOACTIVITY_PROMPT = (
20
- "Extract bioactivity measurements and assay data from the text. "
21
- "Include: potency values (IC50, MIC, EC50, Ki, Kd, GI50, CC50), selectivity ratios, "
22
- "percent inhibition values, and Hill coefficients. "
23
- "For each value, capture the numeric value, unit (nM, µM, mM, µg/mL, ng/mL), and measurement type. "
24
- "Also capture the compound — its name, code name, or identifier (e.g. 'Compound 7', 'BTZ043', "
25
- "'CHEMBL1234', 'SACC-3000') — that the measurement applies to, as a 'compound_name' attribute, "
26
- "if it can be identified from the surrounding context. "
27
- "Also extract assay descriptions: cell lines used (e.g. 'HeLa', 'A549'), assay formats "
28
- "(e.g. 'cell viability', 'binding assay', 'enzymatic assay'), and organisms."
29
- )
30
-
31
- DISEASE_PROMPT = (
32
- "Extract disease names and clinical indications from the text. "
33
- "Include: cancer types (e.g. 'non-small cell lung cancer', 'AML', 'NSCLC'), "
34
- "non-oncology diseases (e.g. 'type 2 diabetes', 'rheumatoid arthritis'), "
35
- "and therapeutic areas (e.g. 'oncology', 'CNS'). "
36
- "Capture both full names and abbreviations. "
37
- "For each disease, note the therapeutic area if discernible from context."
38
- )
39
-
40
- FULL_PROMPT = (
41
- "Extract all drug discovery entities from the text. "
42
- "This includes:\n"
43
- "- Chemical entities: compound names (generic, IUPAC, code names, brand names), "
44
- "SMILES strings (only if explicitly written), CAS numbers, molecular formulas.\n"
45
- "- Biological targets: protein names, gene names, receptor names, enzyme names, pathways.\n"
46
- "- Bioactivity data: IC50, MIC, EC50, Ki, Kd, and other potency/selectivity measurements "
47
- "with their numeric values, units, and the compound name or identifier they apply to "
48
- "(captured as a 'compound_name' attribute).\n"
49
- "- Assay information: cell lines, assay formats, experimental organisms.\n"
50
- "- Diseases and indications: cancer types, disease names, therapeutic areas.\n"
51
- "- Mechanisms of action: binding modes, inhibition types, selectivity descriptions.\n"
52
- "Extract only what is explicitly stated; do not infer or generate values."
53
- )
54
-
55
- # ── Tuberculosis early drug discovery prompts ──────────────────────────
56
-
57
- TB_PROMPT = (
58
- "Extract drug discovery entities from this tuberculosis research text.\n\n"
59
- "For each bioactivity measurement (IC50, MIC, MIC90, EC50, CC50, etc.), capture the "
60
- "compound name, code, or identifier it belongs to as a 'compound_name' attribute "
61
- "(e.g. 'BTZ043 showed a MIC of 1 ng/mL' → compound_name: 'BTZ043').\n\n"
62
- "DISAMBIGUATION RULES:\n"
63
- "- Mycobacterial proteins (e.g. ClpC1, DprE1, InhA, AtpE, MmpL3, QcrB) "
64
- "are biological targets, NOT compounds.\n"
65
- "- Rv locus tags (Rv3790, Rv1484), UniProt IDs (P9WPS1), and PDB codes "
66
- "are accession_number, not target or gene_name.\n"
67
- "- Compound registry and programme identifiers (CHEMBL4521987, ZINC000012345678, "
68
- "DB00945, SACC-3060, TBDA-01187, GSK3036656) are compound_name; only Rv locus "
69
- "tags, UniProt IDs, PDB codes and RefSeq IDs are accession_number.\n"
70
- "- Enzyme descriptions like 'enoyl-ACP reductase' are product, not target.\n"
71
- "- 'cell wall', 'lipid metabolism' are functional_category, not mechanism_of_action.\n"
72
- "- 'fragment screening', 'biochemical assay' are screening_method, not assay.\n"
73
- "- Use target for proteins in a drug-targeting context, gene_name for loci, "
74
- "protein_name for non-drug-target proteins.\n\n"
75
- "Extract only what is explicitly stated; do not infer or generate values."
76
- )
77
-
78
- TB_CHEMISTRY_PROMPT = (
79
- "Extract chemical entities from this tuberculosis drug discovery text. "
80
- "Include compound names, SMILES (only if explicitly written), CAS numbers, "
81
- "and molecular formulas. "
82
- "Mycobacterial proteins (ClpC1, DprE1, InhA, AtpE, etc.) are NOT compounds. "
83
- "Extract only what is explicitly stated; do not infer or generate values."
84
- )
85
-
86
- TB_BIOLOGY_PROMPT = (
87
- "Extract biological entities from this tuberculosis research text. "
88
- "Use target for proteins in a drug-targeting context, gene_name for loci, "
89
- "protein_name for non-drug-target proteins. "
90
- "Rv locus tags and UniProt IDs are accession_number. "
91
- "Enzyme descriptions (e.g. 'enoyl-ACP reductase') are product. "
92
- "Protein functional categories (e.g. 'cell wall', 'lipid metabolism') "
93
- "are functional_category. "
94
- "Extract only what is explicitly stated; do not infer or generate values."
95
- )
File without changes
File without changes