structflo-ner 0.6.0__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. structflo_ner-0.6.0/README.md → structflo_ner-0.7.0/PKG-INFO +25 -3
  2. structflo_ner-0.6.0/PKG-INFO → structflo_ner-0.7.0/README.md +12 -16
  3. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/pyproject.toml +1 -1
  4. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/__init__.py +1 -1
  5. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/_examples.py +96 -0
  6. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/_mapping.py +4 -0
  7. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/_prompts.py +45 -8
  8. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/extractor.py +69 -2
  9. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/tests/test_extractor.py +96 -1
  10. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/uv.lock +1 -1
  11. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/.github/workflows/ci.yml +0 -0
  12. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/.github/workflows/publish.yml +0 -0
  13. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/.gitignore +0 -0
  14. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/Makefile +0 -0
  15. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/coverage.xml +0 -0
  16. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/images/fast-viz.png +0 -0
  17. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/images/local-gen-pandas.png +0 -0
  18. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/images/local-gen-viz.png +0 -0
  19. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/images/local-tb-viz.png +0 -0
  20. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/images/ner_visualization.gif +0 -0
  21. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/images/struct-flo-ner.png +0 -0
  22. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/notebooks/01_quickstart.ipynb +0 -0
  23. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/notebooks/02_fast_ner.ipynb +0 -0
  24. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/_display.py +0 -0
  25. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/_entities.py +0 -0
  26. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/README.md +0 -0
  27. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/__init__.py +0 -0
  28. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/_loader.py +0 -0
  29. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/_matcher.py +0 -0
  30. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/_normalize.py +0 -0
  31. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/extractor.py +0 -0
  32. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/accession_number.yml +0 -0
  33. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/compound_name.yml +0 -0
  34. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/disease.yml +0 -0
  35. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/functional_category.yml +0 -0
  36. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/gene_name.yml +0 -0
  37. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/product.yml +0 -0
  38. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/screening_method.yml +0 -0
  39. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/strain.yml +0 -0
  40. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/target.yml +0 -0
  41. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/profiles.py +0 -0
  42. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/tests/__init__.py +0 -0
  43. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/tests/test_entities.py +0 -0
  44. {structflo_ner-0.6.0 → structflo_ner-0.7.0}/tests/test_fast.py +0 -0
@@ -1,3 +1,16 @@
1
+ Metadata-Version: 2.5
2
+ Name: structflo-ner
3
+ Version: 0.7.0
4
+ Summary: Drug discovery NER wrapper around LangExtract — zero-config entity extraction for chemistry and biology.
5
+ License: Apache-2.0
6
+ Requires-Python: >=3.10
7
+ Requires-Dist: langextract>=1.7.0
8
+ Requires-Dist: pyyaml>=6.0
9
+ Requires-Dist: rapidfuzz>=3.0
10
+ Provides-Extra: dataframe
11
+ Requires-Dist: pandas>=1.5; extra == 'dataframe'
12
+ Description-Content-Type: text/markdown
13
+
1
14
 
2
15
  <h1 align="center">structflo.ner</h1>
3
16
  <p align="center">
@@ -343,10 +356,19 @@ Each bioactivity carries its measurement in `attributes`:
343
356
  | `assay_type` | `CC50` | the endpoint (IC50, MIC90, GI50, ED90, ...) |
344
357
  | `assay` | `HepG2 MTT` | assay, cell line or read-out the value was measured in |
345
358
  | `compound_name` | `8t` | compound the value belongs to |
346
- | `strain` | `H37Rv` | organism or strain (`TB` profile only) |
359
+ | `target` | `InhA` | protein measured against (`TB` profile only) |
360
+ | `strain` | `H37Rv` | organism, strain or virus measured against (`TB` only) |
361
+ | `combination` | `meropenem` | second compound dosed alongside (`TB` profile only) |
362
+
363
+ An attribute the text does not state is absent from `attributes`.
364
+
365
+ A table on one page often names its target only on another. Pass text from
366
+ elsewhere in the document (its title or summary) as `context`; the model uses it
367
+ to fill attributes the page leaves unstated, and extracts nothing from it:
347
368
 
348
- An attribute the text does not state comes back as the string `"None"` with
349
- providers that enforce a strict schema (OpenAI), or is absent otherwise.
369
+ ```python
370
+ result = extractor.extract(page_text, context=document_summary)
371
+ ```
350
372
 
351
373
 
352
374
  ## Notebooks
@@ -1,16 +1,3 @@
1
- Metadata-Version: 2.5
2
- Name: structflo-ner
3
- Version: 0.6.0
4
- Summary: Drug discovery NER wrapper around LangExtract — zero-config entity extraction for chemistry and biology.
5
- License: Apache-2.0
6
- Requires-Python: >=3.10
7
- Requires-Dist: langextract>=1.6.0
8
- Requires-Dist: pyyaml>=6.0
9
- Requires-Dist: rapidfuzz>=3.0
10
- Provides-Extra: dataframe
11
- Requires-Dist: pandas>=1.5; extra == 'dataframe'
12
- Description-Content-Type: text/markdown
13
-
14
1
 
15
2
  <h1 align="center">structflo.ner</h1>
16
3
  <p align="center">
@@ -356,10 +343,19 @@ Each bioactivity carries its measurement in `attributes`:
356
343
  | `assay_type` | `CC50` | the endpoint (IC50, MIC90, GI50, ED90, ...) |
357
344
  | `assay` | `HepG2 MTT` | assay, cell line or read-out the value was measured in |
358
345
  | `compound_name` | `8t` | compound the value belongs to |
359
- | `strain` | `H37Rv` | organism or strain (`TB` profile only) |
346
+ | `target` | `InhA` | protein measured against (`TB` profile only) |
347
+ | `strain` | `H37Rv` | organism, strain or virus measured against (`TB` only) |
348
+ | `combination` | `meropenem` | second compound dosed alongside (`TB` profile only) |
349
+
350
+ An attribute the text does not state is absent from `attributes`.
351
+
352
+ A table on one page often names its target only on another. Pass text from
353
+ elsewhere in the document (its title or summary) as `context`; the model uses it
354
+ to fill attributes the page leaves unstated, and extracts nothing from it:
360
355
 
361
- An attribute the text does not state comes back as the string `"None"` with
362
- providers that enforce a strict schema (OpenAI), or is absent otherwise.
356
+ ```python
357
+ result = extractor.extract(page_text, context=document_summary)
358
+ ```
363
359
 
364
360
 
365
361
  ## Notebooks
@@ -6,7 +6,7 @@ readme = "README.md"
6
6
  requires-python = ">=3.10"
7
7
  license = { text = "Apache-2.0" }
8
8
  dependencies = [
9
- "langextract>=1.6.0",
9
+ "langextract>=1.7.0",
10
10
  "rapidfuzz>=3.0",
11
11
  "PyYAML>=6.0",
12
12
  ]
@@ -63,7 +63,7 @@ from structflo.ner.profiles import (
63
63
  EntityProfile,
64
64
  )
65
65
 
66
- __version__ = "0.6.0"
66
+ __version__ = "0.7.0"
67
67
 
68
68
  __all__ = [
69
69
  # Main classes
@@ -625,6 +625,7 @@ _TB_EXAMPLE_3 = lx.data.ExampleData(
625
625
  "assay_type": "IC50",
626
626
  "compound_name": "Compound 14a",
627
627
  "assay": "biochemical assay",
628
+ "target": "InhA",
628
629
  },
629
630
  ),
630
631
  lx.data.Extraction(
@@ -853,6 +854,7 @@ _TB_EXAMPLE_5 = lx.data.ExampleData(
853
854
  "unit": "nM",
854
855
  "assay_type": "IC50",
855
856
  "compound_name": "SACC-3060",
857
+ "target": "DprE1",
856
858
  },
857
859
  ),
858
860
  lx.data.Extraction(
@@ -980,6 +982,99 @@ _TB_EXAMPLE_6 = lx.data.ExampleData(
980
982
  ],
981
983
  )
982
984
 
985
+ # Strain panel: organism column headers are the strain, a protein column the
986
+ # target, a cytotoxicity column the assay; a column dosed with a second drug
987
+ # names the combination partner. The disease stays a disease entity.
988
+ _TB_EXAMPLE_7 = lx.data.ExampleData(
989
+ text=(
990
+ "Table 3. Antibacterial activity against hospital-acquired pneumonia isolates\n\n"
991
+ "| Cpd | ID | E. coli ATCC 25922 | K. pneumoniae BAA-1705 | A. baumannii ATCC 19606 "
992
+ "| ATCC 19606 + polymyxin B | HepG2 | hERG |\n"
993
+ "| --- | --- | --- | --- | --- | --- | --- | --- |\n"
994
+ "| 9c | CHEMBL5311027 | 0.25 | 1 | >64 | 2 | >100 | 12.5 |\n\n"
995
+ "MIC (µg/mL); polymyxin B at 0.5 µg/mL. HepG2: CC50 (µM). hERG: IC50 (µM)."
996
+ ),
997
+ extractions=[
998
+ lx.data.Extraction(
999
+ extraction_class="disease",
1000
+ extraction_text="hospital-acquired pneumonia",
1001
+ attributes={"therapeutic_area": "infectious disease"},
1002
+ ),
1003
+ lx.data.Extraction(
1004
+ extraction_class="compound_name",
1005
+ extraction_text="CHEMBL5311027",
1006
+ attributes={"synonyms": "9c"},
1007
+ ),
1008
+ lx.data.Extraction(
1009
+ extraction_class="bioactivity",
1010
+ extraction_text="0.25",
1011
+ attributes={
1012
+ "value": "0.25",
1013
+ "unit": "µg/mL",
1014
+ "assay_type": "MIC",
1015
+ "strain": "E. coli ATCC 25922",
1016
+ "compound_name": "CHEMBL5311027",
1017
+ },
1018
+ ),
1019
+ lx.data.Extraction(
1020
+ extraction_class="bioactivity",
1021
+ extraction_text="1",
1022
+ attributes={
1023
+ "value": "1",
1024
+ "unit": "µg/mL",
1025
+ "assay_type": "MIC",
1026
+ "strain": "K. pneumoniae BAA-1705",
1027
+ "compound_name": "CHEMBL5311027",
1028
+ },
1029
+ ),
1030
+ lx.data.Extraction(
1031
+ extraction_class="bioactivity",
1032
+ extraction_text=">64",
1033
+ attributes={
1034
+ "value": ">64",
1035
+ "unit": "µg/mL",
1036
+ "assay_type": "MIC",
1037
+ "strain": "A. baumannii ATCC 19606",
1038
+ "compound_name": "CHEMBL5311027",
1039
+ },
1040
+ ),
1041
+ lx.data.Extraction(
1042
+ extraction_class="bioactivity",
1043
+ extraction_text="2",
1044
+ attributes={
1045
+ "value": "2",
1046
+ "unit": "µg/mL",
1047
+ "assay_type": "MIC",
1048
+ "strain": "ATCC 19606",
1049
+ "combination": "polymyxin B",
1050
+ "compound_name": "CHEMBL5311027",
1051
+ },
1052
+ ),
1053
+ lx.data.Extraction(
1054
+ extraction_class="bioactivity",
1055
+ extraction_text=">100",
1056
+ attributes={
1057
+ "value": ">100",
1058
+ "unit": "µM",
1059
+ "assay_type": "CC50",
1060
+ "compound_name": "CHEMBL5311027",
1061
+ "assay": "HepG2",
1062
+ },
1063
+ ),
1064
+ lx.data.Extraction(
1065
+ extraction_class="bioactivity",
1066
+ extraction_text="12.5",
1067
+ attributes={
1068
+ "value": "12.5",
1069
+ "unit": "µM",
1070
+ "assay_type": "IC50",
1071
+ "compound_name": "CHEMBL5311027",
1072
+ "target": "hERG",
1073
+ },
1074
+ ),
1075
+ ],
1076
+ )
1077
+
983
1078
  TB_EXAMPLES: list[lx.data.ExampleData] = [
984
1079
  _TB_EXAMPLE_1,
985
1080
  _TB_EXAMPLE_2,
@@ -987,6 +1082,7 @@ TB_EXAMPLES: list[lx.data.ExampleData] = [
987
1082
  _TB_EXAMPLE_4,
988
1083
  _TB_EXAMPLE_5,
989
1084
  _TB_EXAMPLE_6,
1085
+ _TB_EXAMPLE_7,
990
1086
  ]
991
1087
 
992
1088
  TB_CHEMISTRY_EXAMPLES: list[lx.data.ExampleData] = [
@@ -29,6 +29,10 @@ def extraction_to_entity(extraction: lx.data.Extraction) -> NEREntity:
29
29
  attributes: dict[str, str] = {}
30
30
  if extraction.attributes:
31
31
  for key, value in extraction.attributes.items():
32
+ # Strict-schema providers (OpenAI) must emit every attribute key, so an
33
+ # unstated one arrives as null; leave it out rather than store "None".
34
+ if value is None:
35
+ continue
32
36
  if isinstance(value, list):
33
37
  attributes[key] = ", ".join(value)
34
38
  else:
@@ -13,20 +13,41 @@ _BIOACTIVITY_ATTRIBUTES = (
13
13
  "MIC, MIC50, MIC90, MIC99, MBC, log10 CFU reduction; CC50, TC50, SI (selectivity index); "
14
14
  "GI50, TGI, LC50; DC50, Dmax (degraders); ED50, ED90, %TGI, T/C (in vivo); "
15
15
  "PRR (parasite reduction ratio), parasite clearance half-life.\n"
16
+ "- compound_name: the compound the value belongs to, by its name or identifier "
17
+ "(e.g. 'bedaquiline', 'BTZ043', 'CHEMBL1234', 'SACC-3000'); a document-local label "
18
+ "('7a', 'Compound 9b') only when nothing else names the compound.\n"
19
+ )
20
+ _ASSAY_ATTRIBUTE = (
16
21
  "- assay: the assay, cell line or read-out the value was measured in, e.g. 'MABA', "
17
22
  "'LORA', 'HepG2 MTT', 'THP-1 macrophage infection assay', 'asexual blood-stage', "
18
23
  "'liver-stage', 'gametocyte', 'NCI-60', 'hERG', 'P. berghei 4-day suppressive test'. "
19
24
  "null when no assay is stated.\n"
20
- "- compound_name: the compound the value belongs to, by its name or identifier "
21
- "(e.g. 'bedaquiline', 'BTZ043', 'CHEMBL1234', 'SACC-3000'); a document-local label "
22
- "('7a', 'Compound 9b') only when nothing else names the compound.\n"
25
+ )
26
+ # TB gives each kind of context its own slot: what the value was measured in
27
+ # (assay), against (target, strain), and with (combination).
28
+ _TB_SLOT_ATTRIBUTES = (
29
+ "- assay: the assay system the value was measured in: a cell line, format or read-out "
30
+ "('MABA', 'LORA', 'HepG2 MTT', 'THP-1 macrophage infection assay', 'asexual blood-stage', "
31
+ "'AlphaScreen', 'FP', 'NCI-60', 'P. berghei 4-day suppressive test'). null when none is "
32
+ "stated.\n"
33
+ "- target: the protein the value was measured against: an enzyme, receptor, channel or "
34
+ "other molecular target, as written ('InhA', 'DprE1', 'hERG', 'KasA', 'PDE4B', "
35
+ "'BACE1'). null for whole-cell, organism and cytotoxicity values unless one is named.\n"
36
+ "- strain: the organism the value was measured against, as written: species, strain, "
37
+ "isolate or virus (H37Rv, M. tuberculosis, E. coli ATCC 25922, A. baumannii ATCC 19606, "
38
+ "Pf3D7, Dd2, S. mansoni, HIV-1, norovirus). An organism always goes here, never in assay or target, even "
39
+ "when it heads a table column or is all that names the assay. null when not stated.\n"
40
+ "- combination: a second compound dosed together with the one measured (a combination "
41
+ "or potentiation partner), as written ('avibactam', 'polymyxin B 0.5 µg/mL'). Not a "
42
+ "reference compound the value is relative to ('% of isoproterenol') or an agonist, substrate "
43
+ "or tracer the assay uses. null for a single agent.\n"
23
44
  )
24
45
  _BIOACTIVITY_TABLES = (
25
46
  "In a data table, extract every cell value as its own bioactivity. An ID or registry "
26
47
  "column, when present, supplies the compound_name and the row label is only its synonym; "
27
- "otherwise the row label is the compound_name. A column header naming an assay or cell "
28
- "line is the assay, and the endpoint and unit often come from the caption or a footnote "
29
- "('CC50 in µM')."
48
+ "otherwise the row label is the compound_name. A column header fills the attribute for "
49
+ "what it names (an assay or cell line, a protein target, an organism or strain), and the "
50
+ "endpoint and unit often come from the caption or a footnote ('CC50 in µM')."
30
51
  )
31
52
 
32
53
  CHEMISTRY_PROMPT = (
@@ -49,6 +70,7 @@ BIOACTIVITY_PROMPT = (
49
70
  "Extract bioactivity measurements and assay data from the text: potency, selectivity, "
50
71
  "cytotoxicity and efficacy values.\n"
51
72
  + _BIOACTIVITY_ATTRIBUTES
73
+ + _ASSAY_ATTRIBUTE
52
74
  + _BIOACTIVITY_TABLES
53
75
  + "\nAlso extract assay descriptions: cell lines used (e.g. 'HeLa', 'A549'), assay formats "
54
76
  "(e.g. 'cell viability', 'binding assay', 'enzymatic assay'), and organisms."
@@ -75,6 +97,7 @@ FULL_PROMPT = (
75
97
  "- Diseases and indications: cancer types, disease names, therapeutic areas.\n"
76
98
  "- Mechanisms of action: binding modes, inhibition types, selectivity descriptions.\n"
77
99
  + _BIOACTIVITY_ATTRIBUTES
100
+ + _ASSAY_ATTRIBUTE
78
101
  + _BIOACTIVITY_TABLES
79
102
  + "\nExtract only what is explicitly stated; do not infer or generate values."
80
103
  )
@@ -84,8 +107,9 @@ FULL_PROMPT = (
84
107
  TB_PROMPT = (
85
108
  "Extract drug discovery entities from this tuberculosis research text.\n\n"
86
109
  + _BIOACTIVITY_ATTRIBUTES
87
- + "- strain: the organism or strain tested (H37Rv, Erdman, Pf3D7, Dd2, K1), "
88
- "null when not stated.\n" + _BIOACTIVITY_TABLES + "\n\n"
110
+ + _TB_SLOT_ATTRIBUTES
111
+ + _BIOACTIVITY_TABLES
112
+ + "\n\n"
89
113
  "DISAMBIGUATION RULES:\n"
90
114
  "- Mycobacterial proteins (e.g. ClpC1, DprE1, InhA, AtpE, MmpL3, QcrB) "
91
115
  "are biological targets, NOT compounds.\n"
@@ -102,6 +126,9 @@ TB_PROMPT = (
102
126
  "- Enzyme descriptions like 'enoyl-ACP reductase' are product, not target.\n"
103
127
  "- 'cell wall', 'lipid metabolism' are functional_category, not mechanism_of_action.\n"
104
128
  "- 'fragment screening', 'biochemical assay' are screening_method, not assay.\n"
129
+ "- A disease or infection is a disease entity wherever it is named, in prose, a heading "
130
+ "or a table cell (tuberculosis, HIV infection, malaria, schistosomiasis, glioblastoma in 'U87 (glioblastoma)'), "
131
+ "even when its organism also fills a bioactivity's strain.\n"
105
132
  "- Use target for proteins in a drug-targeting context, gene_name for loci, "
106
133
  "protein_name for non-drug-target proteins.\n\n"
107
134
  "Extract only what is explicitly stated; do not infer or generate values."
@@ -125,3 +152,13 @@ TB_BIOLOGY_PROMPT = (
125
152
  "are functional_category. "
126
153
  "Extract only what is explicitly stated; do not infer or generate values."
127
154
  )
155
+
156
+ # Framing for NERExtractor.extract(context=...), rendered by langextract ahead of
157
+ # the examples.
158
+ DOCUMENT_CONTEXT = (
159
+ "Document context, from elsewhere in the same document. It is not the text to extract "
160
+ "from: take no entity and no value from it, and do not pair a value in the text with a "
161
+ "number or claim made here. Use it only to fill a bioactivity attribute the text leaves "
162
+ "unstated, such as the target or organism of a table, when the context names one for "
163
+ "the whole document.\n"
164
+ )
@@ -3,15 +3,43 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import logging
6
+ import re
6
7
 
7
8
  import langextract as lx
8
9
 
10
+ from structflo.ner import _prompts
9
11
  from structflo.ner._entities import NERResult
10
12
  from structflo.ner._mapping import annotated_doc_to_result
11
13
  from structflo.ner.profiles import FULL, EntityProfile
12
14
 
13
15
  logger = logging.getLogger(__name__)
14
16
 
17
+ # A table cell holding only a value: '0.94', '>32', '2.4*', '0.62µM'.
18
+ _VALUE_CELL = re.compile(r"^[<>≤≥~=]*\s*\d[\d,]*\.?\d*\s*\*?\s*(%|[nµu]M|µg/mL|mg/kg)?$")
19
+
20
+
21
+ def _bioactivity_count(doc: lx.data.AnnotatedDocument) -> int:
22
+ return sum(1 for ext in doc.extractions if ext.extraction_class == "bioactivity")
23
+
24
+
25
+ def _table_values_dropped(text: str, doc: lx.data.AnnotatedDocument) -> bool:
26
+ """True when fewer than half of a page's table values came back as bioactivities.
27
+
28
+ Models now and then return a complete, valid answer that lists a table's
29
+ compounds but leaves out its values. On the ChEMBL-gold decks this fires on
30
+ ~2% of pages and catches ~60% of such dropouts.
31
+ """
32
+ # ponytail: markdown tables only; values in prose and qualitative results
33
+ # are not checked, so those dropouts still pass silently.
34
+ cells = sum(
35
+ 1
36
+ for line in text.splitlines()
37
+ if line.lstrip().startswith("|")
38
+ for cell in line.strip().strip("|").split("|")
39
+ if _VALUE_CELL.match(cell.strip())
40
+ )
41
+ return cells >= 5 and _bioactivity_count(doc) < 0.5 * cells
42
+
15
43
 
16
44
  class NERExtractor:
17
45
  """Extract drug discovery entities from text with zero configuration.
@@ -90,12 +118,17 @@ class NERExtractor:
90
118
  self,
91
119
  text: str | list[str],
92
120
  profile: EntityProfile | None = None,
121
+ context: str | None = None,
93
122
  ) -> NERResult | list[NERResult]:
94
123
  """Extract drug discovery entities from text.
95
124
 
96
125
  Args:
97
126
  text: Input text (or list of texts) to process.
98
127
  profile: Override the default profile for this call only.
128
+ context: Text from elsewhere in the same document (its title or
129
+ summary). The model reads it to fill attributes the text leaves
130
+ unstated, such as the target of a table whose target is named
131
+ only on another page; no entities are extracted from it.
99
132
 
100
133
  Returns:
101
134
  A :class:`NERResult` for a single string input, or a list of
@@ -107,7 +140,13 @@ class NERExtractor:
107
140
 
108
141
  results = []
109
142
  for single_text in texts:
110
- doc = self._run_extraction(single_text, active_profile)
143
+ doc = self._run_extraction(single_text, active_profile, context)
144
+ if "bioactivity" in active_profile.entity_classes and _table_values_dropped(
145
+ single_text, doc
146
+ ):
147
+ logger.warning("Table values missing from extraction; retrying once.")
148
+ retry = self._run_extraction(single_text, active_profile, context)
149
+ doc = max(doc, retry, key=_bioactivity_count)
111
150
  results.append(annotated_doc_to_result(doc, single_text))
112
151
 
113
152
  return results if is_batch else results[0]
@@ -198,6 +237,7 @@ class NERExtractor:
198
237
  self,
199
238
  text: str,
200
239
  profile: EntityProfile,
240
+ context: str | None = None,
201
241
  ) -> lx.data.AnnotatedDocument:
202
242
  """Call lx.extract and return the AnnotatedDocument."""
203
243
  examples = self._build_examples(profile)
@@ -206,6 +246,8 @@ class NERExtractor:
206
246
  kwargs: dict = dict(self._langextract_kwargs)
207
247
  kwargs.setdefault("use_schema_constraints", True)
208
248
  kwargs.setdefault("show_progress", False)
249
+ if context:
250
+ kwargs["additional_context"] = _prompts.DOCUMENT_CONTEXT + context
209
251
 
210
252
  if self._provider is not None:
211
253
  # Explicit provider → deterministic routing via ModelConfig.
@@ -236,4 +278,29 @@ class NERExtractor:
236
278
 
237
279
  # Post-process: drop extractions with hallucinated entity classes
238
280
  allowed = set(profile.entity_classes)
239
- return self._filter_extractions(doc, allowed)
281
+ doc = self._filter_extractions(doc, allowed)
282
+ if context:
283
+ doc = self._drop_context_copies(doc, context)
284
+ return doc
285
+
286
+ @staticmethod
287
+ def _drop_context_copies(
288
+ doc: lx.data.AnnotatedDocument,
289
+ context: str,
290
+ ) -> lx.data.AnnotatedDocument:
291
+ """Drop extractions the model took from the context instead of the text.
292
+
293
+ The prompt forbids it, but models still extract from the context now and
294
+ then. Such an extraction does not align to the text, and its text occurs
295
+ in the context but not in the text (an unaligned one that is in the text
296
+ is a missed alignment, not a copy).
297
+ """
298
+ ctx, text = context.casefold(), doc.text.casefold()
299
+ kept = [
300
+ ext
301
+ for ext in doc.extractions
302
+ if ext.char_interval is not None
303
+ or ext.extraction_text.casefold() not in ctx
304
+ or ext.extraction_text.casefold() in text
305
+ ]
306
+ return lx.data.AnnotatedDocument(text=doc.text, extractions=kept)
@@ -122,6 +122,50 @@ class TestNERExtractorExtract:
122
122
  result = extractor.extract("My source text")
123
123
  assert result.source_text == "My source text"
124
124
 
125
+ _TABLE = "| cpd | MIC | CC50 |\n| --- | --- | --- |\n" + "".join(
126
+ f"| C{i} | {i}.5 | >50 |\n" for i in range(5)
127
+ )
128
+
129
+ def _bio_doc(self, n: int) -> lx.data.AnnotatedDocument:
130
+ return _make_annotated_doc(
131
+ [
132
+ lx.data.Extraction(extraction_class="bioactivity", extraction_text=f"{i}.5")
133
+ for i in range(n)
134
+ ]
135
+ )
136
+
137
+ def test_retries_once_when_table_values_are_missing(self):
138
+ extractor = NERExtractor(profile=TB)
139
+ extractor._run_extraction = MagicMock(side_effect=[self._bio_doc(0), self._bio_doc(10)])
140
+ result = extractor.extract(self._TABLE)
141
+ assert extractor._run_extraction.call_count == 2
142
+ assert len(result.bioactivities) == 10
143
+
144
+ def test_no_retry_when_table_values_came_back(self):
145
+ extractor = NERExtractor(profile=TB)
146
+ extractor._run_extraction = MagicMock(return_value=self._bio_doc(10))
147
+ extractor.extract(self._TABLE)
148
+ assert extractor._run_extraction.call_count == 1
149
+
150
+ def test_no_retry_for_profiles_without_bioactivity(self):
151
+ extractor = NERExtractor(profile=CHEMISTRY)
152
+ extractor._run_extraction = MagicMock(return_value=self._bio_doc(0))
153
+ extractor.extract(self._TABLE)
154
+ assert extractor._run_extraction.call_count == 1
155
+
156
+ def test_null_attributes_are_absent_not_none_strings(self):
157
+ # OpenAI strict mode returns every schema key, null when unstated.
158
+ extractor = self._extractor_with_mock(
159
+ [
160
+ lx.data.Extraction(
161
+ extraction_class="bioactivity",
162
+ extraction_text="IC50 = 3 nM",
163
+ attributes={"value": "3", "strain": None, "assay": None},
164
+ )
165
+ ]
166
+ )
167
+ assert extractor.extract("text").bioactivities[0].attributes == {"value": "3"}
168
+
125
169
 
126
170
  class TestProviderRouting:
127
171
  """Explicit provider selection routes deterministically via ModelConfig,
@@ -181,6 +225,24 @@ class TestProviderRouting:
181
225
  assert pk["base_url"] == "http://ollama:11434"
182
226
  assert pk["num_ctx"] == 8192
183
227
 
228
+ def test_context_reaches_langextract_framed(self, monkeypatch):
229
+ from structflo.ner import extractor as extractor_mod
230
+
231
+ captured = {}
232
+
233
+ def fake_extract(**kwargs):
234
+ captured.update(kwargs)
235
+ return _make_annotated_doc([])
236
+
237
+ monkeypatch.setattr(extractor_mod.lx, "extract", fake_extract)
238
+ NERExtractor().extract("IC50s table", context="Deck on InhA inhibitors.")
239
+ ctx = captured["additional_context"]
240
+ assert ctx.startswith("Document context") and ctx.endswith("Deck on InhA inhibitors.")
241
+
242
+ def test_no_context_passes_no_additional_context(self, monkeypatch):
243
+ captured = self._run_and_capture(monkeypatch)
244
+ assert "additional_context" not in captured
245
+
184
246
  def test_cloud_provider_kwargs_carry_api_key(self, monkeypatch):
185
247
  captured = self._run_and_capture(
186
248
  monkeypatch, provider="openai", model_id="gpt-4o", api_key="sk-test"
@@ -235,6 +297,27 @@ class TestFilterExtractions:
235
297
  assert len(filtered.extractions) == 0
236
298
 
237
299
 
300
+ class TestDropContextCopies:
301
+ def test_drops_only_unaligned_extractions_found_in_context(self):
302
+ aligned = lx.data.Extraction(
303
+ extraction_class="target",
304
+ extraction_text="InhA",
305
+ char_interval=lx.data.CharInterval(start_pos=0, end_pos=4),
306
+ )
307
+ copied = lx.data.Extraction(extraction_class="compound_name", extraction_text="Isoniazid")
308
+ unaligned = lx.data.Extraction(extraction_class="compound_name", extraction_text="7a")
309
+ # In the context too, but also on the page: a missed alignment, not a copy.
310
+ on_page = lx.data.Extraction(extraction_class="disease", extraction_text="malaria")
311
+ doc = lx.data.AnnotatedDocument(
312
+ text="InhA IC50s for 7a; malaria panel",
313
+ extractions=[aligned, copied, unaligned, on_page],
314
+ )
315
+ kept = NERExtractor._drop_context_copies(
316
+ doc, "InhA inhibitors that match isoniazid, and a malaria screen"
317
+ )
318
+ assert kept.extractions == [aligned, unaligned, on_page]
319
+
320
+
238
321
  class TestBuildExamples:
239
322
  def test_extra_examples_appended(self):
240
323
  extra = lx.data.ExampleData(text="extra", extractions=[])
@@ -298,8 +381,20 @@ class TestTBProfile:
298
381
  }
299
382
  assert {"value", "unit", "assay_type", "compound_name", "assay"} <= keys, profile.name
300
383
 
384
+ def test_tb_bioactivity_schema_has_context_slots(self):
385
+ """TB gives what a value was measured against (target, strain) and with
386
+ (combination) their own slots, apart from assay."""
387
+ keys = {
388
+ k
389
+ for ex in TB.examples
390
+ for e in ex.extractions
391
+ if e.extraction_class == "bioactivity"
392
+ for k in (e.attributes or {})
393
+ }
394
+ assert {"target", "strain", "combination"} <= keys
395
+
301
396
  def test_tb_examples_count(self):
302
- assert len(TB.examples) == 6
397
+ assert len(TB.examples) == 7
303
398
  assert len(TB_CHEMISTRY.examples) == 2
304
399
  assert len(TB_BIOLOGY.examples) == 2
305
400
 
@@ -2170,7 +2170,7 @@ dev = [
2170
2170
 
2171
2171
  [package.metadata]
2172
2172
  requires-dist = [
2173
- { name = "langextract", specifier = ">=1.6.0" },
2173
+ { name = "langextract", specifier = ">=1.7.0" },
2174
2174
  { name = "pandas", marker = "extra == 'dataframe'", specifier = ">=1.5" },
2175
2175
  { name = "pyyaml", specifier = ">=6.0" },
2176
2176
  { name = "rapidfuzz", specifier = ">=3.0" },
File without changes
File without changes