structflo-ner 0.3.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/.gitignore +4 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/PKG-INFO +3 -3
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/notebooks/01_quickstart.ipynb +253 -22
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/pyproject.toml +1 -1
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/__init__.py +1 -1
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/_entities.py +5 -1
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/_examples.py +57 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/_prompts.py +3 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/extractor.py +57 -17
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/README.md +34 -12
- structflo_ner-0.5.0/structflo/ner/fast/_loader.py +191 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/_matcher.py +18 -8
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/extractor.py +6 -8
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/tests/test_extractor.py +69 -1
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/tests/test_fast.py +94 -17
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/uv.lock +17 -17
- structflo_ner-0.3.0/structflo/ner/fast/_loader.py +0 -116
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/.github/workflows/ci.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/.github/workflows/publish.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/Makefile +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/README.md +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/coverage.xml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/images/fast-viz.png +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/images/local-gen-pandas.png +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/images/local-gen-viz.png +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/images/local-tb-viz.png +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/images/ner_visualization.gif +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/images/struct-flo-ner.png +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/notebooks/02_fast_ner.ipynb +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/_display.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/_mapping.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/__init__.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/_normalize.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/gazetteers/accession_number.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/gazetteers/compound_name.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/gazetteers/disease.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/gazetteers/functional_category.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/gazetteers/gene_name.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/gazetteers/product.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/gazetteers/screening_method.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/gazetteers/strain.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/fast/gazetteers/target.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/structflo/ner/profiles.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/tests/__init__.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.5.0}/tests/test_entities.py +0 -0
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: structflo-ner
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: Drug discovery NER wrapper around LangExtract — zero-config entity extraction for chemistry and biology.
|
|
5
5
|
License: Apache-2.0
|
|
6
6
|
Requires-Python: >=3.10
|
|
7
|
-
Requires-Dist: langextract>=1.
|
|
7
|
+
Requires-Dist: langextract>=1.6.0
|
|
8
8
|
Requires-Dist: pyyaml>=6.0
|
|
9
9
|
Requires-Dist: rapidfuzz>=3.0
|
|
10
10
|
Provides-Extra: dataframe
|
|
@@ -28,7 +28,7 @@
|
|
|
28
28
|
},
|
|
29
29
|
{
|
|
30
30
|
"cell_type": "code",
|
|
31
|
-
"execution_count":
|
|
31
|
+
"execution_count": 1,
|
|
32
32
|
"metadata": {},
|
|
33
33
|
"outputs": [],
|
|
34
34
|
"source": [
|
|
@@ -37,7 +37,7 @@
|
|
|
37
37
|
},
|
|
38
38
|
{
|
|
39
39
|
"cell_type": "code",
|
|
40
|
-
"execution_count":
|
|
40
|
+
"execution_count": 2,
|
|
41
41
|
"metadata": {},
|
|
42
42
|
"outputs": [],
|
|
43
43
|
"source": [
|
|
@@ -89,7 +89,7 @@
|
|
|
89
89
|
},
|
|
90
90
|
{
|
|
91
91
|
"cell_type": "code",
|
|
92
|
-
"execution_count":
|
|
92
|
+
"execution_count": 3,
|
|
93
93
|
"metadata": {},
|
|
94
94
|
"outputs": [],
|
|
95
95
|
"source": [
|
|
@@ -101,19 +101,19 @@
|
|
|
101
101
|
},
|
|
102
102
|
{
|
|
103
103
|
"cell_type": "code",
|
|
104
|
-
"execution_count":
|
|
104
|
+
"execution_count": 4,
|
|
105
105
|
"metadata": {},
|
|
106
106
|
"outputs": [
|
|
107
107
|
{
|
|
108
108
|
"data": {
|
|
109
109
|
"text/html": [
|
|
110
|
-
"<div id=\"ner-
|
|
110
|
+
"<div id=\"ner-fdc30d97\" style=\"font-family:system-ui,-apple-system,sans-serif;max-width:900px;padding:20px\"><div style=\"display:flex;flex-wrap:wrap;gap:6px;margin-bottom:12px\"><button class=\"ner-legend-btn\" data-target=\"ChemicalEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #3b82f6;border-radius:16px;background:#dbeafe;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#3b82f6\"></span>Compound</button><button class=\"ner-legend-btn\" data-target=\"TargetEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #22c55e;border-radius:16px;background:#dcfce7;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#22c55e\"></span>Target</button><button class=\"ner-legend-btn\" data-target=\"DiseaseEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #ec4899;border-radius:16px;background:#fce7f3;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#ec4899\"></span>Disease</button><button class=\"ner-legend-btn\" data-target=\"BioactivityEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #f59e0b;border-radius:16px;background:#fef3c7;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#f59e0b\"></span>Bioactivity</button></div><div style=\"font-size:0.8em;color:#64748b;margin-bottom:10px\"><b>3</b> Compound · <b>1</b> Target · <b>1</b> Disease · <b>1</b> Bioactivity</div><div style=\"line-height:2;font-size:1em;white-space:pre-wrap\"><mark class=\"ner-ent\" data-category=\"ChemicalEntity\" style=\"background:#dbeafe;border-bottom:2px solid #3b82f6;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: compound_name | synonyms: ZD1839\">Gefitinib<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#3b82f6\">compound_name</span></mark> (<mark class=\"ner-ent\" data-category=\"ChemicalEntity\" style=\"background:#dbeafe;border-bottom:2px solid #3b82f6;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: compound_name\">ZD1839<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#3b82f6\">compound_name</span></mark>) is a first-generation <mark class=\"ner-ent\" data-category=\"TargetEntity\" style=\"background:#dcfce7;border-bottom:2px solid #22c55e;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: target\">EGFR<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#22c55e\">target</span></mark> inhibitor with <mark class=\"ner-ent\" data-category=\"BioactivityEntity\" style=\"background:#fef3c7;border-bottom:2px solid #f59e0b;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: bioactivity | value: 0.033 | unit: µM | assay_type: IC50 | compound_name: Gefitinib\">IC50 = 0.033 µM<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#f59e0b\">bioactivity</span></mark> approved for <mark class=\"ner-ent\" data-category=\"DiseaseEntity\" style=\"background:#fce7f3;border-bottom:2px solid #ec4899;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: disease | therapeutic_area: oncology\">NSCLC<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#ec4899\">disease</span></mark>.Its SMILES is <mark class=\"ner-ent\" data-category=\"ChemicalEntity\" style=\"background:#dbeafe;border-bottom:2px solid #3b82f6;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: smiles\">COc1cc2ncnc(Nc3ccc(F)c(Cl)c3)c2cc1OCCCN1CCOCC1<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#3b82f6\">smiles</span></mark>.</div><script>(function(){var w=document.getElementById('ner-fdc30d97');w.querySelectorAll('.ner-legend-btn').forEach(function(btn){btn.addEventListener('click',function(){var cat=btn.dataset.target;var active=btn.dataset.active!=='false';btn.dataset.active=active?'false':'true';btn.style.opacity=active?'0.35':'1';w.querySelectorAll('[data-category=\"'+cat+'\"]').forEach(function(el){if(el.classList.contains('ner-ent'))el.style.background=active?'transparent':el.style.borderBottomColor.replace('solid ','');el.style.opacity=active?'0.3':'1';});});});})()</script></div>"
|
|
111
111
|
],
|
|
112
112
|
"text/plain": [
|
|
113
|
-
"NERResult(source_text='Gefitinib (ZD1839) is a first-generation EGFR inhibitor with IC50 = 0.033 µM approved for NSCLC.Its SMILES is COc1cc2ncnc(Nc3ccc(F)c(Cl)c3)c2cc1OCCCN1CCOCC1.', compounds=[ChemicalEntity(text='Gefitinib', entity_type='compound_name', char_start=0, char_end=9, attributes={'synonyms': 'ZD1839'}, alignment='match_exact'), ChemicalEntity(text='ZD1839', entity_type='compound_name', char_start=11, char_end=17, attributes={}, alignment='match_exact'), ChemicalEntity(text='COc1cc2ncnc(Nc3ccc(F)c(Cl)c3)c2cc1OCCCN1CCOCC1', entity_type='smiles', char_start=110, char_end=156, attributes={}, alignment='match_exact')], targets=[TargetEntity(text='EGFR', entity_type='target', char_start=41, char_end=45, attributes={}, alignment='match_exact')], diseases=[DiseaseEntity(text='NSCLC', entity_type='disease', char_start=90, char_end=95, attributes={'therapeutic_area': 'oncology'}, alignment='match_exact')], bioactivities=[BioactivityEntity(text='IC50 = 0.033 µM', entity_type='bioactivity', char_start=61, char_end=76, attributes={'value': '0.033', 'unit': 'µM', 'assay_type': 'IC50'}, alignment='match_exact')], assays=[], mechanisms=[], accessions=[], products=[], functional_categories=[], screening_methods=[], unclassified=[])"
|
|
113
|
+
"NERResult(source_text='Gefitinib (ZD1839) is a first-generation EGFR inhibitor with IC50 = 0.033 µM approved for NSCLC.Its SMILES is COc1cc2ncnc(Nc3ccc(F)c(Cl)c3)c2cc1OCCCN1CCOCC1.', compounds=[ChemicalEntity(text='Gefitinib', entity_type='compound_name', char_start=0, char_end=9, attributes={'synonyms': 'ZD1839'}, alignment='match_exact'), ChemicalEntity(text='ZD1839', entity_type='compound_name', char_start=11, char_end=17, attributes={}, alignment='match_exact'), ChemicalEntity(text='COc1cc2ncnc(Nc3ccc(F)c(Cl)c3)c2cc1OCCCN1CCOCC1', entity_type='smiles', char_start=110, char_end=156, attributes={}, alignment='match_exact')], targets=[TargetEntity(text='EGFR', entity_type='target', char_start=41, char_end=45, attributes={}, alignment='match_exact')], diseases=[DiseaseEntity(text='NSCLC', entity_type='disease', char_start=90, char_end=95, attributes={'therapeutic_area': 'oncology'}, alignment='match_exact')], bioactivities=[BioactivityEntity(text='IC50 = 0.033 µM', entity_type='bioactivity', char_start=61, char_end=76, attributes={'value': '0.033', 'unit': 'µM', 'assay_type': 'IC50', 'compound_name': 'Gefitinib'}, alignment='match_exact')], assays=[], mechanisms=[], accessions=[], products=[], functional_categories=[], screening_methods=[], strains=[], unclassified=[])"
|
|
114
114
|
]
|
|
115
115
|
},
|
|
116
|
-
"execution_count":
|
|
116
|
+
"execution_count": 4,
|
|
117
117
|
"metadata": {},
|
|
118
118
|
"output_type": "execute_result"
|
|
119
119
|
}
|
|
@@ -127,7 +127,7 @@
|
|
|
127
127
|
},
|
|
128
128
|
{
|
|
129
129
|
"cell_type": "code",
|
|
130
|
-
"execution_count":
|
|
130
|
+
"execution_count": 5,
|
|
131
131
|
"metadata": {},
|
|
132
132
|
"outputs": [
|
|
133
133
|
{
|
|
@@ -162,6 +162,7 @@
|
|
|
162
162
|
" <th>value</th>\n",
|
|
163
163
|
" <th>unit</th>\n",
|
|
164
164
|
" <th>assay_type</th>\n",
|
|
165
|
+
" <th>compound_name</th>\n",
|
|
165
166
|
" </tr>\n",
|
|
166
167
|
" </thead>\n",
|
|
167
168
|
" <tbody>\n",
|
|
@@ -178,6 +179,7 @@
|
|
|
178
179
|
" <td>NaN</td>\n",
|
|
179
180
|
" <td>NaN</td>\n",
|
|
180
181
|
" <td>NaN</td>\n",
|
|
182
|
+
" <td>NaN</td>\n",
|
|
181
183
|
" </tr>\n",
|
|
182
184
|
" <tr>\n",
|
|
183
185
|
" <th>1</th>\n",
|
|
@@ -192,6 +194,7 @@
|
|
|
192
194
|
" <td>NaN</td>\n",
|
|
193
195
|
" <td>NaN</td>\n",
|
|
194
196
|
" <td>NaN</td>\n",
|
|
197
|
+
" <td>NaN</td>\n",
|
|
195
198
|
" </tr>\n",
|
|
196
199
|
" <tr>\n",
|
|
197
200
|
" <th>2</th>\n",
|
|
@@ -206,6 +209,7 @@
|
|
|
206
209
|
" <td>NaN</td>\n",
|
|
207
210
|
" <td>NaN</td>\n",
|
|
208
211
|
" <td>NaN</td>\n",
|
|
212
|
+
" <td>NaN</td>\n",
|
|
209
213
|
" </tr>\n",
|
|
210
214
|
" <tr>\n",
|
|
211
215
|
" <th>3</th>\n",
|
|
@@ -220,6 +224,7 @@
|
|
|
220
224
|
" <td>NaN</td>\n",
|
|
221
225
|
" <td>NaN</td>\n",
|
|
222
226
|
" <td>NaN</td>\n",
|
|
227
|
+
" <td>NaN</td>\n",
|
|
223
228
|
" </tr>\n",
|
|
224
229
|
" <tr>\n",
|
|
225
230
|
" <th>4</th>\n",
|
|
@@ -234,6 +239,7 @@
|
|
|
234
239
|
" <td>NaN</td>\n",
|
|
235
240
|
" <td>NaN</td>\n",
|
|
236
241
|
" <td>NaN</td>\n",
|
|
242
|
+
" <td>NaN</td>\n",
|
|
237
243
|
" </tr>\n",
|
|
238
244
|
" <tr>\n",
|
|
239
245
|
" <th>5</th>\n",
|
|
@@ -248,6 +254,7 @@
|
|
|
248
254
|
" <td>0.033</td>\n",
|
|
249
255
|
" <td>µM</td>\n",
|
|
250
256
|
" <td>IC50</td>\n",
|
|
257
|
+
" <td>Gefitinib</td>\n",
|
|
251
258
|
" </tr>\n",
|
|
252
259
|
" </tbody>\n",
|
|
253
260
|
"</table>\n",
|
|
@@ -270,16 +277,16 @@
|
|
|
270
277
|
"4 DiseaseEntity 90 95 match_exact NaN \n",
|
|
271
278
|
"5 BioactivityEntity 61 76 match_exact NaN \n",
|
|
272
279
|
"\n",
|
|
273
|
-
" therapeutic_area value unit assay_type \n",
|
|
274
|
-
"0 NaN NaN NaN NaN \n",
|
|
275
|
-
"1 NaN NaN NaN NaN \n",
|
|
276
|
-
"2 NaN NaN NaN NaN \n",
|
|
277
|
-
"3 NaN NaN NaN NaN \n",
|
|
278
|
-
"4 oncology NaN NaN NaN \n",
|
|
279
|
-
"5 NaN 0.033 µM IC50 "
|
|
280
|
+
" therapeutic_area value unit assay_type compound_name \n",
|
|
281
|
+
"0 NaN NaN NaN NaN NaN \n",
|
|
282
|
+
"1 NaN NaN NaN NaN NaN \n",
|
|
283
|
+
"2 NaN NaN NaN NaN NaN \n",
|
|
284
|
+
"3 NaN NaN NaN NaN NaN \n",
|
|
285
|
+
"4 oncology NaN NaN NaN NaN \n",
|
|
286
|
+
"5 NaN 0.033 µM IC50 Gefitinib "
|
|
280
287
|
]
|
|
281
288
|
},
|
|
282
|
-
"execution_count":
|
|
289
|
+
"execution_count": 5,
|
|
283
290
|
"metadata": {},
|
|
284
291
|
"output_type": "execute_result"
|
|
285
292
|
}
|
|
@@ -311,19 +318,19 @@
|
|
|
311
318
|
},
|
|
312
319
|
{
|
|
313
320
|
"cell_type": "code",
|
|
314
|
-
"execution_count":
|
|
321
|
+
"execution_count": 6,
|
|
315
322
|
"metadata": {},
|
|
316
323
|
"outputs": [
|
|
317
324
|
{
|
|
318
325
|
"data": {
|
|
319
326
|
"text/html": [
|
|
320
|
-
"<div id=\"ner-
|
|
327
|
+
"<div id=\"ner-410d0552\" style=\"font-family:system-ui,-apple-system,sans-serif;max-width:900px;padding:20px\"><div style=\"display:flex;flex-wrap:wrap;gap:6px;margin-bottom:12px\"><button class=\"ner-legend-btn\" data-target=\"ChemicalEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #3b82f6;border-radius:16px;background:#dbeafe;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#3b82f6\"></span>Compound</button><button class=\"ner-legend-btn\" data-target=\"TargetEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #22c55e;border-radius:16px;background:#dcfce7;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#22c55e\"></span>Target</button><button class=\"ner-legend-btn\" data-target=\"DiseaseEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #ec4899;border-radius:16px;background:#fce7f3;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#ec4899\"></span>Disease</button><button class=\"ner-legend-btn\" data-target=\"BioactivityEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #f59e0b;border-radius:16px;background:#fef3c7;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#f59e0b\"></span>Bioactivity</button><button class=\"ner-legend-btn\" data-target=\"AssayEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #6366f1;border-radius:16px;background:#e0e7ff;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#6366f1\"></span>Assay</button></div><div style=\"font-size:0.8em;color:#64748b;margin-bottom:10px\"><b>2</b> Compound · <b>2</b> Target · <b>2</b> Disease · <b>1</b> Bioactivity · <b>1</b> Assay</div><div style=\"line-height:2;font-size:1em;white-space:pre-wrap\"><mark class=\"ner-ent\" data-category=\"ChemicalEntity\" style=\"background:#dbeafe;border-bottom:2px solid #3b82f6;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: compound_name | synonyms: TMC207\">Bedaquiline<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#3b82f6\">compound_name</span></mark> (<mark class=\"ner-ent\" data-category=\"ChemicalEntity\" style=\"background:#dbeafe;border-bottom:2px solid #3b82f6;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: compound_name\">TMC207<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#3b82f6\">compound_name</span></mark>) is a diarylquinoline that inhibits the <mark class=\"ner-ent\" data-category=\"TargetEntity\" style=\"background:#dcfce7;border-bottom:2px solid #22c55e;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: target | gene_name: atpE (Rv1305) | protein_family: ATP synthase\">mycobacterial ATP synthase subunit c<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#22c55e\">target</span></mark> encoded by atpE (Rv1305) with <mark class=\"ner-ent\" data-category=\"BioactivityEntity\" style=\"background:#fef3c7;border-bottom:2px solid #f59e0b;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: bioactivity | value: 0.033 | unit: µM | assay_type: IC50 | compound_name: Bedaquiline\">IC50 = 0.033<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#f59e0b\">bioactivity</span></mark> µMIt shows potent activity against Mycobacterium tuberculosis including <mark class=\"ner-ent\" data-category=\"DiseaseEntity\" style=\"background:#fce7f3;border-bottom:2px solid #ec4899;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: disease | therapeutic_area: infectious disease\">MDR-TB<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#ec4899\">disease</span></mark> and <mark class=\"ner-ent\" data-category=\"DiseaseEntity\" style=\"background:#fce7f3;border-bottom:2px solid #ec4899;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: disease | therapeutic_area: infectious disease\">XDR-TB<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#ec4899\">disease</span></mark>. This compound was identified through <mark class=\"ner-ent\" data-category=\"AssayEntity\" style=\"background:#e0e7ff;border-bottom:2px solid #6366f1;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: assay\">whole-cell screening<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#6366f1\">assay</span></mark> and targets the <mark class=\"ner-ent\" data-category=\"TargetEntity\" style=\"background:#dcfce7;border-bottom:2px solid #22c55e;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: target | protein_family: pathway\">energy metabolism pathway<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#22c55e\">target</span></mark>.</div><script>(function(){var w=document.getElementById('ner-410d0552');w.querySelectorAll('.ner-legend-btn').forEach(function(btn){btn.addEventListener('click',function(){var cat=btn.dataset.target;var active=btn.dataset.active!=='false';btn.dataset.active=active?'false':'true';btn.style.opacity=active?'0.35':'1';w.querySelectorAll('[data-category=\"'+cat+'\"]').forEach(function(el){if(el.classList.contains('ner-ent'))el.style.background=active?'transparent':el.style.borderBottomColor.replace('solid ','');el.style.opacity=active?'0.3':'1';});});});})()</script></div>"
|
|
321
328
|
],
|
|
322
329
|
"text/plain": [
|
|
323
|
-
"NERResult(source_text='Bedaquiline (TMC207) is a diarylquinoline that inhibits the mycobacterial ATP synthase subunit c encoded by atpE (Rv1305).
|
|
330
|
+
"NERResult(source_text='Bedaquiline (TMC207) is a diarylquinoline that inhibits the mycobacterial ATP synthase subunit c encoded by atpE (Rv1305) with IC50 = 0.033 µMIt shows potent activity against Mycobacterium tuberculosis including MDR-TB and XDR-TB. This compound was identified through whole-cell screening and targets the energy metabolism pathway.', compounds=[ChemicalEntity(text='Bedaquiline', entity_type='compound_name', char_start=0, char_end=11, attributes={'synonyms': 'TMC207'}, alignment='match_exact'), ChemicalEntity(text='TMC207', entity_type='compound_name', char_start=13, char_end=19, attributes={}, alignment='match_exact')], targets=[TargetEntity(text='mycobacterial ATP synthase subunit c', entity_type='target', char_start=60, char_end=96, attributes={'gene_name': 'atpE (Rv1305)', 'protein_family': 'ATP synthase'}, alignment='match_exact'), TargetEntity(text='energy metabolism pathway', entity_type='target', char_start=305, char_end=330, attributes={'protein_family': 'pathway'}, alignment='match_exact')], diseases=[DiseaseEntity(text='MDR-TB', entity_type='disease', char_start=212, char_end=218, attributes={'therapeutic_area': 'infectious disease'}, alignment='match_exact'), DiseaseEntity(text='XDR-TB', entity_type='disease', char_start=223, char_end=229, attributes={'therapeutic_area': 'infectious disease'}, alignment='match_exact')], bioactivities=[BioactivityEntity(text='IC50 = 0.033 µM', entity_type='bioactivity', char_start=127, char_end=139, attributes={'value': '0.033', 'unit': 'µM', 'assay_type': 'IC50', 'compound_name': 'Bedaquiline'}, alignment='match_lesser')], assays=[AssayEntity(text='whole-cell screening', entity_type='assay', char_start=268, char_end=288, attributes={}, alignment='match_exact')], mechanisms=[], accessions=[], products=[], functional_categories=[], screening_methods=[], strains=[], unclassified=[])"
|
|
324
331
|
]
|
|
325
332
|
},
|
|
326
|
-
"execution_count":
|
|
333
|
+
"execution_count": 6,
|
|
327
334
|
"metadata": {},
|
|
328
335
|
"output_type": "execute_result"
|
|
329
336
|
}
|
|
@@ -331,7 +338,7 @@
|
|
|
331
338
|
"source": [
|
|
332
339
|
"text = (\n",
|
|
333
340
|
" \"Bedaquiline (TMC207) is a diarylquinoline that inhibits the \"\n",
|
|
334
|
-
" \"mycobacterial ATP synthase subunit c encoded by atpE (Rv1305). \"\n",
|
|
341
|
+
" \"mycobacterial ATP synthase subunit c encoded by atpE (Rv1305) with IC50 = 0.033 µM\"\n",
|
|
335
342
|
" \"It shows potent activity against Mycobacterium tuberculosis \"\n",
|
|
336
343
|
" \"including MDR-TB and XDR-TB. This compound was identified through \"\n",
|
|
337
344
|
" \"whole-cell screening and targets the energy metabolism pathway.\"\n",
|
|
@@ -340,6 +347,230 @@
|
|
|
340
347
|
"result"
|
|
341
348
|
]
|
|
342
349
|
},
|
|
350
|
+
{
|
|
351
|
+
"cell_type": "code",
|
|
352
|
+
"execution_count": 7,
|
|
353
|
+
"metadata": {},
|
|
354
|
+
"outputs": [
|
|
355
|
+
{
|
|
356
|
+
"data": {
|
|
357
|
+
"text/html": [
|
|
358
|
+
"<div>\n",
|
|
359
|
+
"<style scoped>\n",
|
|
360
|
+
" .dataframe tbody tr th:only-of-type {\n",
|
|
361
|
+
" vertical-align: middle;\n",
|
|
362
|
+
" }\n",
|
|
363
|
+
"\n",
|
|
364
|
+
" .dataframe tbody tr th {\n",
|
|
365
|
+
" vertical-align: top;\n",
|
|
366
|
+
" }\n",
|
|
367
|
+
"\n",
|
|
368
|
+
" .dataframe thead th {\n",
|
|
369
|
+
" text-align: right;\n",
|
|
370
|
+
" }\n",
|
|
371
|
+
"</style>\n",
|
|
372
|
+
"<table border=\"1\" class=\"dataframe\">\n",
|
|
373
|
+
" <thead>\n",
|
|
374
|
+
" <tr style=\"text-align: right;\">\n",
|
|
375
|
+
" <th></th>\n",
|
|
376
|
+
" <th>text</th>\n",
|
|
377
|
+
" <th>entity_type</th>\n",
|
|
378
|
+
" <th>entity_class</th>\n",
|
|
379
|
+
" <th>char_start</th>\n",
|
|
380
|
+
" <th>char_end</th>\n",
|
|
381
|
+
" <th>alignment</th>\n",
|
|
382
|
+
" <th>synonyms</th>\n",
|
|
383
|
+
" <th>gene_name</th>\n",
|
|
384
|
+
" <th>protein_family</th>\n",
|
|
385
|
+
" <th>therapeutic_area</th>\n",
|
|
386
|
+
" <th>value</th>\n",
|
|
387
|
+
" <th>unit</th>\n",
|
|
388
|
+
" <th>assay_type</th>\n",
|
|
389
|
+
" <th>compound_name</th>\n",
|
|
390
|
+
" </tr>\n",
|
|
391
|
+
" </thead>\n",
|
|
392
|
+
" <tbody>\n",
|
|
393
|
+
" <tr>\n",
|
|
394
|
+
" <th>0</th>\n",
|
|
395
|
+
" <td>Bedaquiline</td>\n",
|
|
396
|
+
" <td>compound_name</td>\n",
|
|
397
|
+
" <td>ChemicalEntity</td>\n",
|
|
398
|
+
" <td>0</td>\n",
|
|
399
|
+
" <td>11</td>\n",
|
|
400
|
+
" <td>match_exact</td>\n",
|
|
401
|
+
" <td>TMC207</td>\n",
|
|
402
|
+
" <td>NaN</td>\n",
|
|
403
|
+
" <td>NaN</td>\n",
|
|
404
|
+
" <td>NaN</td>\n",
|
|
405
|
+
" <td>NaN</td>\n",
|
|
406
|
+
" <td>NaN</td>\n",
|
|
407
|
+
" <td>NaN</td>\n",
|
|
408
|
+
" <td>NaN</td>\n",
|
|
409
|
+
" </tr>\n",
|
|
410
|
+
" <tr>\n",
|
|
411
|
+
" <th>1</th>\n",
|
|
412
|
+
" <td>TMC207</td>\n",
|
|
413
|
+
" <td>compound_name</td>\n",
|
|
414
|
+
" <td>ChemicalEntity</td>\n",
|
|
415
|
+
" <td>13</td>\n",
|
|
416
|
+
" <td>19</td>\n",
|
|
417
|
+
" <td>match_exact</td>\n",
|
|
418
|
+
" <td>NaN</td>\n",
|
|
419
|
+
" <td>NaN</td>\n",
|
|
420
|
+
" <td>NaN</td>\n",
|
|
421
|
+
" <td>NaN</td>\n",
|
|
422
|
+
" <td>NaN</td>\n",
|
|
423
|
+
" <td>NaN</td>\n",
|
|
424
|
+
" <td>NaN</td>\n",
|
|
425
|
+
" <td>NaN</td>\n",
|
|
426
|
+
" </tr>\n",
|
|
427
|
+
" <tr>\n",
|
|
428
|
+
" <th>2</th>\n",
|
|
429
|
+
" <td>mycobacterial ATP synthase subunit c</td>\n",
|
|
430
|
+
" <td>target</td>\n",
|
|
431
|
+
" <td>TargetEntity</td>\n",
|
|
432
|
+
" <td>60</td>\n",
|
|
433
|
+
" <td>96</td>\n",
|
|
434
|
+
" <td>match_exact</td>\n",
|
|
435
|
+
" <td>NaN</td>\n",
|
|
436
|
+
" <td>atpE (Rv1305)</td>\n",
|
|
437
|
+
" <td>ATP synthase</td>\n",
|
|
438
|
+
" <td>NaN</td>\n",
|
|
439
|
+
" <td>NaN</td>\n",
|
|
440
|
+
" <td>NaN</td>\n",
|
|
441
|
+
" <td>NaN</td>\n",
|
|
442
|
+
" <td>NaN</td>\n",
|
|
443
|
+
" </tr>\n",
|
|
444
|
+
" <tr>\n",
|
|
445
|
+
" <th>3</th>\n",
|
|
446
|
+
" <td>energy metabolism pathway</td>\n",
|
|
447
|
+
" <td>target</td>\n",
|
|
448
|
+
" <td>TargetEntity</td>\n",
|
|
449
|
+
" <td>305</td>\n",
|
|
450
|
+
" <td>330</td>\n",
|
|
451
|
+
" <td>match_exact</td>\n",
|
|
452
|
+
" <td>NaN</td>\n",
|
|
453
|
+
" <td>NaN</td>\n",
|
|
454
|
+
" <td>pathway</td>\n",
|
|
455
|
+
" <td>NaN</td>\n",
|
|
456
|
+
" <td>NaN</td>\n",
|
|
457
|
+
" <td>NaN</td>\n",
|
|
458
|
+
" <td>NaN</td>\n",
|
|
459
|
+
" <td>NaN</td>\n",
|
|
460
|
+
" </tr>\n",
|
|
461
|
+
" <tr>\n",
|
|
462
|
+
" <th>4</th>\n",
|
|
463
|
+
" <td>MDR-TB</td>\n",
|
|
464
|
+
" <td>disease</td>\n",
|
|
465
|
+
" <td>DiseaseEntity</td>\n",
|
|
466
|
+
" <td>212</td>\n",
|
|
467
|
+
" <td>218</td>\n",
|
|
468
|
+
" <td>match_exact</td>\n",
|
|
469
|
+
" <td>NaN</td>\n",
|
|
470
|
+
" <td>NaN</td>\n",
|
|
471
|
+
" <td>NaN</td>\n",
|
|
472
|
+
" <td>infectious disease</td>\n",
|
|
473
|
+
" <td>NaN</td>\n",
|
|
474
|
+
" <td>NaN</td>\n",
|
|
475
|
+
" <td>NaN</td>\n",
|
|
476
|
+
" <td>NaN</td>\n",
|
|
477
|
+
" </tr>\n",
|
|
478
|
+
" <tr>\n",
|
|
479
|
+
" <th>5</th>\n",
|
|
480
|
+
" <td>XDR-TB</td>\n",
|
|
481
|
+
" <td>disease</td>\n",
|
|
482
|
+
" <td>DiseaseEntity</td>\n",
|
|
483
|
+
" <td>223</td>\n",
|
|
484
|
+
" <td>229</td>\n",
|
|
485
|
+
" <td>match_exact</td>\n",
|
|
486
|
+
" <td>NaN</td>\n",
|
|
487
|
+
" <td>NaN</td>\n",
|
|
488
|
+
" <td>NaN</td>\n",
|
|
489
|
+
" <td>infectious disease</td>\n",
|
|
490
|
+
" <td>NaN</td>\n",
|
|
491
|
+
" <td>NaN</td>\n",
|
|
492
|
+
" <td>NaN</td>\n",
|
|
493
|
+
" <td>NaN</td>\n",
|
|
494
|
+
" </tr>\n",
|
|
495
|
+
" <tr>\n",
|
|
496
|
+
" <th>6</th>\n",
|
|
497
|
+
" <td>IC50 = 0.033 µM</td>\n",
|
|
498
|
+
" <td>bioactivity</td>\n",
|
|
499
|
+
" <td>BioactivityEntity</td>\n",
|
|
500
|
+
" <td>127</td>\n",
|
|
501
|
+
" <td>139</td>\n",
|
|
502
|
+
" <td>match_lesser</td>\n",
|
|
503
|
+
" <td>NaN</td>\n",
|
|
504
|
+
" <td>NaN</td>\n",
|
|
505
|
+
" <td>NaN</td>\n",
|
|
506
|
+
" <td>NaN</td>\n",
|
|
507
|
+
" <td>0.033</td>\n",
|
|
508
|
+
" <td>µM</td>\n",
|
|
509
|
+
" <td>IC50</td>\n",
|
|
510
|
+
" <td>Bedaquiline</td>\n",
|
|
511
|
+
" </tr>\n",
|
|
512
|
+
" <tr>\n",
|
|
513
|
+
" <th>7</th>\n",
|
|
514
|
+
" <td>whole-cell screening</td>\n",
|
|
515
|
+
" <td>assay</td>\n",
|
|
516
|
+
" <td>AssayEntity</td>\n",
|
|
517
|
+
" <td>268</td>\n",
|
|
518
|
+
" <td>288</td>\n",
|
|
519
|
+
" <td>match_exact</td>\n",
|
|
520
|
+
" <td>NaN</td>\n",
|
|
521
|
+
" <td>NaN</td>\n",
|
|
522
|
+
" <td>NaN</td>\n",
|
|
523
|
+
" <td>NaN</td>\n",
|
|
524
|
+
" <td>NaN</td>\n",
|
|
525
|
+
" <td>NaN</td>\n",
|
|
526
|
+
" <td>NaN</td>\n",
|
|
527
|
+
" <td>NaN</td>\n",
|
|
528
|
+
" </tr>\n",
|
|
529
|
+
" </tbody>\n",
|
|
530
|
+
"</table>\n",
|
|
531
|
+
"</div>"
|
|
532
|
+
],
|
|
533
|
+
"text/plain": [
|
|
534
|
+
" text entity_type entity_class \\\n",
|
|
535
|
+
"0 Bedaquiline compound_name ChemicalEntity \n",
|
|
536
|
+
"1 TMC207 compound_name ChemicalEntity \n",
|
|
537
|
+
"2 mycobacterial ATP synthase subunit c target TargetEntity \n",
|
|
538
|
+
"3 energy metabolism pathway target TargetEntity \n",
|
|
539
|
+
"4 MDR-TB disease DiseaseEntity \n",
|
|
540
|
+
"5 XDR-TB disease DiseaseEntity \n",
|
|
541
|
+
"6 IC50 = 0.033 µM bioactivity BioactivityEntity \n",
|
|
542
|
+
"7 whole-cell screening assay AssayEntity \n",
|
|
543
|
+
"\n",
|
|
544
|
+
" char_start char_end alignment synonyms gene_name protein_family \\\n",
|
|
545
|
+
"0 0 11 match_exact TMC207 NaN NaN \n",
|
|
546
|
+
"1 13 19 match_exact NaN NaN NaN \n",
|
|
547
|
+
"2 60 96 match_exact NaN atpE (Rv1305) ATP synthase \n",
|
|
548
|
+
"3 305 330 match_exact NaN NaN pathway \n",
|
|
549
|
+
"4 212 218 match_exact NaN NaN NaN \n",
|
|
550
|
+
"5 223 229 match_exact NaN NaN NaN \n",
|
|
551
|
+
"6 127 139 match_lesser NaN NaN NaN \n",
|
|
552
|
+
"7 268 288 match_exact NaN NaN NaN \n",
|
|
553
|
+
"\n",
|
|
554
|
+
" therapeutic_area value unit assay_type compound_name \n",
|
|
555
|
+
"0 NaN NaN NaN NaN NaN \n",
|
|
556
|
+
"1 NaN NaN NaN NaN NaN \n",
|
|
557
|
+
"2 NaN NaN NaN NaN NaN \n",
|
|
558
|
+
"3 NaN NaN NaN NaN NaN \n",
|
|
559
|
+
"4 infectious disease NaN NaN NaN NaN \n",
|
|
560
|
+
"5 infectious disease NaN NaN NaN NaN \n",
|
|
561
|
+
"6 NaN 0.033 µM IC50 Bedaquiline \n",
|
|
562
|
+
"7 NaN NaN NaN NaN NaN "
|
|
563
|
+
]
|
|
564
|
+
},
|
|
565
|
+
"execution_count": 7,
|
|
566
|
+
"metadata": {},
|
|
567
|
+
"output_type": "execute_result"
|
|
568
|
+
}
|
|
569
|
+
],
|
|
570
|
+
"source": [
|
|
571
|
+
"result.to_dataframe()"
|
|
572
|
+
]
|
|
573
|
+
},
|
|
343
574
|
{
|
|
344
575
|
"cell_type": "markdown",
|
|
345
576
|
"metadata": {},
|
|
@@ -836,7 +1067,7 @@
|
|
|
836
1067
|
"name": "python",
|
|
837
1068
|
"nbconvert_exporter": "python",
|
|
838
1069
|
"pygments_lexer": "ipython3",
|
|
839
|
-
"version": "3.12.
|
|
1070
|
+
"version": "3.12.2"
|
|
840
1071
|
}
|
|
841
1072
|
},
|
|
842
1073
|
"nbformat": 4,
|
|
@@ -53,7 +53,11 @@ class MechanismEntity(NEREntity):
|
|
|
53
53
|
|
|
54
54
|
@dataclasses.dataclass(frozen=True)
|
|
55
55
|
class AccessionEntity(NEREntity):
|
|
56
|
-
"""A database accession: Rv locus tag, UniProt ID, or
|
|
56
|
+
"""A biological database accession: Rv locus tag, UniProt ID, PDB code, or RefSeq ID.
|
|
57
|
+
|
|
58
|
+
Chemistry registry identifiers (ChEMBL, ZINC, DrugBank) and compound
|
|
59
|
+
programme codes are :class:`ChemicalEntity`, not accessions.
|
|
60
|
+
"""
|
|
57
61
|
|
|
58
62
|
|
|
59
63
|
@dataclasses.dataclass(frozen=True)
|
|
@@ -804,11 +804,68 @@ _TB_EXAMPLE_4 = lx.data.ExampleData(
|
|
|
804
804
|
],
|
|
805
805
|
)
|
|
806
806
|
|
|
807
|
+
# Contrastive identifier example: chemistry registry ids and programme codes in
|
|
808
|
+
# the same text as biological accessions, so the model sees where each one goes.
|
|
809
|
+
_TB_EXAMPLE_5 = lx.data.ExampleData(
|
|
810
|
+
text=(
|
|
811
|
+
"SACC-3060 (CHEMBL4521987) inhibited DprE1, the decaprenylphosphoryl-beta-D-ribose "
|
|
812
|
+
"oxidase encoded by Rv3790, with an IC50 of 120 nM. Docking used UniProt P9WJG3 "
|
|
813
|
+
"and the PDB structure 4TZK. The related programme compound TBDA-01187 was inactive."
|
|
814
|
+
),
|
|
815
|
+
extractions=[
|
|
816
|
+
lx.data.Extraction(
|
|
817
|
+
extraction_class="compound_name",
|
|
818
|
+
extraction_text="SACC-3060",
|
|
819
|
+
attributes={"synonyms": "CHEMBL4521987"},
|
|
820
|
+
),
|
|
821
|
+
lx.data.Extraction(
|
|
822
|
+
extraction_class="compound_name",
|
|
823
|
+
extraction_text="CHEMBL4521987",
|
|
824
|
+
),
|
|
825
|
+
lx.data.Extraction(
|
|
826
|
+
extraction_class="target",
|
|
827
|
+
extraction_text="DprE1",
|
|
828
|
+
attributes={"gene_name": "Rv3790", "protein_family": "oxidase"},
|
|
829
|
+
),
|
|
830
|
+
lx.data.Extraction(
|
|
831
|
+
extraction_class="product",
|
|
832
|
+
extraction_text="decaprenylphosphoryl-beta-D-ribose oxidase",
|
|
833
|
+
),
|
|
834
|
+
lx.data.Extraction(
|
|
835
|
+
extraction_class="accession_number",
|
|
836
|
+
extraction_text="Rv3790",
|
|
837
|
+
),
|
|
838
|
+
lx.data.Extraction(
|
|
839
|
+
extraction_class="bioactivity",
|
|
840
|
+
extraction_text="IC50 of 120 nM",
|
|
841
|
+
attributes={
|
|
842
|
+
"value": "120",
|
|
843
|
+
"unit": "nM",
|
|
844
|
+
"assay_type": "IC50",
|
|
845
|
+
"compound_name": "SACC-3060",
|
|
846
|
+
},
|
|
847
|
+
),
|
|
848
|
+
lx.data.Extraction(
|
|
849
|
+
extraction_class="accession_number",
|
|
850
|
+
extraction_text="P9WJG3",
|
|
851
|
+
),
|
|
852
|
+
lx.data.Extraction(
|
|
853
|
+
extraction_class="accession_number",
|
|
854
|
+
extraction_text="4TZK",
|
|
855
|
+
),
|
|
856
|
+
lx.data.Extraction(
|
|
857
|
+
extraction_class="compound_name",
|
|
858
|
+
extraction_text="TBDA-01187",
|
|
859
|
+
),
|
|
860
|
+
],
|
|
861
|
+
)
|
|
862
|
+
|
|
807
863
|
TB_EXAMPLES: list[lx.data.ExampleData] = [
|
|
808
864
|
_TB_EXAMPLE_1,
|
|
809
865
|
_TB_EXAMPLE_2,
|
|
810
866
|
_TB_EXAMPLE_3,
|
|
811
867
|
_TB_EXAMPLE_4,
|
|
868
|
+
_TB_EXAMPLE_5,
|
|
812
869
|
]
|
|
813
870
|
|
|
814
871
|
TB_CHEMISTRY_EXAMPLES: list[lx.data.ExampleData] = [
|
|
@@ -64,6 +64,9 @@ TB_PROMPT = (
|
|
|
64
64
|
"are biological targets, NOT compounds.\n"
|
|
65
65
|
"- Rv locus tags (Rv3790, Rv1484), UniProt IDs (P9WPS1), and PDB codes "
|
|
66
66
|
"are accession_number, not target or gene_name.\n"
|
|
67
|
+
"- Compound registry and programme identifiers (CHEMBL4521987, ZINC000012345678, "
|
|
68
|
+
"DB00945, SACC-3060, TBDA-01187, GSK3036656) are compound_name; only Rv locus "
|
|
69
|
+
"tags, UniProt IDs, PDB codes and RefSeq IDs are accession_number.\n"
|
|
67
70
|
"- Enzyme descriptions like 'enoyl-ACP reductase' are product, not target.\n"
|
|
68
71
|
"- 'cell wall', 'lipid metabolism' are functional_category, not mechanism_of_action.\n"
|
|
69
72
|
"- 'fragment screening', 'biochemical assay' are screening_method, not assay.\n"
|