structflo-ner 0.3.0__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- structflo_ner-0.4.0/.claude/settings.local.json +13 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/.gitignore +4 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/PKG-INFO +1 -1
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/notebooks/01_quickstart.ipynb +201 -300
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/__init__.py +1 -1
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/extractor.py +55 -17
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/tests/test_extractor.py +68 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/.github/workflows/ci.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/.github/workflows/publish.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/Makefile +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/README.md +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/coverage.xml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/images/fast-viz.png +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/images/local-gen-pandas.png +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/images/local-gen-viz.png +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/images/local-tb-viz.png +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/images/ner_visualization.gif +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/images/struct-flo-ner.png +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/notebooks/02_fast_ner.ipynb +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/pyproject.toml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/_display.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/_entities.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/_examples.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/_mapping.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/_prompts.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/README.md +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/__init__.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/_loader.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/_matcher.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/_normalize.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/extractor.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/gazetteers/accession_number.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/gazetteers/compound_name.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/gazetteers/disease.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/gazetteers/functional_category.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/gazetteers/gene_name.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/gazetteers/product.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/gazetteers/screening_method.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/gazetteers/strain.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/gazetteers/target.yml +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/profiles.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/tests/__init__.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/tests/test_entities.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/tests/test_fast.py +0 -0
- {structflo_ner-0.3.0 → structflo_ner-0.4.0}/uv.lock +0 -0
|
@@ -28,7 +28,7 @@
|
|
|
28
28
|
},
|
|
29
29
|
{
|
|
30
30
|
"cell_type": "code",
|
|
31
|
-
"execution_count":
|
|
31
|
+
"execution_count": 1,
|
|
32
32
|
"metadata": {},
|
|
33
33
|
"outputs": [],
|
|
34
34
|
"source": [
|
|
@@ -37,7 +37,7 @@
|
|
|
37
37
|
},
|
|
38
38
|
{
|
|
39
39
|
"cell_type": "code",
|
|
40
|
-
"execution_count":
|
|
40
|
+
"execution_count": 2,
|
|
41
41
|
"metadata": {},
|
|
42
42
|
"outputs": [],
|
|
43
43
|
"source": [
|
|
@@ -89,7 +89,7 @@
|
|
|
89
89
|
},
|
|
90
90
|
{
|
|
91
91
|
"cell_type": "code",
|
|
92
|
-
"execution_count":
|
|
92
|
+
"execution_count": 3,
|
|
93
93
|
"metadata": {},
|
|
94
94
|
"outputs": [],
|
|
95
95
|
"source": [
|
|
@@ -101,19 +101,19 @@
|
|
|
101
101
|
},
|
|
102
102
|
{
|
|
103
103
|
"cell_type": "code",
|
|
104
|
-
"execution_count":
|
|
104
|
+
"execution_count": 4,
|
|
105
105
|
"metadata": {},
|
|
106
106
|
"outputs": [
|
|
107
107
|
{
|
|
108
108
|
"data": {
|
|
109
109
|
"text/html": [
|
|
110
|
-
"<div id=\"ner-
|
|
110
|
+
"<div id=\"ner-e5927285\" style=\"font-family:system-ui,-apple-system,sans-serif;max-width:900px;padding:20px\"><div style=\"display:flex;flex-wrap:wrap;gap:6px;margin-bottom:12px\"><button class=\"ner-legend-btn\" data-target=\"ChemicalEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #3b82f6;border-radius:16px;background:#dbeafe;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#3b82f6\"></span>Compound</button><button class=\"ner-legend-btn\" data-target=\"TargetEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #22c55e;border-radius:16px;background:#dcfce7;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#22c55e\"></span>Target</button><button class=\"ner-legend-btn\" data-target=\"DiseaseEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #ec4899;border-radius:16px;background:#fce7f3;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#ec4899\"></span>Disease</button><button class=\"ner-legend-btn\" data-target=\"BioactivityEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #f59e0b;border-radius:16px;background:#fef3c7;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#f59e0b\"></span>Bioactivity</button></div><div style=\"font-size:0.8em;color:#64748b;margin-bottom:10px\"><b>3</b> Compound · <b>1</b> Target · <b>1</b> Disease · <b>1</b> Bioactivity</div><div style=\"line-height:2;font-size:1em;white-space:pre-wrap\"><mark class=\"ner-ent\" data-category=\"ChemicalEntity\" style=\"background:#dbeafe;border-bottom:2px solid #3b82f6;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: compound_name | synonyms: ZD1839\">Gefitinib<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#3b82f6\">compound_name</span></mark> (<mark class=\"ner-ent\" data-category=\"ChemicalEntity\" style=\"background:#dbeafe;border-bottom:2px solid #3b82f6;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: compound_name\">ZD1839<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#3b82f6\">compound_name</span></mark>) is a first-generation <mark class=\"ner-ent\" data-category=\"TargetEntity\" style=\"background:#dcfce7;border-bottom:2px solid #22c55e;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: target\">EGFR<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#22c55e\">target</span></mark> inhibitor with <mark class=\"ner-ent\" data-category=\"BioactivityEntity\" style=\"background:#fef3c7;border-bottom:2px solid #f59e0b;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: bioactivity | value: 0.033 | unit: µM | assay_type: IC50 | compound_name: Gefitinib\">IC50 = 0.033 µM<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#f59e0b\">bioactivity</span></mark> approved for <mark class=\"ner-ent\" data-category=\"DiseaseEntity\" style=\"background:#fce7f3;border-bottom:2px solid #ec4899;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: disease | therapeutic_area: oncology\">NSCLC<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#ec4899\">disease</span></mark>.Its SMILES is <mark class=\"ner-ent\" data-category=\"ChemicalEntity\" style=\"background:#dbeafe;border-bottom:2px solid #3b82f6;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: smiles\">COc1cc2ncnc(Nc3ccc(F)c(Cl)c3)c2cc1OCCCN1CCOCC1<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#3b82f6\">smiles</span></mark>.</div><script>(function(){var w=document.getElementById('ner-e5927285');w.querySelectorAll('.ner-legend-btn').forEach(function(btn){btn.addEventListener('click',function(){var cat=btn.dataset.target;var active=btn.dataset.active!=='false';btn.dataset.active=active?'false':'true';btn.style.opacity=active?'0.35':'1';w.querySelectorAll('[data-category=\"'+cat+'\"]').forEach(function(el){if(el.classList.contains('ner-ent'))el.style.background=active?'transparent':el.style.borderBottomColor.replace('solid ','');el.style.opacity=active?'0.3':'1';});});});})()</script></div>"
|
|
111
111
|
],
|
|
112
112
|
"text/plain": [
|
|
113
|
-
"NERResult(source_text='Gefitinib (ZD1839) is a first-generation EGFR inhibitor with IC50 = 0.033 µM approved for NSCLC.Its SMILES is COc1cc2ncnc(Nc3ccc(F)c(Cl)c3)c2cc1OCCCN1CCOCC1.', compounds=[ChemicalEntity(text='Gefitinib', entity_type='compound_name', char_start=0, char_end=9, attributes={'synonyms': 'ZD1839'}, alignment='match_exact'), ChemicalEntity(text='ZD1839', entity_type='compound_name', char_start=11, char_end=17, attributes={}, alignment='match_exact'), ChemicalEntity(text='COc1cc2ncnc(Nc3ccc(F)c(Cl)c3)c2cc1OCCCN1CCOCC1', entity_type='smiles', char_start=110, char_end=156, attributes={}, alignment='match_exact')], targets=[TargetEntity(text='EGFR', entity_type='target', char_start=41, char_end=45, attributes={}, alignment='match_exact')], diseases=[DiseaseEntity(text='NSCLC', entity_type='disease', char_start=90, char_end=95, attributes={'therapeutic_area': 'oncology'}, alignment='match_exact')], bioactivities=[BioactivityEntity(text='IC50 = 0.033 µM', entity_type='bioactivity', char_start=61, char_end=76, attributes={'value': '0.033', 'unit': 'µM', 'assay_type': 'IC50'}, alignment='match_exact')], assays=[], mechanisms=[], accessions=[], products=[], functional_categories=[], screening_methods=[], unclassified=[])"
|
|
113
|
+
"NERResult(source_text='Gefitinib (ZD1839) is a first-generation EGFR inhibitor with IC50 = 0.033 µM approved for NSCLC.Its SMILES is COc1cc2ncnc(Nc3ccc(F)c(Cl)c3)c2cc1OCCCN1CCOCC1.', compounds=[ChemicalEntity(text='Gefitinib', entity_type='compound_name', char_start=0, char_end=9, attributes={'synonyms': 'ZD1839'}, alignment='match_exact'), ChemicalEntity(text='ZD1839', entity_type='compound_name', char_start=11, char_end=17, attributes={}, alignment='match_exact'), ChemicalEntity(text='COc1cc2ncnc(Nc3ccc(F)c(Cl)c3)c2cc1OCCCN1CCOCC1', entity_type='smiles', char_start=110, char_end=156, attributes={}, alignment='match_exact')], targets=[TargetEntity(text='EGFR', entity_type='target', char_start=41, char_end=45, attributes={}, alignment='match_exact')], diseases=[DiseaseEntity(text='NSCLC', entity_type='disease', char_start=90, char_end=95, attributes={'therapeutic_area': 'oncology'}, alignment='match_exact')], bioactivities=[BioactivityEntity(text='IC50 = 0.033 µM', entity_type='bioactivity', char_start=61, char_end=76, attributes={'value': '0.033', 'unit': 'µM', 'assay_type': 'IC50', 'compound_name': 'Gefitinib'}, alignment='match_exact')], assays=[], mechanisms=[], accessions=[], products=[], functional_categories=[], screening_methods=[], strains=[], unclassified=[])"
|
|
114
114
|
]
|
|
115
115
|
},
|
|
116
|
-
"execution_count":
|
|
116
|
+
"execution_count": 4,
|
|
117
117
|
"metadata": {},
|
|
118
118
|
"output_type": "execute_result"
|
|
119
119
|
}
|
|
@@ -127,7 +127,7 @@
|
|
|
127
127
|
},
|
|
128
128
|
{
|
|
129
129
|
"cell_type": "code",
|
|
130
|
-
"execution_count":
|
|
130
|
+
"execution_count": 5,
|
|
131
131
|
"metadata": {},
|
|
132
132
|
"outputs": [
|
|
133
133
|
{
|
|
@@ -162,6 +162,7 @@
|
|
|
162
162
|
" <th>value</th>\n",
|
|
163
163
|
" <th>unit</th>\n",
|
|
164
164
|
" <th>assay_type</th>\n",
|
|
165
|
+
" <th>compound_name</th>\n",
|
|
165
166
|
" </tr>\n",
|
|
166
167
|
" </thead>\n",
|
|
167
168
|
" <tbody>\n",
|
|
@@ -178,6 +179,7 @@
|
|
|
178
179
|
" <td>NaN</td>\n",
|
|
179
180
|
" <td>NaN</td>\n",
|
|
180
181
|
" <td>NaN</td>\n",
|
|
182
|
+
" <td>NaN</td>\n",
|
|
181
183
|
" </tr>\n",
|
|
182
184
|
" <tr>\n",
|
|
183
185
|
" <th>1</th>\n",
|
|
@@ -192,6 +194,7 @@
|
|
|
192
194
|
" <td>NaN</td>\n",
|
|
193
195
|
" <td>NaN</td>\n",
|
|
194
196
|
" <td>NaN</td>\n",
|
|
197
|
+
" <td>NaN</td>\n",
|
|
195
198
|
" </tr>\n",
|
|
196
199
|
" <tr>\n",
|
|
197
200
|
" <th>2</th>\n",
|
|
@@ -206,6 +209,7 @@
|
|
|
206
209
|
" <td>NaN</td>\n",
|
|
207
210
|
" <td>NaN</td>\n",
|
|
208
211
|
" <td>NaN</td>\n",
|
|
212
|
+
" <td>NaN</td>\n",
|
|
209
213
|
" </tr>\n",
|
|
210
214
|
" <tr>\n",
|
|
211
215
|
" <th>3</th>\n",
|
|
@@ -220,6 +224,7 @@
|
|
|
220
224
|
" <td>NaN</td>\n",
|
|
221
225
|
" <td>NaN</td>\n",
|
|
222
226
|
" <td>NaN</td>\n",
|
|
227
|
+
" <td>NaN</td>\n",
|
|
223
228
|
" </tr>\n",
|
|
224
229
|
" <tr>\n",
|
|
225
230
|
" <th>4</th>\n",
|
|
@@ -234,6 +239,7 @@
|
|
|
234
239
|
" <td>NaN</td>\n",
|
|
235
240
|
" <td>NaN</td>\n",
|
|
236
241
|
" <td>NaN</td>\n",
|
|
242
|
+
" <td>NaN</td>\n",
|
|
237
243
|
" </tr>\n",
|
|
238
244
|
" <tr>\n",
|
|
239
245
|
" <th>5</th>\n",
|
|
@@ -248,6 +254,7 @@
|
|
|
248
254
|
" <td>0.033</td>\n",
|
|
249
255
|
" <td>µM</td>\n",
|
|
250
256
|
" <td>IC50</td>\n",
|
|
257
|
+
" <td>Gefitinib</td>\n",
|
|
251
258
|
" </tr>\n",
|
|
252
259
|
" </tbody>\n",
|
|
253
260
|
"</table>\n",
|
|
@@ -270,16 +277,16 @@
|
|
|
270
277
|
"4 DiseaseEntity 90 95 match_exact NaN \n",
|
|
271
278
|
"5 BioactivityEntity 61 76 match_exact NaN \n",
|
|
272
279
|
"\n",
|
|
273
|
-
" therapeutic_area value unit assay_type \n",
|
|
274
|
-
"0 NaN NaN NaN NaN \n",
|
|
275
|
-
"1 NaN NaN NaN NaN \n",
|
|
276
|
-
"2 NaN NaN NaN NaN \n",
|
|
277
|
-
"3 NaN NaN NaN NaN \n",
|
|
278
|
-
"4 oncology NaN NaN NaN \n",
|
|
279
|
-
"5 NaN 0.033 µM IC50 "
|
|
280
|
+
" therapeutic_area value unit assay_type compound_name \n",
|
|
281
|
+
"0 NaN NaN NaN NaN NaN \n",
|
|
282
|
+
"1 NaN NaN NaN NaN NaN \n",
|
|
283
|
+
"2 NaN NaN NaN NaN NaN \n",
|
|
284
|
+
"3 NaN NaN NaN NaN NaN \n",
|
|
285
|
+
"4 oncology NaN NaN NaN NaN \n",
|
|
286
|
+
"5 NaN 0.033 µM IC50 Gefitinib "
|
|
280
287
|
]
|
|
281
288
|
},
|
|
282
|
-
"execution_count":
|
|
289
|
+
"execution_count": 5,
|
|
283
290
|
"metadata": {},
|
|
284
291
|
"output_type": "execute_result"
|
|
285
292
|
}
|
|
@@ -311,19 +318,19 @@
|
|
|
311
318
|
},
|
|
312
319
|
{
|
|
313
320
|
"cell_type": "code",
|
|
314
|
-
"execution_count":
|
|
321
|
+
"execution_count": 7,
|
|
315
322
|
"metadata": {},
|
|
316
323
|
"outputs": [
|
|
317
324
|
{
|
|
318
325
|
"data": {
|
|
319
326
|
"text/html": [
|
|
320
|
-
"<div id=\"ner-
|
|
327
|
+
"<div id=\"ner-742315f5\" style=\"font-family:system-ui,-apple-system,sans-serif;max-width:900px;padding:20px\"><div style=\"display:flex;flex-wrap:wrap;gap:6px;margin-bottom:12px\"><button class=\"ner-legend-btn\" data-target=\"ChemicalEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #3b82f6;border-radius:16px;background:#dbeafe;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#3b82f6\"></span>Compound</button><button class=\"ner-legend-btn\" data-target=\"TargetEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #22c55e;border-radius:16px;background:#dcfce7;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#22c55e\"></span>Target</button><button class=\"ner-legend-btn\" data-target=\"DiseaseEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #ec4899;border-radius:16px;background:#fce7f3;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#ec4899\"></span>Disease</button><button class=\"ner-legend-btn\" data-target=\"BioactivityEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #f59e0b;border-radius:16px;background:#fef3c7;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#f59e0b\"></span>Bioactivity</button><button class=\"ner-legend-btn\" data-target=\"AccessionEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #14b8a6;border-radius:16px;background:#ccfbf1;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#14b8a6\"></span>Accession</button><button class=\"ner-legend-btn\" data-target=\"FunctionalCategoryEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #f43f5e;border-radius:16px;background:#ffe4e6;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#f43f5e\"></span>Function</button><button class=\"ner-legend-btn\" data-target=\"ScreeningMethodEntity\" style=\"display:inline-flex;align-items:center;gap:4px;padding:4px 10px;border:1.5px solid #10b981;border-radius:16px;background:#d1fae5;cursor:pointer;font-size:0.8em;font-weight:500;font-family:inherit\"><span style=\"width:8px;height:8px;border-radius:50%;background:#10b981\"></span>Screening</button></div><div style=\"font-size:0.8em;color:#64748b;margin-bottom:10px\"><b>2</b> Compound · <b>1</b> Target · <b>2</b> Disease · <b>1</b> Bioactivity · <b>1</b> Accession · <b>1</b> Function · <b>1</b> Screening</div><div style=\"line-height:2;font-size:1em;white-space:pre-wrap\"><mark class=\"ner-ent\" data-category=\"ChemicalEntity\" style=\"background:#dbeafe;border-bottom:2px solid #3b82f6;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: compound_name | synonyms: TMC207\">Bedaquiline<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#3b82f6\">compound_name</span></mark> (<mark class=\"ner-ent\" data-category=\"ChemicalEntity\" style=\"background:#dbeafe;border-bottom:2px solid #3b82f6;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: compound_name\">TMC207<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#3b82f6\">compound_name</span></mark>) is a diarylquinoline that inhibits the mycobacterial <mark class=\"ner-ent\" data-category=\"TargetEntity\" style=\"background:#dcfce7;border-bottom:2px solid #22c55e;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: target | gene_name: atpE | protein_family: ATP synthase\">ATP synthase subunit c<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#22c55e\">target</span></mark> encoded by atpE (<mark class=\"ner-ent\" data-category=\"AccessionEntity\" style=\"background:#ccfbf1;border-bottom:2px solid #14b8a6;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: accession_number\">Rv1305<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#14b8a6\">accession_number</span></mark>) with <mark class=\"ner-ent\" data-category=\"BioactivityEntity\" style=\"background:#fef3c7;border-bottom:2px solid #f59e0b;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: bioactivity | value: 0.033 | unit: µM | assay_type: IC50 | compound_name: Bedaquiline\">IC50 = 0.033<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#f59e0b\">bioactivity</span></mark> µMIt shows potent activity against Mycobacterium tuberculosis including <mark class=\"ner-ent\" data-category=\"DiseaseEntity\" style=\"background:#fce7f3;border-bottom:2px solid #ec4899;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: disease | therapeutic_area: infectious disease\">MDR-TB<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#ec4899\">disease</span></mark> and <mark class=\"ner-ent\" data-category=\"DiseaseEntity\" style=\"background:#fce7f3;border-bottom:2px solid #ec4899;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: disease | therapeutic_area: infectious disease\">XDR-TB<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#ec4899\">disease</span></mark>. This compound was identified through <mark class=\"ner-ent\" data-category=\"ScreeningMethodEntity\" style=\"background:#d1fae5;border-bottom:2px solid #10b981;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: screening_method\">whole-cell screening<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#10b981\">screening_method</span></mark> and targets the <mark class=\"ner-ent\" data-category=\"FunctionalCategoryEntity\" style=\"background:#ffe4e6;border-bottom:2px solid #f43f5e;padding:2px 4px;border-radius:4px;cursor:default\" title=\"type: functional_category\">energy metabolism pathway<span style=\"font-size:0.7em;font-weight:600;vertical-align:super;margin-left:2px;color:#f43f5e\">functional_category</span></mark>.</div><script>(function(){var w=document.getElementById('ner-742315f5');w.querySelectorAll('.ner-legend-btn').forEach(function(btn){btn.addEventListener('click',function(){var cat=btn.dataset.target;var active=btn.dataset.active!=='false';btn.dataset.active=active?'false':'true';btn.style.opacity=active?'0.35':'1';w.querySelectorAll('[data-category=\"'+cat+'\"]').forEach(function(el){if(el.classList.contains('ner-ent'))el.style.background=active?'transparent':el.style.borderBottomColor.replace('solid ','');el.style.opacity=active?'0.3':'1';});});});})()</script></div>"
|
|
321
328
|
],
|
|
322
329
|
"text/plain": [
|
|
323
|
-
"NERResult(source_text='Bedaquiline (TMC207) is a diarylquinoline that inhibits the mycobacterial ATP synthase subunit c encoded by atpE (Rv1305).
|
|
330
|
+
"NERResult(source_text='Bedaquiline (TMC207) is a diarylquinoline that inhibits the mycobacterial ATP synthase subunit c encoded by atpE (Rv1305) with IC50 = 0.033 µMIt shows potent activity against Mycobacterium tuberculosis including MDR-TB and XDR-TB. This compound was identified through whole-cell screening and targets the energy metabolism pathway.', compounds=[ChemicalEntity(text='Bedaquiline', entity_type='compound_name', char_start=0, char_end=11, attributes={'synonyms': 'TMC207'}, alignment='match_exact'), ChemicalEntity(text='TMC207', entity_type='compound_name', char_start=13, char_end=19, attributes={}, alignment='match_exact')], targets=[TargetEntity(text='ATP synthase subunit c', entity_type='target', char_start=74, char_end=96, attributes={'gene_name': 'atpE', 'protein_family': 'ATP synthase'}, alignment='match_exact')], diseases=[DiseaseEntity(text='MDR-TB', entity_type='disease', char_start=212, char_end=218, attributes={'therapeutic_area': 'infectious disease'}, alignment='match_exact'), DiseaseEntity(text='XDR-TB', entity_type='disease', char_start=223, char_end=229, attributes={'therapeutic_area': 'infectious disease'}, alignment='match_exact')], bioactivities=[BioactivityEntity(text='IC50 = 0.033 µM', entity_type='bioactivity', char_start=127, char_end=139, attributes={'value': '0.033', 'unit': 'µM', 'assay_type': 'IC50', 'compound_name': 'Bedaquiline'}, alignment='match_lesser')], assays=[], mechanisms=[], accessions=[AccessionEntity(text='Rv1305', entity_type='accession_number', char_start=114, char_end=120, attributes={}, alignment='match_exact')], products=[], functional_categories=[FunctionalCategoryEntity(text='energy metabolism pathway', entity_type='functional_category', char_start=305, char_end=330, attributes={}, alignment='match_exact')], screening_methods=[ScreeningMethodEntity(text='whole-cell screening', entity_type='screening_method', char_start=268, char_end=288, attributes={}, alignment='match_exact')], strains=[], unclassified=[])"
|
|
324
331
|
]
|
|
325
332
|
},
|
|
326
|
-
"execution_count":
|
|
333
|
+
"execution_count": 7,
|
|
327
334
|
"metadata": {},
|
|
328
335
|
"output_type": "execute_result"
|
|
329
336
|
}
|
|
@@ -331,7 +338,7 @@
|
|
|
331
338
|
"source": [
|
|
332
339
|
"text = (\n",
|
|
333
340
|
" \"Bedaquiline (TMC207) is a diarylquinoline that inhibits the \"\n",
|
|
334
|
-
" \"mycobacterial ATP synthase subunit c encoded by atpE (Rv1305). \"\n",
|
|
341
|
+
" \"mycobacterial ATP synthase subunit c encoded by atpE (Rv1305) with IC50 = 0.033 µM\"\n",
|
|
335
342
|
" \"It shows potent activity against Mycobacterium tuberculosis \"\n",
|
|
336
343
|
" \"including MDR-TB and XDR-TB. This compound was identified through \"\n",
|
|
337
344
|
" \"whole-cell screening and targets the energy metabolism pathway.\"\n",
|
|
@@ -340,126 +347,9 @@
|
|
|
340
347
|
"result"
|
|
341
348
|
]
|
|
342
349
|
},
|
|
343
|
-
{
|
|
344
|
-
"cell_type": "markdown",
|
|
345
|
-
"metadata": {},
|
|
346
|
-
"source": [
|
|
347
|
-
"## 3. Built-in profiles\n",
|
|
348
|
-
"\n",
|
|
349
|
-
"Profiles control which entity types are extracted. Use them to focus the model on specific categories."
|
|
350
|
-
]
|
|
351
|
-
},
|
|
352
|
-
{
|
|
353
|
-
"cell_type": "code",
|
|
354
|
-
"execution_count": 12,
|
|
355
|
-
"metadata": {},
|
|
356
|
-
"outputs": [
|
|
357
|
-
{
|
|
358
|
-
"name": "stdout",
|
|
359
|
-
"output_type": "stream",
|
|
360
|
-
"text": [
|
|
361
|
-
"Compounds: [ChemicalEntity(text='Bedaquiline', entity_type='compound_name', char_start=0, char_end=11, attributes={'synonyms': 'TMC207'}, alignment='match_exact'), ChemicalEntity(text='TMC207', entity_type='compound_name', char_start=13, char_end=19, attributes={}, alignment='match_exact')]\n",
|
|
362
|
-
"Targets: []\n"
|
|
363
|
-
]
|
|
364
|
-
}
|
|
365
|
-
],
|
|
366
|
-
"source": [
|
|
367
|
-
"from structflo.ner import CHEMISTRY, BIOLOGY\n",
|
|
368
|
-
"# Extract only chemical entities\n",
|
|
369
|
-
"chem_result = extractor.extract(text, profile=CHEMISTRY)\n",
|
|
370
|
-
"print(\"Compounds:\", chem_result.compounds)\n",
|
|
371
|
-
"print(\"Targets:\", chem_result.targets) # empty — not in CHEMISTRY profile"
|
|
372
|
-
]
|
|
373
|
-
},
|
|
374
|
-
{
|
|
375
|
-
"cell_type": "code",
|
|
376
|
-
"execution_count": 13,
|
|
377
|
-
"metadata": {},
|
|
378
|
-
"outputs": [
|
|
379
|
-
{
|
|
380
|
-
"name": "stdout",
|
|
381
|
-
"output_type": "stream",
|
|
382
|
-
"text": [
|
|
383
|
-
"Profile: chemistry+biology\n",
|
|
384
|
-
"Entity classes: ['compound_name', 'smiles', 'cas_number', 'molecular_formula', 'target', 'gene_name', 'protein_name']\n",
|
|
385
|
-
"Compounds: [ChemicalEntity(text='Bedaquiline', entity_type='compound_name', char_start=0, char_end=11, attributes={'synonyms': 'TMC207'}, alignment='match_exact'), ChemicalEntity(text='TMC207', entity_type='compound_name', char_start=13, char_end=19, attributes={}, alignment='match_exact')]\n",
|
|
386
|
-
"Targets: [TargetEntity(text='mycobacterial ATP synthase subunit c', entity_type='target', char_start=60, char_end=96, attributes={'gene_name': 'atpE (Rv1305)', 'organism': 'Mycobacterium tuberculosis'}, alignment='match_exact'), TargetEntity(text='energy metabolism pathway', entity_type='target', char_start=286, char_end=311, attributes={'protein_family': 'pathway', 'organism': 'Mycobacterium tuberculosis'}, alignment='match_exact')]\n"
|
|
387
|
-
]
|
|
388
|
-
}
|
|
389
|
-
],
|
|
390
|
-
"source": [
|
|
391
|
-
"# Merge profiles to combine entity types\n",
|
|
392
|
-
"combined = CHEMISTRY.merge(BIOLOGY)\n",
|
|
393
|
-
"print(f\"Profile: {combined.name}\")\n",
|
|
394
|
-
"print(f\"Entity classes: {combined.entity_classes}\")\n",
|
|
395
|
-
"\n",
|
|
396
|
-
"combined_result = extractor.extract(text, profile=combined)\n",
|
|
397
|
-
"print(\"Compounds:\", combined_result.compounds)\n",
|
|
398
|
-
"print(\"Targets:\", combined_result.targets)"
|
|
399
|
-
]
|
|
400
|
-
},
|
|
401
|
-
{
|
|
402
|
-
"cell_type": "markdown",
|
|
403
|
-
"metadata": {},
|
|
404
|
-
"source": [
|
|
405
|
-
"## 5. Working with results"
|
|
406
|
-
]
|
|
407
|
-
},
|
|
408
350
|
{
|
|
409
351
|
"cell_type": "code",
|
|
410
|
-
"execution_count":
|
|
411
|
-
"metadata": {},
|
|
412
|
-
"outputs": [
|
|
413
|
-
{
|
|
414
|
-
"name": "stdout",
|
|
415
|
-
"output_type": "stream",
|
|
416
|
-
"text": [
|
|
417
|
-
"Compounds: [ChemicalEntity(text='Bedaquiline', entity_type='compound_name', char_start=0, char_end=11, attributes={'synonyms': 'TMC207'}, alignment='match_exact'), ChemicalEntity(text='TMC207', entity_type='compound_name', char_start=13, char_end=19, attributes={}, alignment='match_exact')]\n",
|
|
418
|
-
"Targets: [TargetEntity(text='ATP synthase subunit c', entity_type='target', char_start=74, char_end=96, attributes={'gene_name': 'atpE', 'protein_family': 'ATP synthase'}, alignment='match_exact')]\n",
|
|
419
|
-
"Bioactivities: []\n",
|
|
420
|
-
"Diseases: [DiseaseEntity(text='MDR-TB', entity_type='disease', char_start=193, char_end=199, attributes={'therapeutic_area': 'infectious disease'}, alignment='match_exact'), DiseaseEntity(text='XDR-TB', entity_type='disease', char_start=204, char_end=210, attributes={'therapeutic_area': 'infectious disease'}, alignment='match_exact')]\n",
|
|
421
|
-
"Mechanisms: []\n"
|
|
422
|
-
]
|
|
423
|
-
}
|
|
424
|
-
],
|
|
425
|
-
"source": [
|
|
426
|
-
"# Access typed entity lists\n",
|
|
427
|
-
"print(\"Compounds:\", result.compounds)\n",
|
|
428
|
-
"print(\"Targets:\", result.targets)\n",
|
|
429
|
-
"print(\"Bioactivities:\", result.bioactivities)\n",
|
|
430
|
-
"print(\"Diseases:\", result.diseases)\n",
|
|
431
|
-
"print(\"Mechanisms:\", result.mechanisms)"
|
|
432
|
-
]
|
|
433
|
-
},
|
|
434
|
-
{
|
|
435
|
-
"cell_type": "code",
|
|
436
|
-
"execution_count": 16,
|
|
437
|
-
"metadata": {},
|
|
438
|
-
"outputs": [
|
|
439
|
-
{
|
|
440
|
-
"name": "stdout",
|
|
441
|
-
"output_type": "stream",
|
|
442
|
-
"text": [
|
|
443
|
-
"compound_name | Bedaquiline\n",
|
|
444
|
-
"compound_name | TMC207\n",
|
|
445
|
-
"target | ATP synthase subunit c\n",
|
|
446
|
-
"disease | MDR-TB\n",
|
|
447
|
-
"disease | XDR-TB\n",
|
|
448
|
-
"accession_number | Rv1305\n",
|
|
449
|
-
"functional_category | energy metabolism pathway\n",
|
|
450
|
-
"screening_method | whole-cell screening\n"
|
|
451
|
-
]
|
|
452
|
-
}
|
|
453
|
-
],
|
|
454
|
-
"source": [
|
|
455
|
-
"# Flat list of all entities\n",
|
|
456
|
-
"for entity in result.all_entities():\n",
|
|
457
|
-
" print(f\"{entity.entity_type:20s} | {entity.text}\")"
|
|
458
|
-
]
|
|
459
|
-
},
|
|
460
|
-
{
|
|
461
|
-
"cell_type": "code",
|
|
462
|
-
"execution_count": 17,
|
|
352
|
+
"execution_count": 8,
|
|
463
353
|
"metadata": {},
|
|
464
354
|
"outputs": [
|
|
465
355
|
{
|
|
@@ -493,6 +383,10 @@
|
|
|
493
383
|
" <th>gene_name</th>\n",
|
|
494
384
|
" <th>protein_family</th>\n",
|
|
495
385
|
" <th>therapeutic_area</th>\n",
|
|
386
|
+
" <th>value</th>\n",
|
|
387
|
+
" <th>unit</th>\n",
|
|
388
|
+
" <th>assay_type</th>\n",
|
|
389
|
+
" <th>compound_name</th>\n",
|
|
496
390
|
" </tr>\n",
|
|
497
391
|
" </thead>\n",
|
|
498
392
|
" <tbody>\n",
|
|
@@ -508,6 +402,10 @@
|
|
|
508
402
|
" <td>NaN</td>\n",
|
|
509
403
|
" <td>NaN</td>\n",
|
|
510
404
|
" <td>NaN</td>\n",
|
|
405
|
+
" <td>NaN</td>\n",
|
|
406
|
+
" <td>NaN</td>\n",
|
|
407
|
+
" <td>NaN</td>\n",
|
|
408
|
+
" <td>NaN</td>\n",
|
|
511
409
|
" </tr>\n",
|
|
512
410
|
" <tr>\n",
|
|
513
411
|
" <th>1</th>\n",
|
|
@@ -521,6 +419,10 @@
|
|
|
521
419
|
" <td>NaN</td>\n",
|
|
522
420
|
" <td>NaN</td>\n",
|
|
523
421
|
" <td>NaN</td>\n",
|
|
422
|
+
" <td>NaN</td>\n",
|
|
423
|
+
" <td>NaN</td>\n",
|
|
424
|
+
" <td>NaN</td>\n",
|
|
425
|
+
" <td>NaN</td>\n",
|
|
524
426
|
" </tr>\n",
|
|
525
427
|
" <tr>\n",
|
|
526
428
|
" <th>2</th>\n",
|
|
@@ -534,35 +436,64 @@
|
|
|
534
436
|
" <td>atpE</td>\n",
|
|
535
437
|
" <td>ATP synthase</td>\n",
|
|
536
438
|
" <td>NaN</td>\n",
|
|
439
|
+
" <td>NaN</td>\n",
|
|
440
|
+
" <td>NaN</td>\n",
|
|
441
|
+
" <td>NaN</td>\n",
|
|
442
|
+
" <td>NaN</td>\n",
|
|
537
443
|
" </tr>\n",
|
|
538
444
|
" <tr>\n",
|
|
539
445
|
" <th>3</th>\n",
|
|
540
446
|
" <td>MDR-TB</td>\n",
|
|
541
447
|
" <td>disease</td>\n",
|
|
542
448
|
" <td>DiseaseEntity</td>\n",
|
|
543
|
-
" <td>
|
|
544
|
-
" <td>
|
|
449
|
+
" <td>212</td>\n",
|
|
450
|
+
" <td>218</td>\n",
|
|
545
451
|
" <td>match_exact</td>\n",
|
|
546
452
|
" <td>NaN</td>\n",
|
|
547
453
|
" <td>NaN</td>\n",
|
|
548
454
|
" <td>NaN</td>\n",
|
|
549
455
|
" <td>infectious disease</td>\n",
|
|
456
|
+
" <td>NaN</td>\n",
|
|
457
|
+
" <td>NaN</td>\n",
|
|
458
|
+
" <td>NaN</td>\n",
|
|
459
|
+
" <td>NaN</td>\n",
|
|
550
460
|
" </tr>\n",
|
|
551
461
|
" <tr>\n",
|
|
552
462
|
" <th>4</th>\n",
|
|
553
463
|
" <td>XDR-TB</td>\n",
|
|
554
464
|
" <td>disease</td>\n",
|
|
555
465
|
" <td>DiseaseEntity</td>\n",
|
|
556
|
-
" <td>
|
|
557
|
-
" <td>
|
|
466
|
+
" <td>223</td>\n",
|
|
467
|
+
" <td>229</td>\n",
|
|
558
468
|
" <td>match_exact</td>\n",
|
|
559
469
|
" <td>NaN</td>\n",
|
|
560
470
|
" <td>NaN</td>\n",
|
|
561
471
|
" <td>NaN</td>\n",
|
|
562
472
|
" <td>infectious disease</td>\n",
|
|
473
|
+
" <td>NaN</td>\n",
|
|
474
|
+
" <td>NaN</td>\n",
|
|
475
|
+
" <td>NaN</td>\n",
|
|
476
|
+
" <td>NaN</td>\n",
|
|
563
477
|
" </tr>\n",
|
|
564
478
|
" <tr>\n",
|
|
565
479
|
" <th>5</th>\n",
|
|
480
|
+
" <td>IC50 = 0.033 µM</td>\n",
|
|
481
|
+
" <td>bioactivity</td>\n",
|
|
482
|
+
" <td>BioactivityEntity</td>\n",
|
|
483
|
+
" <td>127</td>\n",
|
|
484
|
+
" <td>139</td>\n",
|
|
485
|
+
" <td>match_lesser</td>\n",
|
|
486
|
+
" <td>NaN</td>\n",
|
|
487
|
+
" <td>NaN</td>\n",
|
|
488
|
+
" <td>NaN</td>\n",
|
|
489
|
+
" <td>NaN</td>\n",
|
|
490
|
+
" <td>0.033</td>\n",
|
|
491
|
+
" <td>µM</td>\n",
|
|
492
|
+
" <td>IC50</td>\n",
|
|
493
|
+
" <td>Bedaquiline</td>\n",
|
|
494
|
+
" </tr>\n",
|
|
495
|
+
" <tr>\n",
|
|
496
|
+
" <th>6</th>\n",
|
|
566
497
|
" <td>Rv1305</td>\n",
|
|
567
498
|
" <td>accession_number</td>\n",
|
|
568
499
|
" <td>AccessionEntity</td>\n",
|
|
@@ -573,32 +504,44 @@
|
|
|
573
504
|
" <td>NaN</td>\n",
|
|
574
505
|
" <td>NaN</td>\n",
|
|
575
506
|
" <td>NaN</td>\n",
|
|
507
|
+
" <td>NaN</td>\n",
|
|
508
|
+
" <td>NaN</td>\n",
|
|
509
|
+
" <td>NaN</td>\n",
|
|
510
|
+
" <td>NaN</td>\n",
|
|
576
511
|
" </tr>\n",
|
|
577
512
|
" <tr>\n",
|
|
578
|
-
" <th>
|
|
513
|
+
" <th>7</th>\n",
|
|
579
514
|
" <td>energy metabolism pathway</td>\n",
|
|
580
515
|
" <td>functional_category</td>\n",
|
|
581
516
|
" <td>FunctionalCategoryEntity</td>\n",
|
|
582
|
-
" <td>
|
|
583
|
-
" <td>
|
|
517
|
+
" <td>305</td>\n",
|
|
518
|
+
" <td>330</td>\n",
|
|
584
519
|
" <td>match_exact</td>\n",
|
|
585
520
|
" <td>NaN</td>\n",
|
|
586
521
|
" <td>NaN</td>\n",
|
|
587
522
|
" <td>NaN</td>\n",
|
|
588
523
|
" <td>NaN</td>\n",
|
|
524
|
+
" <td>NaN</td>\n",
|
|
525
|
+
" <td>NaN</td>\n",
|
|
526
|
+
" <td>NaN</td>\n",
|
|
527
|
+
" <td>NaN</td>\n",
|
|
589
528
|
" </tr>\n",
|
|
590
529
|
" <tr>\n",
|
|
591
|
-
" <th>
|
|
530
|
+
" <th>8</th>\n",
|
|
592
531
|
" <td>whole-cell screening</td>\n",
|
|
593
532
|
" <td>screening_method</td>\n",
|
|
594
533
|
" <td>ScreeningMethodEntity</td>\n",
|
|
595
|
-
" <td>
|
|
596
|
-
" <td>
|
|
534
|
+
" <td>268</td>\n",
|
|
535
|
+
" <td>288</td>\n",
|
|
597
536
|
" <td>match_exact</td>\n",
|
|
598
537
|
" <td>NaN</td>\n",
|
|
599
538
|
" <td>NaN</td>\n",
|
|
600
539
|
" <td>NaN</td>\n",
|
|
601
540
|
" <td>NaN</td>\n",
|
|
541
|
+
" <td>NaN</td>\n",
|
|
542
|
+
" <td>NaN</td>\n",
|
|
543
|
+
" <td>NaN</td>\n",
|
|
544
|
+
" <td>NaN</td>\n",
|
|
602
545
|
" </tr>\n",
|
|
603
546
|
" </tbody>\n",
|
|
604
547
|
"</table>\n",
|
|
@@ -611,36 +554,118 @@
|
|
|
611
554
|
"2 ATP synthase subunit c target TargetEntity \n",
|
|
612
555
|
"3 MDR-TB disease DiseaseEntity \n",
|
|
613
556
|
"4 XDR-TB disease DiseaseEntity \n",
|
|
614
|
-
"5
|
|
615
|
-
"6
|
|
616
|
-
"7
|
|
557
|
+
"5 IC50 = 0.033 µM bioactivity BioactivityEntity \n",
|
|
558
|
+
"6 Rv1305 accession_number AccessionEntity \n",
|
|
559
|
+
"7 energy metabolism pathway functional_category FunctionalCategoryEntity \n",
|
|
560
|
+
"8 whole-cell screening screening_method ScreeningMethodEntity \n",
|
|
617
561
|
"\n",
|
|
618
|
-
" char_start char_end
|
|
619
|
-
"0 0 11
|
|
620
|
-
"1 13 19
|
|
621
|
-
"2 74 96
|
|
622
|
-
"3
|
|
623
|
-
"4
|
|
624
|
-
"5
|
|
625
|
-
"6
|
|
626
|
-
"7
|
|
562
|
+
" char_start char_end alignment synonyms gene_name protein_family \\\n",
|
|
563
|
+
"0 0 11 match_exact TMC207 NaN NaN \n",
|
|
564
|
+
"1 13 19 match_exact NaN NaN NaN \n",
|
|
565
|
+
"2 74 96 match_exact NaN atpE ATP synthase \n",
|
|
566
|
+
"3 212 218 match_exact NaN NaN NaN \n",
|
|
567
|
+
"4 223 229 match_exact NaN NaN NaN \n",
|
|
568
|
+
"5 127 139 match_lesser NaN NaN NaN \n",
|
|
569
|
+
"6 114 120 match_exact NaN NaN NaN \n",
|
|
570
|
+
"7 305 330 match_exact NaN NaN NaN \n",
|
|
571
|
+
"8 268 288 match_exact NaN NaN NaN \n",
|
|
627
572
|
"\n",
|
|
628
|
-
" therapeutic_area \n",
|
|
629
|
-
"0 NaN \n",
|
|
630
|
-
"1 NaN \n",
|
|
631
|
-
"2 NaN \n",
|
|
632
|
-
"3 infectious disease \n",
|
|
633
|
-
"4 infectious disease \n",
|
|
634
|
-
"5 NaN \n",
|
|
635
|
-
"6 NaN \n",
|
|
636
|
-
"7 NaN "
|
|
573
|
+
" therapeutic_area value unit assay_type compound_name \n",
|
|
574
|
+
"0 NaN NaN NaN NaN NaN \n",
|
|
575
|
+
"1 NaN NaN NaN NaN NaN \n",
|
|
576
|
+
"2 NaN NaN NaN NaN NaN \n",
|
|
577
|
+
"3 infectious disease NaN NaN NaN NaN \n",
|
|
578
|
+
"4 infectious disease NaN NaN NaN NaN \n",
|
|
579
|
+
"5 NaN 0.033 µM IC50 Bedaquiline \n",
|
|
580
|
+
"6 NaN NaN NaN NaN NaN \n",
|
|
581
|
+
"7 NaN NaN NaN NaN NaN \n",
|
|
582
|
+
"8 NaN NaN NaN NaN NaN "
|
|
637
583
|
]
|
|
638
584
|
},
|
|
639
|
-
"execution_count":
|
|
585
|
+
"execution_count": 8,
|
|
640
586
|
"metadata": {},
|
|
641
587
|
"output_type": "execute_result"
|
|
642
588
|
}
|
|
643
589
|
],
|
|
590
|
+
"source": [
|
|
591
|
+
"result.to_dataframe()"
|
|
592
|
+
]
|
|
593
|
+
},
|
|
594
|
+
{
|
|
595
|
+
"cell_type": "markdown",
|
|
596
|
+
"metadata": {},
|
|
597
|
+
"source": [
|
|
598
|
+
"## 3. Built-in profiles\n",
|
|
599
|
+
"\n",
|
|
600
|
+
"Profiles control which entity types are extracted. Use them to focus the model on specific categories."
|
|
601
|
+
]
|
|
602
|
+
},
|
|
603
|
+
{
|
|
604
|
+
"cell_type": "code",
|
|
605
|
+
"execution_count": null,
|
|
606
|
+
"metadata": {},
|
|
607
|
+
"outputs": [],
|
|
608
|
+
"source": [
|
|
609
|
+
"from structflo.ner import CHEMISTRY, BIOLOGY\n",
|
|
610
|
+
"# Extract only chemical entities\n",
|
|
611
|
+
"chem_result = extractor.extract(text, profile=CHEMISTRY)\n",
|
|
612
|
+
"print(\"Compounds:\", chem_result.compounds)\n",
|
|
613
|
+
"print(\"Targets:\", chem_result.targets) # empty — not in CHEMISTRY profile"
|
|
614
|
+
]
|
|
615
|
+
},
|
|
616
|
+
{
|
|
617
|
+
"cell_type": "code",
|
|
618
|
+
"execution_count": null,
|
|
619
|
+
"metadata": {},
|
|
620
|
+
"outputs": [],
|
|
621
|
+
"source": [
|
|
622
|
+
"# Merge profiles to combine entity types\n",
|
|
623
|
+
"combined = CHEMISTRY.merge(BIOLOGY)\n",
|
|
624
|
+
"print(f\"Profile: {combined.name}\")\n",
|
|
625
|
+
"print(f\"Entity classes: {combined.entity_classes}\")\n",
|
|
626
|
+
"\n",
|
|
627
|
+
"combined_result = extractor.extract(text, profile=combined)\n",
|
|
628
|
+
"print(\"Compounds:\", combined_result.compounds)\n",
|
|
629
|
+
"print(\"Targets:\", combined_result.targets)"
|
|
630
|
+
]
|
|
631
|
+
},
|
|
632
|
+
{
|
|
633
|
+
"cell_type": "markdown",
|
|
634
|
+
"metadata": {},
|
|
635
|
+
"source": [
|
|
636
|
+
"## 5. Working with results"
|
|
637
|
+
]
|
|
638
|
+
},
|
|
639
|
+
{
|
|
640
|
+
"cell_type": "code",
|
|
641
|
+
"execution_count": null,
|
|
642
|
+
"metadata": {},
|
|
643
|
+
"outputs": [],
|
|
644
|
+
"source": [
|
|
645
|
+
"# Access typed entity lists\n",
|
|
646
|
+
"print(\"Compounds:\", result.compounds)\n",
|
|
647
|
+
"print(\"Targets:\", result.targets)\n",
|
|
648
|
+
"print(\"Bioactivities:\", result.bioactivities)\n",
|
|
649
|
+
"print(\"Diseases:\", result.diseases)\n",
|
|
650
|
+
"print(\"Mechanisms:\", result.mechanisms)"
|
|
651
|
+
]
|
|
652
|
+
},
|
|
653
|
+
{
|
|
654
|
+
"cell_type": "code",
|
|
655
|
+
"execution_count": null,
|
|
656
|
+
"metadata": {},
|
|
657
|
+
"outputs": [],
|
|
658
|
+
"source": [
|
|
659
|
+
"# Flat list of all entities\n",
|
|
660
|
+
"for entity in result.all_entities():\n",
|
|
661
|
+
" print(f\"{entity.entity_type:20s} | {entity.text}\")"
|
|
662
|
+
]
|
|
663
|
+
},
|
|
664
|
+
{
|
|
665
|
+
"cell_type": "code",
|
|
666
|
+
"execution_count": null,
|
|
667
|
+
"metadata": {},
|
|
668
|
+
"outputs": [],
|
|
644
669
|
"source": [
|
|
645
670
|
"# Export to pandas DataFrame\n",
|
|
646
671
|
"df = result.to_dataframe()\n",
|
|
@@ -649,109 +674,9 @@
|
|
|
649
674
|
},
|
|
650
675
|
{
|
|
651
676
|
"cell_type": "code",
|
|
652
|
-
"execution_count":
|
|
677
|
+
"execution_count": null,
|
|
653
678
|
"metadata": {},
|
|
654
|
-
"outputs": [
|
|
655
|
-
{
|
|
656
|
-
"name": "stdout",
|
|
657
|
-
"output_type": "stream",
|
|
658
|
-
"text": [
|
|
659
|
-
"{\n",
|
|
660
|
-
" \"source_text\": \"Bedaquiline (TMC207) is a diarylquinoline that inhibits the mycobacterial ATP synthase subunit c encoded by atpE (Rv1305). It shows potent activity against Mycobacterium tuberculosis including MDR-TB and XDR-TB. This compound was identified through whole-cell screening and targets the energy metabolism pathway.\",\n",
|
|
661
|
-
" \"compounds\": [\n",
|
|
662
|
-
" {\n",
|
|
663
|
-
" \"text\": \"Bedaquiline\",\n",
|
|
664
|
-
" \"entity_type\": \"compound_name\",\n",
|
|
665
|
-
" \"char_start\": 0,\n",
|
|
666
|
-
" \"char_end\": 11,\n",
|
|
667
|
-
" \"attributes\": {\n",
|
|
668
|
-
" \"synonyms\": \"TMC207\"\n",
|
|
669
|
-
" },\n",
|
|
670
|
-
" \"alignment\": \"match_exact\"\n",
|
|
671
|
-
" },\n",
|
|
672
|
-
" {\n",
|
|
673
|
-
" \"text\": \"TMC207\",\n",
|
|
674
|
-
" \"entity_type\": \"compound_name\",\n",
|
|
675
|
-
" \"char_start\": 13,\n",
|
|
676
|
-
" \"char_end\": 19,\n",
|
|
677
|
-
" \"attributes\": {},\n",
|
|
678
|
-
" \"alignment\": \"match_exact\"\n",
|
|
679
|
-
" }\n",
|
|
680
|
-
" ],\n",
|
|
681
|
-
" \"targets\": [\n",
|
|
682
|
-
" {\n",
|
|
683
|
-
" \"text\": \"ATP synthase subunit c\",\n",
|
|
684
|
-
" \"entity_type\": \"target\",\n",
|
|
685
|
-
" \"char_start\": 74,\n",
|
|
686
|
-
" \"char_end\": 96,\n",
|
|
687
|
-
" \"attributes\": {\n",
|
|
688
|
-
" \"gene_name\": \"atpE\",\n",
|
|
689
|
-
" \"protein_family\": \"ATP synthase\"\n",
|
|
690
|
-
" },\n",
|
|
691
|
-
" \"alignment\": \"match_exact\"\n",
|
|
692
|
-
" }\n",
|
|
693
|
-
" ],\n",
|
|
694
|
-
" \"diseases\": [\n",
|
|
695
|
-
" {\n",
|
|
696
|
-
" \"text\": \"MDR-TB\",\n",
|
|
697
|
-
" \"entity_type\": \"disease\",\n",
|
|
698
|
-
" \"char_start\": 193,\n",
|
|
699
|
-
" \"char_end\": 199,\n",
|
|
700
|
-
" \"attributes\": {\n",
|
|
701
|
-
" \"therapeutic_area\": \"infectious disease\"\n",
|
|
702
|
-
" },\n",
|
|
703
|
-
" \"alignment\": \"match_exact\"\n",
|
|
704
|
-
" },\n",
|
|
705
|
-
" {\n",
|
|
706
|
-
" \"text\": \"XDR-TB\",\n",
|
|
707
|
-
" \"entity_type\": \"disease\",\n",
|
|
708
|
-
" \"char_start\": 204,\n",
|
|
709
|
-
" \"char_end\": 210,\n",
|
|
710
|
-
" \"attributes\": {\n",
|
|
711
|
-
" \"therapeutic_area\": \"infectious disease\"\n",
|
|
712
|
-
" },\n",
|
|
713
|
-
" \"alignment\": \"match_exact\"\n",
|
|
714
|
-
" }\n",
|
|
715
|
-
" ],\n",
|
|
716
|
-
" \"bioactivities\": [],\n",
|
|
717
|
-
" \"assays\": [],\n",
|
|
718
|
-
" \"mechanisms\": [],\n",
|
|
719
|
-
" \"accessions\": [\n",
|
|
720
|
-
" {\n",
|
|
721
|
-
" \"text\": \"Rv1305\",\n",
|
|
722
|
-
" \"entity_type\": \"accession_number\",\n",
|
|
723
|
-
" \"char_start\": 114,\n",
|
|
724
|
-
" \"char_end\": 120,\n",
|
|
725
|
-
" \"attributes\": {},\n",
|
|
726
|
-
" \"alignment\": \"match_exact\"\n",
|
|
727
|
-
" }\n",
|
|
728
|
-
" ],\n",
|
|
729
|
-
" \"products\": [],\n",
|
|
730
|
-
" \"functional_categories\": [\n",
|
|
731
|
-
" {\n",
|
|
732
|
-
" \"text\": \"energy metabolism pathway\",\n",
|
|
733
|
-
" \"entity_type\": \"functional_category\",\n",
|
|
734
|
-
" \"char_start\": 286,\n",
|
|
735
|
-
" \"char_end\": 311,\n",
|
|
736
|
-
" \"attributes\": {},\n",
|
|
737
|
-
" \"alignment\": \"match_exact\"\n",
|
|
738
|
-
" }\n",
|
|
739
|
-
" ],\n",
|
|
740
|
-
" \"screening_methods\": [\n",
|
|
741
|
-
" {\n",
|
|
742
|
-
" \"text\": \"whole-cell screening\",\n",
|
|
743
|
-
" \"entity_type\": \"screening_method\",\n",
|
|
744
|
-
" \"char_start\": 249,\n",
|
|
745
|
-
" \"char_end\": 269,\n",
|
|
746
|
-
" \"attributes\": {},\n",
|
|
747
|
-
" \"alignment\": \"match_exact\"\n",
|
|
748
|
-
" }\n",
|
|
749
|
-
" ],\n",
|
|
750
|
-
" \"unclassified\": []\n",
|
|
751
|
-
"}\n"
|
|
752
|
-
]
|
|
753
|
-
}
|
|
754
|
-
],
|
|
679
|
+
"outputs": [],
|
|
755
680
|
"source": [
|
|
756
681
|
"# Serialize to dict (useful for JSON export)\n",
|
|
757
682
|
"import json\n",
|
|
@@ -770,33 +695,9 @@
|
|
|
770
695
|
},
|
|
771
696
|
{
|
|
772
697
|
"cell_type": "code",
|
|
773
|
-
"execution_count":
|
|
698
|
+
"execution_count": null,
|
|
774
699
|
"metadata": {},
|
|
775
|
-
"outputs": [
|
|
776
|
-
{
|
|
777
|
-
"name": "stdout",
|
|
778
|
-
"output_type": "stream",
|
|
779
|
-
"text": [
|
|
780
|
-
"\n",
|
|
781
|
-
"--- Text 1 ---\n",
|
|
782
|
-
" compound_name | Imatinib\n",
|
|
783
|
-
" target | BCR-ABL\n",
|
|
784
|
-
" disease | CML\n",
|
|
785
|
-
" bioactivity | IC50 = 0.6 µM\n",
|
|
786
|
-
"\n",
|
|
787
|
-
"--- Text 2 ---\n",
|
|
788
|
-
" compound_name | Trastuzumab\n",
|
|
789
|
-
" target | HER2\n",
|
|
790
|
-
" disease | breast cancer\n",
|
|
791
|
-
"\n",
|
|
792
|
-
"--- Text 3 ---\n",
|
|
793
|
-
" compound_name | Remdesivir\n",
|
|
794
|
-
" compound_name | GS-5734\n",
|
|
795
|
-
" disease | SARS-CoV-2\n",
|
|
796
|
-
" bioactivity | EC50 = 0.77 µM\n"
|
|
797
|
-
]
|
|
798
|
-
}
|
|
799
|
-
],
|
|
700
|
+
"outputs": [],
|
|
800
701
|
"source": [
|
|
801
702
|
"texts = [\n",
|
|
802
703
|
" \"Imatinib inhibits BCR-ABL with IC50 = 0.6 µM in CML.\",\n",
|
|
@@ -836,7 +737,7 @@
|
|
|
836
737
|
"name": "python",
|
|
837
738
|
"nbconvert_exporter": "python",
|
|
838
739
|
"pygments_lexer": "ipython3",
|
|
839
|
-
"version": "3.12.
|
|
740
|
+
"version": "3.12.2"
|
|
840
741
|
}
|
|
841
742
|
},
|
|
842
743
|
"nbformat": 4,
|
|
@@ -46,6 +46,12 @@ class NERExtractor:
|
|
|
46
46
|
model_url: Base URL for self-hosted models (e.g. Ollama at
|
|
47
47
|
``"http://localhost:11434"``). When set, langextract routes
|
|
48
48
|
requests to this endpoint instead of a cloud API.
|
|
49
|
+
provider: Explicit langextract provider name (``"ollama"``,
|
|
50
|
+
``"openai"``, ``"gemini"``). When set, routing is deterministic and
|
|
51
|
+
no longer depends on ``model_id`` matching a built-in regex — use
|
|
52
|
+
this for non-standard model names (e.g. ``"medgemma:latest"``) that
|
|
53
|
+
would otherwise fail to resolve. When ``None`` (default), the
|
|
54
|
+
provider is inferred from ``model_id`` as before.
|
|
49
55
|
profile: Default :class:`EntityProfile` to use when no per-call
|
|
50
56
|
profile is specified. Defaults to :data:`FULL`.
|
|
51
57
|
extra_examples: Additional :class:`lx.data.ExampleData` objects that
|
|
@@ -63,6 +69,7 @@ class NERExtractor:
|
|
|
63
69
|
model_id: str = "gemini-2.5-flash",
|
|
64
70
|
api_key: str | None = None,
|
|
65
71
|
model_url: str | None = None,
|
|
72
|
+
provider: str | None = None,
|
|
66
73
|
profile: EntityProfile = FULL,
|
|
67
74
|
extra_examples: list[lx.data.ExampleData] | None = None,
|
|
68
75
|
langextract_kwargs: dict | None = None,
|
|
@@ -70,6 +77,7 @@ class NERExtractor:
|
|
|
70
77
|
self._model_id = model_id
|
|
71
78
|
self._api_key = api_key
|
|
72
79
|
self._model_url = model_url
|
|
80
|
+
self._provider = provider
|
|
73
81
|
self._default_profile = profile
|
|
74
82
|
self._extra_examples = extra_examples or []
|
|
75
83
|
self._langextract_kwargs = langextract_kwargs or {}
|
|
@@ -159,7 +167,30 @@ class NERExtractor:
|
|
|
159
167
|
@property
|
|
160
168
|
def _is_ollama(self) -> bool:
|
|
161
169
|
"""Return True when routing to an Ollama endpoint."""
|
|
162
|
-
return self._model_url is not None
|
|
170
|
+
return self._provider == "ollama" or (self._provider is None and self._model_url is not None)
|
|
171
|
+
|
|
172
|
+
def _build_config(self) -> "lx.factory.ModelConfig":
|
|
173
|
+
"""Build an explicit ModelConfig for deterministic provider routing.
|
|
174
|
+
|
|
175
|
+
Provider-specific settings travel in ``provider_kwargs`` (passed to the
|
|
176
|
+
provider constructor): ``api_key`` for cloud providers, ``base_url`` +
|
|
177
|
+
``num_ctx`` for Ollama.
|
|
178
|
+
"""
|
|
179
|
+
from langextract import factory
|
|
180
|
+
|
|
181
|
+
provider_kwargs: dict = {}
|
|
182
|
+
if self._api_key is not None:
|
|
183
|
+
provider_kwargs["api_key"] = self._api_key
|
|
184
|
+
if self._model_url is not None:
|
|
185
|
+
provider_kwargs["base_url"] = self._model_url
|
|
186
|
+
if self._is_ollama:
|
|
187
|
+
# Ollama defaults to num_ctx=2048, far too small for few-shot NER.
|
|
188
|
+
provider_kwargs.setdefault("num_ctx", 8192)
|
|
189
|
+
return factory.ModelConfig(
|
|
190
|
+
model_id=self._model_id,
|
|
191
|
+
provider=self._provider,
|
|
192
|
+
provider_kwargs=provider_kwargs,
|
|
193
|
+
)
|
|
163
194
|
|
|
164
195
|
def _run_extraction(
|
|
165
196
|
self,
|
|
@@ -174,22 +205,29 @@ class NERExtractor:
|
|
|
174
205
|
kwargs.setdefault("use_schema_constraints", True)
|
|
175
206
|
kwargs.setdefault("show_progress", False)
|
|
176
207
|
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
208
|
+
if self._provider is not None:
|
|
209
|
+
# Explicit provider → deterministic routing via ModelConfig.
|
|
210
|
+
result = lx.extract(
|
|
211
|
+
text_or_documents=text,
|
|
212
|
+
prompt_description=prompt,
|
|
213
|
+
examples=examples,
|
|
214
|
+
config=self._build_config(),
|
|
215
|
+
**kwargs,
|
|
216
|
+
)
|
|
217
|
+
else:
|
|
218
|
+
# Legacy path: provider inferred from model_id regex by langextract.
|
|
219
|
+
if self._is_ollama:
|
|
220
|
+
lm_params = kwargs.setdefault("language_model_params", {})
|
|
221
|
+
lm_params.setdefault("num_ctx", 8192)
|
|
222
|
+
result = lx.extract(
|
|
223
|
+
text_or_documents=text,
|
|
224
|
+
prompt_description=prompt,
|
|
225
|
+
examples=examples,
|
|
226
|
+
model_id=self._model_id,
|
|
227
|
+
api_key=self._api_key,
|
|
228
|
+
model_url=self._model_url,
|
|
229
|
+
**kwargs,
|
|
230
|
+
)
|
|
193
231
|
|
|
194
232
|
# lx.extract returns a list when given a list; we always pass a single string
|
|
195
233
|
doc = result[0] if isinstance(result, list) else result
|
|
@@ -122,6 +122,74 @@ class TestNERExtractorExtract:
|
|
|
122
122
|
assert result.source_text == "My source text"
|
|
123
123
|
|
|
124
124
|
|
|
125
|
+
class TestProviderRouting:
|
|
126
|
+
"""Explicit provider selection routes deterministically via ModelConfig,
|
|
127
|
+
instead of relying on langextract's model_id regex inference."""
|
|
128
|
+
|
|
129
|
+
def _run_and_capture(self, monkeypatch, **init_kwargs):
|
|
130
|
+
from structflo.ner import extractor as extractor_mod
|
|
131
|
+
|
|
132
|
+
captured = {}
|
|
133
|
+
|
|
134
|
+
def fake_extract(**kwargs):
|
|
135
|
+
captured.update(kwargs)
|
|
136
|
+
return _make_annotated_doc([])
|
|
137
|
+
|
|
138
|
+
monkeypatch.setattr(extractor_mod.lx, "extract", fake_extract)
|
|
139
|
+
NERExtractor(**init_kwargs).extract("some text")
|
|
140
|
+
return captured
|
|
141
|
+
|
|
142
|
+
def test_provider_defaults_to_none(self):
|
|
143
|
+
assert NERExtractor()._provider is None
|
|
144
|
+
|
|
145
|
+
def test_provider_stored(self):
|
|
146
|
+
assert NERExtractor(provider="openai")._provider == "openai"
|
|
147
|
+
|
|
148
|
+
def test_no_provider_uses_legacy_model_id_path(self, monkeypatch):
|
|
149
|
+
captured = self._run_and_capture(
|
|
150
|
+
monkeypatch, model_id="gemma3:27b", model_url="http://ollama:11434"
|
|
151
|
+
)
|
|
152
|
+
assert captured["model_id"] == "gemma3:27b"
|
|
153
|
+
assert captured["model_url"] == "http://ollama:11434"
|
|
154
|
+
assert "config" not in captured
|
|
155
|
+
|
|
156
|
+
def test_explicit_provider_uses_config_not_top_level_model_id(self, monkeypatch):
|
|
157
|
+
captured = self._run_and_capture(
|
|
158
|
+
monkeypatch,
|
|
159
|
+
provider="ollama",
|
|
160
|
+
model_id="medgemma:latest",
|
|
161
|
+
model_url="http://ollama:11434",
|
|
162
|
+
)
|
|
163
|
+
# Deterministic routing: config carries provider + model_id, and the
|
|
164
|
+
# top-level model_id/model_url are NOT passed (would conflict with config).
|
|
165
|
+
assert "config" in captured
|
|
166
|
+
assert "model_id" not in captured
|
|
167
|
+
assert "model_url" not in captured
|
|
168
|
+
config = captured["config"]
|
|
169
|
+
assert config.provider == "ollama"
|
|
170
|
+
assert config.model_id == "medgemma:latest"
|
|
171
|
+
|
|
172
|
+
def test_ollama_provider_kwargs_carry_base_url_and_num_ctx(self, monkeypatch):
|
|
173
|
+
captured = self._run_and_capture(
|
|
174
|
+
monkeypatch,
|
|
175
|
+
provider="ollama",
|
|
176
|
+
model_id="medgemma:latest",
|
|
177
|
+
model_url="http://ollama:11434",
|
|
178
|
+
)
|
|
179
|
+
pk = captured["config"].provider_kwargs
|
|
180
|
+
assert pk["base_url"] == "http://ollama:11434"
|
|
181
|
+
assert pk["num_ctx"] == 8192
|
|
182
|
+
|
|
183
|
+
def test_cloud_provider_kwargs_carry_api_key(self, monkeypatch):
|
|
184
|
+
captured = self._run_and_capture(
|
|
185
|
+
monkeypatch, provider="openai", model_id="gpt-4o", api_key="sk-test"
|
|
186
|
+
)
|
|
187
|
+
pk = captured["config"].provider_kwargs
|
|
188
|
+
assert pk["api_key"] == "sk-test"
|
|
189
|
+
assert "num_ctx" not in pk
|
|
190
|
+
assert "base_url" not in pk
|
|
191
|
+
|
|
192
|
+
|
|
125
193
|
class TestBuildPrompt:
|
|
126
194
|
def test_prompt_includes_entity_class_constraint(self):
|
|
127
195
|
extractor = NERExtractor()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/gazetteers/accession_number.yml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/gazetteers/functional_category.yml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structflo_ner-0.3.0 → structflo_ner-0.4.0}/structflo/ner/fast/gazetteers/screening_method.yml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|