encode-toolkit 0.3.2__py3-none-any.whl → 0.3.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- encode_connector/__init__.py +6 -1
- encode_connector/client/constants.py +55 -4
- encode_connector/client/encode_client.py +66 -34
- encode_connector/client/models.py +27 -5
- encode_connector/client/tracker.py +48 -10
- encode_connector/server/main.py +39 -11
- {encode_toolkit-0.3.2.dist-info → encode_toolkit-0.3.4.dist-info}/METADATA +9 -9
- encode_toolkit-0.3.4.dist-info/RECORD +18 -0
- {encode_toolkit-0.3.2.dist-info → encode_toolkit-0.3.4.dist-info}/WHEEL +1 -1
- encode_toolkit-0.3.2.dist-info/RECORD +0 -18
- {encode_toolkit-0.3.2.dist-info → encode_toolkit-0.3.4.dist-info}/entry_points.txt +0 -0
- {encode_toolkit-0.3.2.dist-info → encode_toolkit-0.3.4.dist-info}/licenses/LICENSE +0 -0
encode_connector/__init__.py
CHANGED
|
@@ -1,4 +1,9 @@
|
|
|
1
1
|
"""ENCODE Project connector - MCP server and Python client."""
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
4
|
+
|
|
5
|
+
try:
|
|
6
|
+
__version__ = version("encode-toolkit")
|
|
7
|
+
except PackageNotFoundError: # a source tree that is not installed
|
|
8
|
+
__version__ = "0+unknown"
|
|
4
9
|
__author__ = "Dr. Alex M. Mawla, PhD"
|
|
@@ -12,12 +12,21 @@ DOWNLOAD_CONCURRENCY = 3
|
|
|
12
12
|
DEFAULT_TIMEOUT = 30.0
|
|
13
13
|
DOWNLOAD_TIMEOUT = 300.0
|
|
14
14
|
DEFAULT_LIMIT = 25
|
|
15
|
+
# Frame for a single experiment. "page" is "embedded" plus the "audit" property; with
|
|
16
|
+
# "embedded" ENCODE omits the audits and every experiment looks free of errors and warnings.
|
|
17
|
+
EXPERIMENT_FRAME = "page"
|
|
18
|
+
# experiments read per request while a file search walks the experiments of one organism
|
|
19
|
+
EXPERIMENT_PAGE_SIZE = 200
|
|
20
|
+
# ...and how many experiments it reads at most: each one costs a request to ENCODE
|
|
21
|
+
MAX_EXPERIMENTS_SCANNED = 1000
|
|
22
|
+
# files of one experiment read per request during that walk
|
|
23
|
+
FILES_PAGE_SIZE = 200
|
|
15
24
|
try:
|
|
16
25
|
import importlib.metadata
|
|
17
26
|
|
|
18
27
|
_version = importlib.metadata.version("encode-toolkit")
|
|
19
28
|
except importlib.metadata.PackageNotFoundError:
|
|
20
|
-
_version = "0.3.
|
|
29
|
+
_version = "0.3.4"
|
|
21
30
|
USER_AGENT = f"encode-toolkit/{_version} (MCP; +https://github.com/ammawla/encode-toolkit)"
|
|
22
31
|
|
|
23
32
|
# Keyring service name for credential storage
|
|
@@ -239,8 +248,18 @@ FILE_FORMATS = [
|
|
|
239
248
|
"vcf",
|
|
240
249
|
"bigInteract",
|
|
241
250
|
"idx",
|
|
242
|
-
"dat",
|
|
243
251
|
"txt",
|
|
252
|
+
"h5ad",
|
|
253
|
+
"hdf5",
|
|
254
|
+
"sam",
|
|
255
|
+
"wig",
|
|
256
|
+
"starch",
|
|
257
|
+
"chain",
|
|
258
|
+
"PWM",
|
|
259
|
+
"btr",
|
|
260
|
+
"cndb",
|
|
261
|
+
"nucle3d",
|
|
262
|
+
"yaml",
|
|
244
263
|
]
|
|
245
264
|
|
|
246
265
|
OUTPUT_TYPES = [
|
|
@@ -277,7 +296,6 @@ OUTPUT_TYPES = [
|
|
|
277
296
|
"pseudoreplicated peaks",
|
|
278
297
|
"pseudoreplicated IDR thresholded peaks",
|
|
279
298
|
"replicated peaks",
|
|
280
|
-
"stable peaks",
|
|
281
299
|
"hotspots",
|
|
282
300
|
"footprints",
|
|
283
301
|
"peaks and background as input for IDR",
|
|
@@ -289,6 +307,10 @@ OUTPUT_TYPES = [
|
|
|
289
307
|
"filtered peaks",
|
|
290
308
|
# Quantifications
|
|
291
309
|
"gene quantifications",
|
|
310
|
+
"sparse gene count matrix of unique reads",
|
|
311
|
+
"sparse gene count matrix of all reads",
|
|
312
|
+
"unfiltered sparse gene count matrix of unique reads",
|
|
313
|
+
"unfiltered sparse gene count matrix of all reads",
|
|
292
314
|
"transcript quantifications",
|
|
293
315
|
"exon quantifications",
|
|
294
316
|
"microRNA quantifications",
|
|
@@ -347,7 +369,6 @@ OUTPUT_CATEGORIES = [
|
|
|
347
369
|
"annotation",
|
|
348
370
|
"quantification",
|
|
349
371
|
"reference",
|
|
350
|
-
"quality metric",
|
|
351
372
|
]
|
|
352
373
|
|
|
353
374
|
FILE_STATUSES = [
|
|
@@ -381,6 +402,14 @@ ASSEMBLIES = [
|
|
|
381
402
|
"dm3",
|
|
382
403
|
"ce11",
|
|
383
404
|
"ce10",
|
|
405
|
+
"GRCh38-minimal",
|
|
406
|
+
"mm10-minimal",
|
|
407
|
+
"T2T-CHM13",
|
|
408
|
+
"J02459.1",
|
|
409
|
+
"ENC001.1",
|
|
410
|
+
"ENC002.1",
|
|
411
|
+
"ENC003.1",
|
|
412
|
+
"ENC004.1",
|
|
384
413
|
]
|
|
385
414
|
|
|
386
415
|
LIFE_STAGES = [
|
|
@@ -416,6 +445,28 @@ METADATA_MAP = {
|
|
|
416
445
|
}
|
|
417
446
|
|
|
418
447
|
# ENCODE API parameter name mapping (user-friendly -> API param)
|
|
448
|
+
# Fields requested from /search/?type=Experiment. They are what ExperimentSummary.from_api
|
|
449
|
+
# reads; naming them makes the API embed labels instead of returning object paths.
|
|
450
|
+
EXPERIMENT_SEARCH_FIELDS = (
|
|
451
|
+
"accession",
|
|
452
|
+
"assay_title",
|
|
453
|
+
"target.label",
|
|
454
|
+
"biosample_summary",
|
|
455
|
+
"biosample_ontology.classification",
|
|
456
|
+
"biosample_ontology.organ_slims",
|
|
457
|
+
"replicates.library.biosample.organism.scientific_name",
|
|
458
|
+
"status",
|
|
459
|
+
"date_released",
|
|
460
|
+
"description",
|
|
461
|
+
"lab.title",
|
|
462
|
+
"files.@id",
|
|
463
|
+
"replication_type",
|
|
464
|
+
"life_stage_age",
|
|
465
|
+
"assembly",
|
|
466
|
+
"audit",
|
|
467
|
+
"dbxrefs",
|
|
468
|
+
)
|
|
469
|
+
|
|
419
470
|
EXPERIMENT_FILTER_MAP = {
|
|
420
471
|
"assay_title": "assay_title",
|
|
421
472
|
"organism": "replicates.library.biosample.donor.organism.scientific_name",
|
|
@@ -19,7 +19,12 @@ from encode_connector.client.constants import (
|
|
|
19
19
|
DEFAULT_LIMIT,
|
|
20
20
|
DEFAULT_TIMEOUT,
|
|
21
21
|
EXPERIMENT_FILTER_MAP,
|
|
22
|
+
EXPERIMENT_FRAME,
|
|
23
|
+
EXPERIMENT_PAGE_SIZE,
|
|
24
|
+
EXPERIMENT_SEARCH_FIELDS,
|
|
22
25
|
FILE_FILTER_MAP,
|
|
26
|
+
FILES_PAGE_SIZE,
|
|
27
|
+
MAX_EXPERIMENTS_SCANNED,
|
|
23
28
|
MAX_REQUESTS_PER_SECOND,
|
|
24
29
|
METADATA_MAP,
|
|
25
30
|
USER_AGENT,
|
|
@@ -236,10 +241,14 @@ class EncodeClient:
|
|
|
236
241
|
Returns dict with 'results' (list of ExperimentSummary) and 'total' count.
|
|
237
242
|
"""
|
|
238
243
|
limit = clamp_limit(limit)
|
|
244
|
+
offset = max(0, offset)
|
|
245
|
+
# frame=object replaces linked objects with paths ("/targets/H3K4me1-human/"), which
|
|
246
|
+
# loses the labels the filters use, the organ, the organism and the assemblies. Naming
|
|
247
|
+
# the fields makes the API embed exactly those.
|
|
239
248
|
params: dict[str, Any] = {
|
|
240
249
|
"type": "Experiment",
|
|
241
250
|
"format": "json",
|
|
242
|
-
"
|
|
251
|
+
"field": list(EXPERIMENT_SEARCH_FIELDS),
|
|
243
252
|
"limit": limit,
|
|
244
253
|
}
|
|
245
254
|
if offset > 0:
|
|
@@ -297,11 +306,11 @@ class EncodeClient:
|
|
|
297
306
|
}
|
|
298
307
|
|
|
299
308
|
async def get_experiment_raw(self, accession: str) -> dict:
|
|
300
|
-
"""Get raw experiment data
|
|
309
|
+
"""Get raw experiment data: linked objects embedded, audits included."""
|
|
301
310
|
validate_accession(accession)
|
|
302
311
|
return await self._request(
|
|
303
312
|
f"/experiments/{accession}/",
|
|
304
|
-
{"format": "json", "frame":
|
|
313
|
+
{"format": "json", "frame": EXPERIMENT_FRAME},
|
|
305
314
|
)
|
|
306
315
|
|
|
307
316
|
async def get_experiment(self, accession: str) -> ExperimentDetail:
|
|
@@ -310,7 +319,7 @@ class EncodeClient:
|
|
|
310
319
|
# Get experiment data
|
|
311
320
|
exp_data = await self._request(
|
|
312
321
|
f"/experiments/{accession}/",
|
|
313
|
-
{"format": "json", "frame":
|
|
322
|
+
{"format": "json", "frame": EXPERIMENT_FRAME},
|
|
314
323
|
)
|
|
315
324
|
|
|
316
325
|
# Get files for this experiment
|
|
@@ -343,6 +352,7 @@ class EncodeClient:
|
|
|
343
352
|
status: str | None = None,
|
|
344
353
|
preferred_default: bool | None = None,
|
|
345
354
|
limit: int = 200,
|
|
355
|
+
offset: int = 0,
|
|
346
356
|
) -> list[FileSummary]:
|
|
347
357
|
"""List files for a specific experiment with optional filters."""
|
|
348
358
|
validate_accession(experiment_accession)
|
|
@@ -354,6 +364,8 @@ class EncodeClient:
|
|
|
354
364
|
"frame": "object",
|
|
355
365
|
"limit": limit,
|
|
356
366
|
}
|
|
367
|
+
if offset > 0:
|
|
368
|
+
params["from"] = offset
|
|
357
369
|
|
|
358
370
|
filter_values = {
|
|
359
371
|
"file_format": file_format,
|
|
@@ -395,6 +407,7 @@ class EncodeClient:
|
|
|
395
407
|
) -> dict[str, Any]:
|
|
396
408
|
"""Search files across all experiments with combined filters."""
|
|
397
409
|
limit = clamp_limit(limit)
|
|
410
|
+
offset = max(0, offset)
|
|
398
411
|
params: dict[str, Any] = {
|
|
399
412
|
"type": "File",
|
|
400
413
|
"format": "json",
|
|
@@ -434,41 +447,60 @@ class EncodeClient:
|
|
|
434
447
|
# For organism filtering on files, we need a two-step approach:
|
|
435
448
|
# search experiments first, then get files from matching experiments
|
|
436
449
|
if organism:
|
|
437
|
-
#
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
experiment_accession=exp.accession,
|
|
456
|
-
file_format=file_format,
|
|
457
|
-
file_type=file_type,
|
|
458
|
-
output_type=output_type,
|
|
459
|
-
output_category=output_category,
|
|
460
|
-
assembly=assembly,
|
|
461
|
-
status=status,
|
|
462
|
-
preferred_default=preferred_default,
|
|
450
|
+
# Collect one file past the requested page so the caller can tell that more exist.
|
|
451
|
+
# Matching files may belong to experiments far down the list, so keep reading
|
|
452
|
+
# experiment pages until the page is full or the experiments run out.
|
|
453
|
+
wanted = offset + limit + 1
|
|
454
|
+
all_files: list[FileSummary] = []
|
|
455
|
+
experiment_offset = 0
|
|
456
|
+
experiment_total = 0
|
|
457
|
+
while len(all_files) < wanted and experiment_offset < MAX_EXPERIMENTS_SCANNED:
|
|
458
|
+
exp_result = await self.search_experiments(
|
|
459
|
+
assay_title=assay_title,
|
|
460
|
+
organism=organism,
|
|
461
|
+
organ=organ,
|
|
462
|
+
biosample_type=biosample_type,
|
|
463
|
+
target=target,
|
|
464
|
+
status=status or "released",
|
|
465
|
+
search_term=search_term,
|
|
466
|
+
limit=EXPERIMENT_PAGE_SIZE,
|
|
467
|
+
offset=experiment_offset,
|
|
463
468
|
)
|
|
464
|
-
|
|
465
|
-
|
|
469
|
+
experiments = exp_result["results"]
|
|
470
|
+
experiment_total = exp_result.get("total", 0)
|
|
471
|
+
for exp in experiments:
|
|
472
|
+
# one experiment can hold more files than a request returns
|
|
473
|
+
file_offset = 0
|
|
474
|
+
while len(all_files) < wanted:
|
|
475
|
+
exp_files = await self.list_files(
|
|
476
|
+
experiment_accession=exp.accession,
|
|
477
|
+
file_format=file_format,
|
|
478
|
+
file_type=file_type,
|
|
479
|
+
output_type=output_type,
|
|
480
|
+
output_category=output_category,
|
|
481
|
+
assembly=assembly,
|
|
482
|
+
status=status,
|
|
483
|
+
preferred_default=preferred_default,
|
|
484
|
+
limit=FILES_PAGE_SIZE,
|
|
485
|
+
offset=file_offset,
|
|
486
|
+
)
|
|
487
|
+
all_files.extend(exp_files)
|
|
488
|
+
file_offset += len(exp_files)
|
|
489
|
+
if len(exp_files) < FILES_PAGE_SIZE:
|
|
490
|
+
break
|
|
491
|
+
if len(all_files) >= wanted:
|
|
492
|
+
break
|
|
493
|
+
experiment_offset += len(experiments)
|
|
494
|
+
if not experiments or experiment_offset >= experiment_total:
|
|
466
495
|
break
|
|
467
496
|
|
|
497
|
+
total_note = "Lower bound: files collected so far from matching experiments, not the full count"
|
|
498
|
+
if len(all_files) < wanted and experiment_total > experiment_offset >= MAX_EXPERIMENTS_SCANNED:
|
|
499
|
+
total_note += f"; stopped after reading {experiment_offset} of {experiment_total} experiments"
|
|
468
500
|
return {
|
|
469
|
-
"results": all_files[:limit],
|
|
501
|
+
"results": all_files[offset : offset + limit],
|
|
470
502
|
"total": len(all_files),
|
|
471
|
-
"total_note":
|
|
503
|
+
"total_note": total_note,
|
|
472
504
|
"limit": limit,
|
|
473
505
|
"offset": offset,
|
|
474
506
|
}
|
|
@@ -12,6 +12,30 @@ def _extract_assemblies(file_list: list | None) -> list[str]:
|
|
|
12
12
|
return sorted(set(f.get("assembly", "") for f in file_list if isinstance(f, dict) and f.get("assembly")))
|
|
13
13
|
|
|
14
14
|
|
|
15
|
+
def _experiment_assemblies(data: dict, file_list: list) -> list[str]:
|
|
16
|
+
"""Assemblies of an experiment: its own ``assembly`` list, else those of its embedded files."""
|
|
17
|
+
own = data.get("assembly")
|
|
18
|
+
if isinstance(own, list) and own:
|
|
19
|
+
return sorted({name for name in own if isinstance(name, str) and name})
|
|
20
|
+
return _extract_assemblies(file_list)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _experiment_organism(data: dict) -> str:
|
|
24
|
+
"""Organism of an experiment. Search results carry it on the replicates' biosamples."""
|
|
25
|
+
organism = data.get("organism")
|
|
26
|
+
if isinstance(organism, dict) and organism.get("scientific_name"):
|
|
27
|
+
return organism["scientific_name"]
|
|
28
|
+
names: list[str] = []
|
|
29
|
+
for replicate in data.get("replicates") or []:
|
|
30
|
+
if not isinstance(replicate, dict):
|
|
31
|
+
continue
|
|
32
|
+
biosample = (replicate.get("library") or {}).get("biosample") or {}
|
|
33
|
+
name = (biosample.get("organism") or {}).get("scientific_name", "") if isinstance(biosample, dict) else ""
|
|
34
|
+
if name and name not in names:
|
|
35
|
+
names.append(name)
|
|
36
|
+
return ", ".join(names)
|
|
37
|
+
|
|
38
|
+
|
|
15
39
|
def _extract_audit_counts(data: dict) -> dict[str, int]:
|
|
16
40
|
"""Extract all 4 ENCODE audit level counts from an experiment dict.
|
|
17
41
|
|
|
@@ -57,7 +81,7 @@ class ExperimentSummary(BaseModel):
|
|
|
57
81
|
|
|
58
82
|
@classmethod
|
|
59
83
|
def from_api(cls, data: dict) -> ExperimentSummary:
|
|
60
|
-
"""Parse
|
|
84
|
+
"""Parse a search hit (the fields EXPERIMENT_SEARCH_FIELDS asks for) or an embedded object."""
|
|
61
85
|
# Extract target label
|
|
62
86
|
target = ""
|
|
63
87
|
if data.get("target"):
|
|
@@ -90,7 +114,7 @@ class ExperimentSummary(BaseModel):
|
|
|
90
114
|
biosample_type = ont
|
|
91
115
|
|
|
92
116
|
file_list = data.get("files") or []
|
|
93
|
-
assemblies =
|
|
117
|
+
assemblies = _experiment_assemblies(data, file_list)
|
|
94
118
|
audit_counts = _extract_audit_counts(data)
|
|
95
119
|
|
|
96
120
|
# Extract dbxrefs (GEO accessions, etc.)
|
|
@@ -103,9 +127,7 @@ class ExperimentSummary(BaseModel):
|
|
|
103
127
|
assay_title=data.get("assay_title", ""),
|
|
104
128
|
target=target,
|
|
105
129
|
biosample_summary=data.get("biosample_summary", ""),
|
|
106
|
-
organism=data
|
|
107
|
-
if isinstance(data.get("organism"), dict)
|
|
108
|
-
else "",
|
|
130
|
+
organism=_experiment_organism(data),
|
|
109
131
|
organ=organ,
|
|
110
132
|
biosample_type=biosample_type,
|
|
111
133
|
status=data.get("status", ""),
|
|
@@ -23,6 +23,21 @@ logger = logging.getLogger(__name__)
|
|
|
23
23
|
|
|
24
24
|
DEFAULT_DB_PATH = Path.home() / ".encode_connector" / "tracker.db"
|
|
25
25
|
|
|
26
|
+
ASSEMBLY_SEPARATOR = ", "
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _assembly_text(assembly: str | list[str] | None) -> str:
|
|
30
|
+
"""Return the assembly value as the text the assembly column stores.
|
|
31
|
+
|
|
32
|
+
Experiment models carry ``assembly`` as a list (an experiment can have files on several
|
|
33
|
+
assemblies), and SQLite cannot bind a list, so the names are joined into one string.
|
|
34
|
+
"""
|
|
35
|
+
if not assembly:
|
|
36
|
+
return ""
|
|
37
|
+
if isinstance(assembly, str):
|
|
38
|
+
return assembly
|
|
39
|
+
return ASSEMBLY_SEPARATOR.join(assembly)
|
|
40
|
+
|
|
26
41
|
|
|
27
42
|
class ExperimentTracker:
|
|
28
43
|
"""SQLite-backed experiment tracker for ENCODE data."""
|
|
@@ -204,7 +219,7 @@ class ExperimentTracker:
|
|
|
204
219
|
experiment_data.get("description", ""),
|
|
205
220
|
experiment_data.get("lab", ""),
|
|
206
221
|
experiment_data.get("award", ""),
|
|
207
|
-
experiment_data.get("assembly"
|
|
222
|
+
_assembly_text(experiment_data.get("assembly")),
|
|
208
223
|
experiment_data.get("replication_type", ""),
|
|
209
224
|
experiment_data.get("life_stage", ""),
|
|
210
225
|
experiment_data.get("url", ""),
|
|
@@ -237,7 +252,7 @@ class ExperimentTracker:
|
|
|
237
252
|
experiment_data.get("description", ""),
|
|
238
253
|
experiment_data.get("lab", ""),
|
|
239
254
|
experiment_data.get("award", ""),
|
|
240
|
-
experiment_data.get("assembly"
|
|
255
|
+
_assembly_text(experiment_data.get("assembly")),
|
|
241
256
|
experiment_data.get("replication_type", ""),
|
|
242
257
|
experiment_data.get("life_stage", ""),
|
|
243
258
|
experiment_data.get("url", ""),
|
|
@@ -332,6 +347,17 @@ class ExperimentTracker:
|
|
|
332
347
|
conn = self._get_conn()
|
|
333
348
|
count = 0
|
|
334
349
|
for pub in publications:
|
|
350
|
+
# The table is unique on (experiment, pmid). ENCODE lists some papers without a
|
|
351
|
+
# PMID; stored as "" they would all replace each other, so they are stored as NULL
|
|
352
|
+
# (which SQLite never treats as equal) and matched on DOI and title instead.
|
|
353
|
+
pmid = pub.get("pmid") or None
|
|
354
|
+
if pmid is None:
|
|
355
|
+
conn.execute(
|
|
356
|
+
# older versions stored a missing PMID as "": replace such a row too
|
|
357
|
+
"DELETE FROM publications WHERE experiment_accession = ? "
|
|
358
|
+
"AND (pmid IS NULL OR pmid = '') AND doi = ? AND title = ?",
|
|
359
|
+
(accession, pub.get("doi", ""), pub.get("title", "")),
|
|
360
|
+
)
|
|
335
361
|
try:
|
|
336
362
|
conn.execute(
|
|
337
363
|
"""
|
|
@@ -341,7 +367,7 @@ class ExperimentTracker:
|
|
|
341
367
|
""",
|
|
342
368
|
(
|
|
343
369
|
accession,
|
|
344
|
-
|
|
370
|
+
pmid,
|
|
345
371
|
pub.get("doi", ""),
|
|
346
372
|
pub.get("title", ""),
|
|
347
373
|
pub.get("authors", ""),
|
|
@@ -363,7 +389,8 @@ class ExperimentTracker:
|
|
|
363
389
|
"SELECT * FROM publications WHERE experiment_accession = ?",
|
|
364
390
|
(accession,),
|
|
365
391
|
).fetchall()
|
|
366
|
-
|
|
392
|
+
# a missing PMID is stored as NULL; callers keep getting the empty string they always got
|
|
393
|
+
return [{**dict(r), "pmid": r["pmid"] or ""} for r in rows]
|
|
367
394
|
|
|
368
395
|
# ------------------------------------------------------------------
|
|
369
396
|
# Pipeline info
|
|
@@ -584,15 +611,19 @@ class ExperimentTracker:
|
|
|
584
611
|
else:
|
|
585
612
|
compatible_aspects.append(f"Same organism: {exp1['organism']}")
|
|
586
613
|
|
|
587
|
-
# Check assembly
|
|
614
|
+
# Check assembly. An experiment can have files on several assemblies, stored as one
|
|
615
|
+
# joined string, so two experiments are comparable when they share at least one.
|
|
588
616
|
if exp1.get("assembly") and exp2.get("assembly"):
|
|
589
|
-
|
|
617
|
+
shared = sorted(
|
|
618
|
+
set(exp1["assembly"].split(ASSEMBLY_SEPARATOR)) & set(exp2["assembly"].split(ASSEMBLY_SEPARATOR))
|
|
619
|
+
)
|
|
620
|
+
if not shared:
|
|
590
621
|
issues.append(
|
|
591
622
|
f"Different genome assemblies: {exp1['assembly']} vs {exp2['assembly']}. "
|
|
592
623
|
"Coordinate liftover needed before comparison."
|
|
593
624
|
)
|
|
594
625
|
else:
|
|
595
|
-
compatible_aspects.append(f"Same assembly: {
|
|
626
|
+
compatible_aspects.append(f"Same assembly: {ASSEMBLY_SEPARATOR.join(shared)}")
|
|
596
627
|
|
|
597
628
|
# Check assay type
|
|
598
629
|
if exp1.get("assay_title") and exp2.get("assay_title"):
|
|
@@ -698,7 +729,8 @@ class ExperimentTracker:
|
|
|
698
729
|
if pub.get("title"):
|
|
699
730
|
entry += f" title = {{{pub['title']}}},\n"
|
|
700
731
|
if pub.get("authors"):
|
|
701
|
-
|
|
732
|
+
# stored as "A, B, C"; BibTeX separates names with " and "
|
|
733
|
+
entry += f" author = {{{' and '.join(pub['authors'].split(', '))}}},\n"
|
|
702
734
|
if pub.get("journal"):
|
|
703
735
|
entry += f" journal = {{{pub['journal']}}},\n"
|
|
704
736
|
if pub.get("year"):
|
|
@@ -755,8 +787,14 @@ class ExperimentTracker:
|
|
|
755
787
|
# ------------------------------------------------------------------
|
|
756
788
|
|
|
757
789
|
def get_metadata_table(self, accessions: list[str] | None = None) -> list[dict]:
|
|
758
|
-
"""Get a metadata table of tracked experiments for analysis.
|
|
790
|
+
"""Get a metadata table of tracked experiments for analysis.
|
|
791
|
+
|
|
792
|
+
``None`` selects every tracked experiment; a list selects those accessions, so an empty
|
|
793
|
+
list (a filter that matched nothing) selects nothing.
|
|
794
|
+
"""
|
|
759
795
|
conn = self._get_conn()
|
|
796
|
+
if accessions is not None and not accessions:
|
|
797
|
+
return []
|
|
760
798
|
if accessions:
|
|
761
799
|
placeholders = ",".join("?" for _ in accessions)
|
|
762
800
|
rows = conn.execute(
|
|
@@ -891,7 +929,7 @@ class ExperimentTracker:
|
|
|
891
929
|
organ=organ,
|
|
892
930
|
)
|
|
893
931
|
|
|
894
|
-
table = self.get_metadata_table([e["accession"] for e in experiments]
|
|
932
|
+
table = self.get_metadata_table([e["accession"] for e in experiments])
|
|
895
933
|
|
|
896
934
|
# Enrich with external reference counts and PMIDs
|
|
897
935
|
conn = self._get_conn()
|
encode_connector/server/main.py
CHANGED
|
@@ -273,6 +273,7 @@ async def encode_search_experiments(
|
|
|
273
273
|
"""
|
|
274
274
|
client = await _get_client()
|
|
275
275
|
limit = clamp_limit(limit)
|
|
276
|
+
offset = max(0, offset)
|
|
276
277
|
|
|
277
278
|
filter_warnings = _validate_filters(assay_title, organ, biosample_type)
|
|
278
279
|
|
|
@@ -321,11 +322,13 @@ async def encode_search_experiments(
|
|
|
321
322
|
async def encode_get_experiment(accession: str) -> str:
|
|
322
323
|
"""Get full details for a specific ENCODE experiment by accession ID.
|
|
323
324
|
|
|
324
|
-
Returns
|
|
325
|
-
|
|
325
|
+
Returns the experiment's metadata, all associated files, the accessions of its possible
|
|
326
|
+
controls, replicate counts, and the number of audit flags at each level (ERROR,
|
|
327
|
+
NOT_COMPLIANT, WARNING, INTERNAL_ACTION). It does not return QC metric values such as
|
|
328
|
+
FRiP or NSC; those are on the experiment's page at encodeproject.org.
|
|
326
329
|
|
|
327
330
|
WHEN TO USE: Use when you have a specific accession and need full details
|
|
328
|
-
including files,
|
|
331
|
+
including files, controls, and audit counts.
|
|
329
332
|
RELATED TOOLS: encode_list_files, encode_track_experiment, encode_compare_experiments
|
|
330
333
|
|
|
331
334
|
Args:
|
|
@@ -471,6 +474,7 @@ async def encode_search_files(
|
|
|
471
474
|
"""
|
|
472
475
|
client = await _get_client()
|
|
473
476
|
limit = clamp_limit(limit)
|
|
477
|
+
offset = max(0, offset)
|
|
474
478
|
filter_warnings = _validate_filters(assay_title, organ, biosample_type)
|
|
475
479
|
result = await client.search_files(
|
|
476
480
|
file_format=file_format,
|
|
@@ -648,6 +652,7 @@ async def encode_batch_download(
|
|
|
648
652
|
verify_md5: bool = True,
|
|
649
653
|
limit: int = 100,
|
|
650
654
|
dry_run: bool = True,
|
|
655
|
+
offset: int = 0,
|
|
651
656
|
) -> str:
|
|
652
657
|
"""Search for files and download them all in batch.
|
|
653
658
|
|
|
@@ -686,6 +691,8 @@ async def encode_batch_download(
|
|
|
686
691
|
verify_md5: Verify downloads with MD5 checksums (default True)
|
|
687
692
|
limit: Max files to download (default 100, safety limit)
|
|
688
693
|
dry_run: If True (default), only preview what would be downloaded. Set False to download.
|
|
694
|
+
offset: Skip the first N matching files; pass the next_offset of the previous reply to
|
|
695
|
+
continue a search that has more files than limit
|
|
689
696
|
|
|
690
697
|
Returns:
|
|
691
698
|
JSON with download preview (dry_run=True) or download results (dry_run=False).
|
|
@@ -694,6 +701,7 @@ async def encode_batch_download(
|
|
|
694
701
|
downloader = _get_downloader()
|
|
695
702
|
validate_organize_by(organize_by)
|
|
696
703
|
limit = clamp_limit(limit)
|
|
704
|
+
offset = max(0, offset)
|
|
697
705
|
filter_warnings = _validate_filters(assay_title, organ, biosample_type)
|
|
698
706
|
|
|
699
707
|
# Search for files
|
|
@@ -710,18 +718,34 @@ async def encode_batch_download(
|
|
|
710
718
|
status="released",
|
|
711
719
|
preferred_default=preferred_default,
|
|
712
720
|
limit=limit,
|
|
721
|
+
offset=offset,
|
|
713
722
|
)
|
|
714
723
|
|
|
715
724
|
files = search_result["results"]
|
|
725
|
+
# the file search may stop before it has read every experiment; keep its note in every reply
|
|
726
|
+
total_note = search_result.get("total_note")
|
|
716
727
|
|
|
717
728
|
if not files:
|
|
729
|
+
# with an offset, an empty page means "past the last match", not "nothing matches"
|
|
730
|
+
search_total = search_result.get("total", 0)
|
|
731
|
+
if search_total:
|
|
732
|
+
message = f"No files at offset {offset}: the search matches {search_total} file(s). Use a smaller offset."
|
|
733
|
+
suggestion = "Call again with offset=0 to start from the first matching file."
|
|
734
|
+
else:
|
|
735
|
+
message = "No files found matching the search criteria."
|
|
736
|
+
suggestion = (
|
|
737
|
+
"Try broadening your search filters. Use encode_get_facets to see what data is "
|
|
738
|
+
"available for your criteria."
|
|
739
|
+
)
|
|
718
740
|
empty_result = {
|
|
719
|
-
"message":
|
|
720
|
-
"total":
|
|
741
|
+
"message": message,
|
|
742
|
+
"total": search_total,
|
|
721
743
|
"has_more": False,
|
|
722
744
|
"next_offset": None,
|
|
723
|
-
"suggestion":
|
|
745
|
+
"suggestion": suggestion,
|
|
724
746
|
}
|
|
747
|
+
if total_note:
|
|
748
|
+
empty_result["total_note"] = total_note
|
|
725
749
|
if filter_warnings:
|
|
726
750
|
empty_result["filter_warnings"] = filter_warnings
|
|
727
751
|
return json.dumps(empty_result, indent=2)
|
|
@@ -734,8 +758,10 @@ async def encode_batch_download(
|
|
|
734
758
|
f"Found {preview['file_count']} files ({preview['total_size_human']}). Set dry_run=False to download."
|
|
735
759
|
)
|
|
736
760
|
preview["search_total"] = search_total
|
|
737
|
-
preview["has_more"] = search_total > limit
|
|
738
|
-
preview["next_offset"] = limit if search_total > limit else None
|
|
761
|
+
preview["has_more"] = search_total > offset + limit
|
|
762
|
+
preview["next_offset"] = offset + limit if search_total > offset + limit else None
|
|
763
|
+
if total_note:
|
|
764
|
+
preview["total_note"] = total_note
|
|
739
765
|
if filter_warnings:
|
|
740
766
|
preview["filter_warnings"] = filter_warnings
|
|
741
767
|
return json.dumps(_serialize(preview), indent=2)
|
|
@@ -754,9 +780,11 @@ async def encode_batch_download(
|
|
|
754
780
|
"total_size": sum(r.file_size for r in results if r.success),
|
|
755
781
|
"total_size_human": _human_size(sum(r.file_size for r in results if r.success)),
|
|
756
782
|
},
|
|
757
|
-
"has_more": search_total > limit,
|
|
758
|
-
"next_offset": limit if search_total > limit else None,
|
|
783
|
+
"has_more": search_total > offset + limit,
|
|
784
|
+
"next_offset": offset + limit if search_total > offset + limit else None,
|
|
759
785
|
}
|
|
786
|
+
if total_note:
|
|
787
|
+
output["total_note"] = total_note
|
|
760
788
|
if filter_warnings:
|
|
761
789
|
output["filter_warnings"] = filter_warnings
|
|
762
790
|
return json.dumps(output, indent=2)
|
|
@@ -1124,7 +1152,7 @@ async def encode_list_tracked(
|
|
|
1124
1152
|
)
|
|
1125
1153
|
|
|
1126
1154
|
# Build metadata table
|
|
1127
|
-
table = tracker.get_metadata_table([e["accession"] for e in experiments]
|
|
1155
|
+
table = tracker.get_metadata_table([e["accession"] for e in experiments])
|
|
1128
1156
|
|
|
1129
1157
|
# Remove raw_metadata from output
|
|
1130
1158
|
for row in table:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: encode-toolkit
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.4
|
|
4
4
|
Summary: MCP server for querying and downloading ENCODE Project genomics data directly from Claude
|
|
5
5
|
Project-URL: Homepage, https://github.com/ammawla/encode-toolkit
|
|
6
6
|
Project-URL: Repository, https://github.com/ammawla/encode-toolkit
|
|
@@ -37,7 +37,7 @@ Description-Content-Type: text/markdown
|
|
|
37
37
|
|
|
38
38
|
[](LICENSE)
|
|
39
39
|
[](https://www.python.org/downloads/)
|
|
40
|
-
[](CHANGELOG.md)
|
|
41
41
|
[]()
|
|
42
42
|
[](docs/skill-vignettes/)
|
|
43
43
|
[](src/encode_connector/server/main.py)
|
|
@@ -331,7 +331,7 @@ Search ENCODE experiments with 20+ filters.
|
|
|
331
331
|
|
|
332
332
|
| Parameter | Type | Description |
|
|
333
333
|
|-----------|------|-------------|
|
|
334
|
-
| `assay_title` | string | Assay type: "Histone ChIP-seq", "ATAC-seq", "RNA-seq", "Hi-C", etc. |
|
|
334
|
+
| `assay_title` | string | Assay type: "Histone ChIP-seq", "ATAC-seq", "total RNA-seq", "Hi-C", etc. |
|
|
335
335
|
| `organism` | string | Species (default: "Homo sapiens") |
|
|
336
336
|
| `organ` | string | Organ: "pancreas", "brain", "liver", "heart", "kidney", etc. |
|
|
337
337
|
| `biosample_type` | string | "tissue", "cell line", "primary cell", "organoid" |
|
|
@@ -496,7 +496,7 @@ Get grouped statistics of your tracked experiment collection.
|
|
|
496
496
|
</details>
|
|
497
497
|
|
|
498
498
|
<details>
|
|
499
|
-
<summary><strong>Provenance and export tools (
|
|
499
|
+
<summary><strong>Provenance and export tools (5)</strong></summary>
|
|
500
500
|
|
|
501
501
|
### `encode_log_derived_file`
|
|
502
502
|
|
|
@@ -624,7 +624,7 @@ When installed as a Claude Code plugin, ENCODE Toolkit includes 47 literature-ba
|
|
|
624
624
|
</details>
|
|
625
625
|
|
|
626
626
|
<details>
|
|
627
|
-
<summary><strong>Workflow skills (
|
|
627
|
+
<summary><strong>Workflow skills (10)</strong></summary>
|
|
628
628
|
|
|
629
629
|
| Skill | Description |
|
|
630
630
|
|-------|-------------|
|
|
@@ -681,7 +681,7 @@ Each pipeline includes a SKILL.md overview, 5-stage reference files (preprocessi
|
|
|
681
681
|
| File | Description |
|
|
682
682
|
|-------|-------------|
|
|
683
683
|
| `skills/histone-aggregation/references/histone-marks-reference.md` | Comprehensive chromatin biology catalog (1,442 lines) — 21 histone marks with writers/erasers/readers, 5 novel acylation marks, ChromHMM state models (5 to 51 states), TF co-binding patterns, chromatin remodeling complexes, DNA methylation-chromatin interplay, nucleosome dynamics, 3D genome organization, chromatin in disease. 74 primary references |
|
|
684
|
-
| `skills/*/references/literature.md` |
|
|
684
|
+
| `skills/*/references/literature.md` | 47 per-skill literature reference documents — about 240 papers cataloged with DOI, PMID, citation counts, and skill-relevant key findings |
|
|
685
685
|
|
|
686
686
|
</details>
|
|
687
687
|
|
|
@@ -710,12 +710,12 @@ Most genomics tools give you one thing. ENCODE Toolkit gives you the full resear
|
|
|
710
710
|
| Category | Assays |
|
|
711
711
|
|----------|--------|
|
|
712
712
|
| **Histone/Chromatin** | Histone ChIP-seq, TF ChIP-seq, ATAC-seq, DNase-seq, CUT&RUN, CUT&Tag, MNase-seq |
|
|
713
|
-
| **Transcription** | RNA-seq,
|
|
713
|
+
| **Transcription** | total RNA-seq, polyA plus RNA-seq, small RNA-seq, long read RNA-seq, CAGE, RAMPAGE, PRO-seq, GRO-seq |
|
|
714
714
|
| **3D Genome** | Hi-C, intact Hi-C, Micro-C, ChIA-PET, HiChIP, PLAC-seq, 5C |
|
|
715
715
|
| **DNA Methylation** | WGBS, RRBS, MeDIP-seq, MRE-seq |
|
|
716
716
|
| **Functional** | STARR-seq, MPRA, CRISPR screen, eCLIP, iCLIP |
|
|
717
|
-
| **Single Cell** | scRNA-seq, snATAC-seq,
|
|
718
|
-
| **Perturbation** | CRISPRi
|
|
717
|
+
| **Single Cell** | scRNA-seq, snATAC-seq, snRNA-seq, long read scRNA-seq |
|
|
718
|
+
| **Perturbation** | CRISPRi RNA-seq, shRNA RNA-seq, siRNA RNA-seq, CRISPR RNA-seq |
|
|
719
719
|
|
|
720
720
|
**Supported file formats**: `fastq` `bam` `bed` `bigWig` `bigBed` `tsv` `csv` `hic` `tagAlign` `bedpe` `pairs` `fasta` `vcf` `tar`
|
|
721
721
|
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
encode_connector/__init__.py,sha256=m2jzlw8g0WFGPmR0OLkpNcOo5ao8U4Bwzp9g2nQnFac,311
|
|
2
|
+
encode_connector/__main__.py,sha256=hX7Hv1cKkwjnZrQkgsV-Kkx-60ywaNmobOuLuRfLZ-o,106
|
|
3
|
+
encode_connector/client/__init__.py,sha256=wyIOQXMUZoUqGdv8_h5lW7vfWZ6Ohq_ttz2aLN7Mvo8,216
|
|
4
|
+
encode_connector/client/auth.py,sha256=SnQZ3Hkq7BhpQD6vBZudAi6Lr2zCB47WXEWeFy_4xZg,9576
|
|
5
|
+
encode_connector/client/constants.py,sha256=8l5kUaNPEoVC0bDbQW4IaTEQJfsAfit8IVqX-ZUHmUo,11259
|
|
6
|
+
encode_connector/client/downloader.py,sha256=hDPLkGs8MO3Jr5gxJnPxhl211I5iZkoytZd7m7EeTFk,11877
|
|
7
|
+
encode_connector/client/encode_client.py,sha256=yom0wAPQogRMF0euLN-ItczgRjDleLLMfF5zqqaROPM,23303
|
|
8
|
+
encode_connector/client/models.py,sha256=lkt-2lr0dqBCPDBlSb7ozvSvvrlp-0JH7KtSpyjZil8,14830
|
|
9
|
+
encode_connector/client/tracker.py,sha256=ts_B4fOuGtL4Xz1GrCph7XyPjfcyeUNGJRO_rFOyAIs,46777
|
|
10
|
+
encode_connector/client/validation.py,sha256=vssV9tu9iTQes98jxCUvfSNsABtn_o8AzGKEJ2RqWf0,7814
|
|
11
|
+
encode_connector/server/__init__.py,sha256=qE1SX7iMPz3bwjVuuvMlXNN1g_iHB-sIUQY-LtGgAPw,33
|
|
12
|
+
encode_connector/server/__main__.py,sha256=gVilW3ql6i1R9fJ34xUi5RmGFDy2E_fyy9j0wcQ3tUg,124
|
|
13
|
+
encode_connector/server/main.py,sha256=dWFNjaX-eNbKeV4BwcbycA-0fkftypmkRtUWHl2G72I,59200
|
|
14
|
+
encode_toolkit-0.3.4.dist-info/METADATA,sha256=c7VqORxOgAQTex5aSHzeBmrFAtm7lyFZgSX1yDkmN-c,34459
|
|
15
|
+
encode_toolkit-0.3.4.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
16
|
+
encode_toolkit-0.3.4.dist-info/entry_points.txt,sha256=ZNqGYIpnig7ZLZKSP1xZTRZ1FdTSRLt8BcY9cqmCOW8,69
|
|
17
|
+
encode_toolkit-0.3.4.dist-info/licenses/LICENSE,sha256=hIahDEOTzuHCU5J2nd07LWwkLW7Hko4UFO__ffsvB-8,34523
|
|
18
|
+
encode_toolkit-0.3.4.dist-info/RECORD,,
|
|
@@ -1,18 +0,0 @@
|
|
|
1
|
-
encode_connector/__init__.py,sha256=A1xhCKU_XGnBkePtv1NmXWFX-A0WTcm62YQfMgaYwE8,124
|
|
2
|
-
encode_connector/__main__.py,sha256=hX7Hv1cKkwjnZrQkgsV-Kkx-60ywaNmobOuLuRfLZ-o,106
|
|
3
|
-
encode_connector/client/__init__.py,sha256=wyIOQXMUZoUqGdv8_h5lW7vfWZ6Ohq_ttz2aLN7Mvo8,216
|
|
4
|
-
encode_connector/client/auth.py,sha256=SnQZ3Hkq7BhpQD6vBZudAi6Lr2zCB47WXEWeFy_4xZg,9576
|
|
5
|
-
encode_connector/client/constants.py,sha256=x3XWhgHICZv_osWAyQrvR8GiYA8byZ3a9F0qscZ8GKc,9709
|
|
6
|
-
encode_connector/client/downloader.py,sha256=hDPLkGs8MO3Jr5gxJnPxhl211I5iZkoytZd7m7EeTFk,11877
|
|
7
|
-
encode_connector/client/encode_client.py,sha256=PiSGepIWoaxryX1YhcYfLTuKhJxrm4fgACmZNglAUs8,21314
|
|
8
|
-
encode_connector/client/models.py,sha256=hNukcSisnM2Dw9xms-trt72Ymj1H1_yhi2iAxn3jfn0,13804
|
|
9
|
-
encode_connector/client/tracker.py,sha256=8x6JNlla6UIpzgVDj8HOGnssfjQPRp-WdpmiirjLDI8,44804
|
|
10
|
-
encode_connector/client/validation.py,sha256=vssV9tu9iTQes98jxCUvfSNsABtn_o8AzGKEJ2RqWf0,7814
|
|
11
|
-
encode_connector/server/__init__.py,sha256=qE1SX7iMPz3bwjVuuvMlXNN1g_iHB-sIUQY-LtGgAPw,33
|
|
12
|
-
encode_connector/server/__main__.py,sha256=gVilW3ql6i1R9fJ34xUi5RmGFDy2E_fyy9j0wcQ3tUg,124
|
|
13
|
-
encode_connector/server/main.py,sha256=2oYZKnDM_XxW-z_NOXxlGdGOmauw5NuQdu7d3-aAbow,57826
|
|
14
|
-
encode_toolkit-0.3.2.dist-info/METADATA,sha256=rZIBBJ6Yazo_ogyJb6tNCO9raY15xjMCrUgHfFIPdiY,34436
|
|
15
|
-
encode_toolkit-0.3.2.dist-info/WHEEL,sha256=THafob7ofN-NsuMN7Mg4qZyHaQI7KkD-QlcQatYhXPo,87
|
|
16
|
-
encode_toolkit-0.3.2.dist-info/entry_points.txt,sha256=ZNqGYIpnig7ZLZKSP1xZTRZ1FdTSRLt8BcY9cqmCOW8,69
|
|
17
|
-
encode_toolkit-0.3.2.dist-info/licenses/LICENSE,sha256=hIahDEOTzuHCU5J2nd07LWwkLW7Hko4UFO__ffsvB-8,34523
|
|
18
|
-
encode_toolkit-0.3.2.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|