encode-toolkit 0.3.2__py3-none-any.whl → 0.3.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,9 @@
1
1
  """ENCODE Project connector - MCP server and Python client."""
2
2
 
3
- __version__ = "0.2.1"
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+ try:
6
+ __version__ = version("encode-toolkit")
7
+ except PackageNotFoundError: # a source tree that is not installed
8
+ __version__ = "0+unknown"
4
9
  __author__ = "Dr. Alex M. Mawla, PhD"
@@ -12,12 +12,21 @@ DOWNLOAD_CONCURRENCY = 3
12
12
  DEFAULT_TIMEOUT = 30.0
13
13
  DOWNLOAD_TIMEOUT = 300.0
14
14
  DEFAULT_LIMIT = 25
15
+ # Frame for a single experiment. "page" is "embedded" plus the "audit" property; with
16
+ # "embedded" ENCODE omits the audits and every experiment looks free of errors and warnings.
17
+ EXPERIMENT_FRAME = "page"
18
+ # experiments read per request while a file search walks the experiments of one organism
19
+ EXPERIMENT_PAGE_SIZE = 200
20
+ # ...and how many experiments it reads at most: each one costs a request to ENCODE
21
+ MAX_EXPERIMENTS_SCANNED = 1000
22
+ # files of one experiment read per request during that walk
23
+ FILES_PAGE_SIZE = 200
15
24
  try:
16
25
  import importlib.metadata
17
26
 
18
27
  _version = importlib.metadata.version("encode-toolkit")
19
28
  except importlib.metadata.PackageNotFoundError:
20
- _version = "0.3.2"
29
+ _version = "0.3.4"
21
30
  USER_AGENT = f"encode-toolkit/{_version} (MCP; +https://github.com/ammawla/encode-toolkit)"
22
31
 
23
32
  # Keyring service name for credential storage
@@ -239,8 +248,18 @@ FILE_FORMATS = [
239
248
  "vcf",
240
249
  "bigInteract",
241
250
  "idx",
242
- "dat",
243
251
  "txt",
252
+ "h5ad",
253
+ "hdf5",
254
+ "sam",
255
+ "wig",
256
+ "starch",
257
+ "chain",
258
+ "PWM",
259
+ "btr",
260
+ "cndb",
261
+ "nucle3d",
262
+ "yaml",
244
263
  ]
245
264
 
246
265
  OUTPUT_TYPES = [
@@ -277,7 +296,6 @@ OUTPUT_TYPES = [
277
296
  "pseudoreplicated peaks",
278
297
  "pseudoreplicated IDR thresholded peaks",
279
298
  "replicated peaks",
280
- "stable peaks",
281
299
  "hotspots",
282
300
  "footprints",
283
301
  "peaks and background as input for IDR",
@@ -289,6 +307,10 @@ OUTPUT_TYPES = [
289
307
  "filtered peaks",
290
308
  # Quantifications
291
309
  "gene quantifications",
310
+ "sparse gene count matrix of unique reads",
311
+ "sparse gene count matrix of all reads",
312
+ "unfiltered sparse gene count matrix of unique reads",
313
+ "unfiltered sparse gene count matrix of all reads",
292
314
  "transcript quantifications",
293
315
  "exon quantifications",
294
316
  "microRNA quantifications",
@@ -347,7 +369,6 @@ OUTPUT_CATEGORIES = [
347
369
  "annotation",
348
370
  "quantification",
349
371
  "reference",
350
- "quality metric",
351
372
  ]
352
373
 
353
374
  FILE_STATUSES = [
@@ -381,6 +402,14 @@ ASSEMBLIES = [
381
402
  "dm3",
382
403
  "ce11",
383
404
  "ce10",
405
+ "GRCh38-minimal",
406
+ "mm10-minimal",
407
+ "T2T-CHM13",
408
+ "J02459.1",
409
+ "ENC001.1",
410
+ "ENC002.1",
411
+ "ENC003.1",
412
+ "ENC004.1",
384
413
  ]
385
414
 
386
415
  LIFE_STAGES = [
@@ -416,6 +445,28 @@ METADATA_MAP = {
416
445
  }
417
446
 
418
447
  # ENCODE API parameter name mapping (user-friendly -> API param)
448
+ # Fields requested from /search/?type=Experiment. They are what ExperimentSummary.from_api
449
+ # reads; naming them makes the API embed labels instead of returning object paths.
450
+ EXPERIMENT_SEARCH_FIELDS = (
451
+ "accession",
452
+ "assay_title",
453
+ "target.label",
454
+ "biosample_summary",
455
+ "biosample_ontology.classification",
456
+ "biosample_ontology.organ_slims",
457
+ "replicates.library.biosample.organism.scientific_name",
458
+ "status",
459
+ "date_released",
460
+ "description",
461
+ "lab.title",
462
+ "files.@id",
463
+ "replication_type",
464
+ "life_stage_age",
465
+ "assembly",
466
+ "audit",
467
+ "dbxrefs",
468
+ )
469
+
419
470
  EXPERIMENT_FILTER_MAP = {
420
471
  "assay_title": "assay_title",
421
472
  "organism": "replicates.library.biosample.donor.organism.scientific_name",
@@ -19,7 +19,12 @@ from encode_connector.client.constants import (
19
19
  DEFAULT_LIMIT,
20
20
  DEFAULT_TIMEOUT,
21
21
  EXPERIMENT_FILTER_MAP,
22
+ EXPERIMENT_FRAME,
23
+ EXPERIMENT_PAGE_SIZE,
24
+ EXPERIMENT_SEARCH_FIELDS,
22
25
  FILE_FILTER_MAP,
26
+ FILES_PAGE_SIZE,
27
+ MAX_EXPERIMENTS_SCANNED,
23
28
  MAX_REQUESTS_PER_SECOND,
24
29
  METADATA_MAP,
25
30
  USER_AGENT,
@@ -236,10 +241,14 @@ class EncodeClient:
236
241
  Returns dict with 'results' (list of ExperimentSummary) and 'total' count.
237
242
  """
238
243
  limit = clamp_limit(limit)
244
+ offset = max(0, offset)
245
+ # frame=object replaces linked objects with paths ("/targets/H3K4me1-human/"), which
246
+ # loses the labels the filters use, the organ, the organism and the assemblies. Naming
247
+ # the fields makes the API embed exactly those.
239
248
  params: dict[str, Any] = {
240
249
  "type": "Experiment",
241
250
  "format": "json",
242
- "frame": "object",
251
+ "field": list(EXPERIMENT_SEARCH_FIELDS),
243
252
  "limit": limit,
244
253
  }
245
254
  if offset > 0:
@@ -297,11 +306,11 @@ class EncodeClient:
297
306
  }
298
307
 
299
308
  async def get_experiment_raw(self, accession: str) -> dict:
300
- """Get raw experiment data with embedded frame."""
309
+ """Get raw experiment data: linked objects embedded, audits included."""
301
310
  validate_accession(accession)
302
311
  return await self._request(
303
312
  f"/experiments/{accession}/",
304
- {"format": "json", "frame": "embedded"},
313
+ {"format": "json", "frame": EXPERIMENT_FRAME},
305
314
  )
306
315
 
307
316
  async def get_experiment(self, accession: str) -> ExperimentDetail:
@@ -310,7 +319,7 @@ class EncodeClient:
310
319
  # Get experiment data
311
320
  exp_data = await self._request(
312
321
  f"/experiments/{accession}/",
313
- {"format": "json", "frame": "embedded"},
322
+ {"format": "json", "frame": EXPERIMENT_FRAME},
314
323
  )
315
324
 
316
325
  # Get files for this experiment
@@ -343,6 +352,7 @@ class EncodeClient:
343
352
  status: str | None = None,
344
353
  preferred_default: bool | None = None,
345
354
  limit: int = 200,
355
+ offset: int = 0,
346
356
  ) -> list[FileSummary]:
347
357
  """List files for a specific experiment with optional filters."""
348
358
  validate_accession(experiment_accession)
@@ -354,6 +364,8 @@ class EncodeClient:
354
364
  "frame": "object",
355
365
  "limit": limit,
356
366
  }
367
+ if offset > 0:
368
+ params["from"] = offset
357
369
 
358
370
  filter_values = {
359
371
  "file_format": file_format,
@@ -395,6 +407,7 @@ class EncodeClient:
395
407
  ) -> dict[str, Any]:
396
408
  """Search files across all experiments with combined filters."""
397
409
  limit = clamp_limit(limit)
410
+ offset = max(0, offset)
398
411
  params: dict[str, Any] = {
399
412
  "type": "File",
400
413
  "format": "json",
@@ -434,41 +447,60 @@ class EncodeClient:
434
447
  # For organism filtering on files, we need a two-step approach:
435
448
  # search experiments first, then get files from matching experiments
436
449
  if organism:
437
- # For non-human organisms, do a two-step search
438
- exp_result = await self.search_experiments(
439
- assay_title=assay_title,
440
- organism=organism,
441
- organ=organ,
442
- biosample_type=biosample_type,
443
- target=target,
444
- status=status or "released",
445
- search_term=search_term,
446
- limit=200,
447
- )
448
- if not exp_result["results"]:
449
- return {"results": [], "total": 0, "limit": limit, "offset": offset}
450
-
451
- # Get files from matching experiments
452
- all_files = []
453
- for exp in exp_result["results"]:
454
- exp_files = await self.list_files(
455
- experiment_accession=exp.accession,
456
- file_format=file_format,
457
- file_type=file_type,
458
- output_type=output_type,
459
- output_category=output_category,
460
- assembly=assembly,
461
- status=status,
462
- preferred_default=preferred_default,
450
+ # Collect one file past the requested page so the caller can tell that more exist.
451
+ # Matching files may belong to experiments far down the list, so keep reading
452
+ # experiment pages until the page is full or the experiments run out.
453
+ wanted = offset + limit + 1
454
+ all_files: list[FileSummary] = []
455
+ experiment_offset = 0
456
+ experiment_total = 0
457
+ while len(all_files) < wanted and experiment_offset < MAX_EXPERIMENTS_SCANNED:
458
+ exp_result = await self.search_experiments(
459
+ assay_title=assay_title,
460
+ organism=organism,
461
+ organ=organ,
462
+ biosample_type=biosample_type,
463
+ target=target,
464
+ status=status or "released",
465
+ search_term=search_term,
466
+ limit=EXPERIMENT_PAGE_SIZE,
467
+ offset=experiment_offset,
463
468
  )
464
- all_files.extend(exp_files)
465
- if len(all_files) >= limit:
469
+ experiments = exp_result["results"]
470
+ experiment_total = exp_result.get("total", 0)
471
+ for exp in experiments:
472
+ # one experiment can hold more files than a request returns
473
+ file_offset = 0
474
+ while len(all_files) < wanted:
475
+ exp_files = await self.list_files(
476
+ experiment_accession=exp.accession,
477
+ file_format=file_format,
478
+ file_type=file_type,
479
+ output_type=output_type,
480
+ output_category=output_category,
481
+ assembly=assembly,
482
+ status=status,
483
+ preferred_default=preferred_default,
484
+ limit=FILES_PAGE_SIZE,
485
+ offset=file_offset,
486
+ )
487
+ all_files.extend(exp_files)
488
+ file_offset += len(exp_files)
489
+ if len(exp_files) < FILES_PAGE_SIZE:
490
+ break
491
+ if len(all_files) >= wanted:
492
+ break
493
+ experiment_offset += len(experiments)
494
+ if not experiments or experiment_offset >= experiment_total:
466
495
  break
467
496
 
497
+ total_note = "Lower bound: files collected so far from matching experiments, not the full count"
498
+ if len(all_files) < wanted and experiment_total > experiment_offset >= MAX_EXPERIMENTS_SCANNED:
499
+ total_note += f"; stopped after reading {experiment_offset} of {experiment_total} experiments"
468
500
  return {
469
- "results": all_files[:limit],
501
+ "results": all_files[offset : offset + limit],
470
502
  "total": len(all_files),
471
- "total_note": "Approximate — based on files collected from matching experiments",
503
+ "total_note": total_note,
472
504
  "limit": limit,
473
505
  "offset": offset,
474
506
  }
@@ -12,6 +12,30 @@ def _extract_assemblies(file_list: list | None) -> list[str]:
12
12
  return sorted(set(f.get("assembly", "") for f in file_list if isinstance(f, dict) and f.get("assembly")))
13
13
 
14
14
 
15
+ def _experiment_assemblies(data: dict, file_list: list) -> list[str]:
16
+ """Assemblies of an experiment: its own ``assembly`` list, else those of its embedded files."""
17
+ own = data.get("assembly")
18
+ if isinstance(own, list) and own:
19
+ return sorted({name for name in own if isinstance(name, str) and name})
20
+ return _extract_assemblies(file_list)
21
+
22
+
23
+ def _experiment_organism(data: dict) -> str:
24
+ """Organism of an experiment. Search results carry it on the replicates' biosamples."""
25
+ organism = data.get("organism")
26
+ if isinstance(organism, dict) and organism.get("scientific_name"):
27
+ return organism["scientific_name"]
28
+ names: list[str] = []
29
+ for replicate in data.get("replicates") or []:
30
+ if not isinstance(replicate, dict):
31
+ continue
32
+ biosample = (replicate.get("library") or {}).get("biosample") or {}
33
+ name = (biosample.get("organism") or {}).get("scientific_name", "") if isinstance(biosample, dict) else ""
34
+ if name and name not in names:
35
+ names.append(name)
36
+ return ", ".join(names)
37
+
38
+
15
39
  def _extract_audit_counts(data: dict) -> dict[str, int]:
16
40
  """Extract all 4 ENCODE audit level counts from an experiment dict.
17
41
 
@@ -57,7 +81,7 @@ class ExperimentSummary(BaseModel):
57
81
 
58
82
  @classmethod
59
83
  def from_api(cls, data: dict) -> ExperimentSummary:
60
- """Parse from ENCODE API experiment object (frame=object)."""
84
+ """Parse a search hit (the fields EXPERIMENT_SEARCH_FIELDS asks for) or an embedded object."""
61
85
  # Extract target label
62
86
  target = ""
63
87
  if data.get("target"):
@@ -90,7 +114,7 @@ class ExperimentSummary(BaseModel):
90
114
  biosample_type = ont
91
115
 
92
116
  file_list = data.get("files") or []
93
- assemblies = _extract_assemblies(file_list)
117
+ assemblies = _experiment_assemblies(data, file_list)
94
118
  audit_counts = _extract_audit_counts(data)
95
119
 
96
120
  # Extract dbxrefs (GEO accessions, etc.)
@@ -103,9 +127,7 @@ class ExperimentSummary(BaseModel):
103
127
  assay_title=data.get("assay_title", ""),
104
128
  target=target,
105
129
  biosample_summary=data.get("biosample_summary", ""),
106
- organism=data.get("organism", {}).get("scientific_name", "")
107
- if isinstance(data.get("organism"), dict)
108
- else "",
130
+ organism=_experiment_organism(data),
109
131
  organ=organ,
110
132
  biosample_type=biosample_type,
111
133
  status=data.get("status", ""),
@@ -23,6 +23,21 @@ logger = logging.getLogger(__name__)
23
23
 
24
24
  DEFAULT_DB_PATH = Path.home() / ".encode_connector" / "tracker.db"
25
25
 
26
+ ASSEMBLY_SEPARATOR = ", "
27
+
28
+
29
+ def _assembly_text(assembly: str | list[str] | None) -> str:
30
+ """Return the assembly value as the text the assembly column stores.
31
+
32
+ Experiment models carry ``assembly`` as a list (an experiment can have files on several
33
+ assemblies), and SQLite cannot bind a list, so the names are joined into one string.
34
+ """
35
+ if not assembly:
36
+ return ""
37
+ if isinstance(assembly, str):
38
+ return assembly
39
+ return ASSEMBLY_SEPARATOR.join(assembly)
40
+
26
41
 
27
42
  class ExperimentTracker:
28
43
  """SQLite-backed experiment tracker for ENCODE data."""
@@ -204,7 +219,7 @@ class ExperimentTracker:
204
219
  experiment_data.get("description", ""),
205
220
  experiment_data.get("lab", ""),
206
221
  experiment_data.get("award", ""),
207
- experiment_data.get("assembly", ""),
222
+ _assembly_text(experiment_data.get("assembly")),
208
223
  experiment_data.get("replication_type", ""),
209
224
  experiment_data.get("life_stage", ""),
210
225
  experiment_data.get("url", ""),
@@ -237,7 +252,7 @@ class ExperimentTracker:
237
252
  experiment_data.get("description", ""),
238
253
  experiment_data.get("lab", ""),
239
254
  experiment_data.get("award", ""),
240
- experiment_data.get("assembly", ""),
255
+ _assembly_text(experiment_data.get("assembly")),
241
256
  experiment_data.get("replication_type", ""),
242
257
  experiment_data.get("life_stage", ""),
243
258
  experiment_data.get("url", ""),
@@ -332,6 +347,17 @@ class ExperimentTracker:
332
347
  conn = self._get_conn()
333
348
  count = 0
334
349
  for pub in publications:
350
+ # The table is unique on (experiment, pmid). ENCODE lists some papers without a
351
+ # PMID; stored as "" they would all replace each other, so they are stored as NULL
352
+ # (which SQLite never treats as equal) and matched on DOI and title instead.
353
+ pmid = pub.get("pmid") or None
354
+ if pmid is None:
355
+ conn.execute(
356
+ # older versions stored a missing PMID as "": replace such a row too
357
+ "DELETE FROM publications WHERE experiment_accession = ? "
358
+ "AND (pmid IS NULL OR pmid = '') AND doi = ? AND title = ?",
359
+ (accession, pub.get("doi", ""), pub.get("title", "")),
360
+ )
335
361
  try:
336
362
  conn.execute(
337
363
  """
@@ -341,7 +367,7 @@ class ExperimentTracker:
341
367
  """,
342
368
  (
343
369
  accession,
344
- pub.get("pmid", ""),
370
+ pmid,
345
371
  pub.get("doi", ""),
346
372
  pub.get("title", ""),
347
373
  pub.get("authors", ""),
@@ -363,7 +389,8 @@ class ExperimentTracker:
363
389
  "SELECT * FROM publications WHERE experiment_accession = ?",
364
390
  (accession,),
365
391
  ).fetchall()
366
- return [dict(r) for r in rows]
392
+ # a missing PMID is stored as NULL; callers keep getting the empty string they always got
393
+ return [{**dict(r), "pmid": r["pmid"] or ""} for r in rows]
367
394
 
368
395
  # ------------------------------------------------------------------
369
396
  # Pipeline info
@@ -584,15 +611,19 @@ class ExperimentTracker:
584
611
  else:
585
612
  compatible_aspects.append(f"Same organism: {exp1['organism']}")
586
613
 
587
- # Check assembly
614
+ # Check assembly. An experiment can have files on several assemblies, stored as one
615
+ # joined string, so two experiments are comparable when they share at least one.
588
616
  if exp1.get("assembly") and exp2.get("assembly"):
589
- if exp1["assembly"] != exp2["assembly"]:
617
+ shared = sorted(
618
+ set(exp1["assembly"].split(ASSEMBLY_SEPARATOR)) & set(exp2["assembly"].split(ASSEMBLY_SEPARATOR))
619
+ )
620
+ if not shared:
590
621
  issues.append(
591
622
  f"Different genome assemblies: {exp1['assembly']} vs {exp2['assembly']}. "
592
623
  "Coordinate liftover needed before comparison."
593
624
  )
594
625
  else:
595
- compatible_aspects.append(f"Same assembly: {exp1['assembly']}")
626
+ compatible_aspects.append(f"Same assembly: {ASSEMBLY_SEPARATOR.join(shared)}")
596
627
 
597
628
  # Check assay type
598
629
  if exp1.get("assay_title") and exp2.get("assay_title"):
@@ -698,7 +729,8 @@ class ExperimentTracker:
698
729
  if pub.get("title"):
699
730
  entry += f" title = {{{pub['title']}}},\n"
700
731
  if pub.get("authors"):
701
- entry += f" author = {{{pub['authors']}}},\n"
732
+ # stored as "A, B, C"; BibTeX separates names with " and "
733
+ entry += f" author = {{{' and '.join(pub['authors'].split(', '))}}},\n"
702
734
  if pub.get("journal"):
703
735
  entry += f" journal = {{{pub['journal']}}},\n"
704
736
  if pub.get("year"):
@@ -755,8 +787,14 @@ class ExperimentTracker:
755
787
  # ------------------------------------------------------------------
756
788
 
757
789
  def get_metadata_table(self, accessions: list[str] | None = None) -> list[dict]:
758
- """Get a metadata table of tracked experiments for analysis."""
790
+ """Get a metadata table of tracked experiments for analysis.
791
+
792
+ ``None`` selects every tracked experiment; a list selects those accessions, so an empty
793
+ list (a filter that matched nothing) selects nothing.
794
+ """
759
795
  conn = self._get_conn()
796
+ if accessions is not None and not accessions:
797
+ return []
760
798
  if accessions:
761
799
  placeholders = ",".join("?" for _ in accessions)
762
800
  rows = conn.execute(
@@ -891,7 +929,7 @@ class ExperimentTracker:
891
929
  organ=organ,
892
930
  )
893
931
 
894
- table = self.get_metadata_table([e["accession"] for e in experiments] if experiments else None)
932
+ table = self.get_metadata_table([e["accession"] for e in experiments])
895
933
 
896
934
  # Enrich with external reference counts and PMIDs
897
935
  conn = self._get_conn()
@@ -273,6 +273,7 @@ async def encode_search_experiments(
273
273
  """
274
274
  client = await _get_client()
275
275
  limit = clamp_limit(limit)
276
+ offset = max(0, offset)
276
277
 
277
278
  filter_warnings = _validate_filters(assay_title, organ, biosample_type)
278
279
 
@@ -321,11 +322,13 @@ async def encode_search_experiments(
321
322
  async def encode_get_experiment(accession: str) -> str:
322
323
  """Get full details for a specific ENCODE experiment by accession ID.
323
324
 
324
- Returns complete experiment metadata including all associated files,
325
- quality metrics, controls, replicate information, and audit status.
325
+ Returns the experiment's metadata, all associated files, the accessions of its possible
326
+ controls, replicate counts, and the number of audit flags at each level (ERROR,
327
+ NOT_COMPLIANT, WARNING, INTERNAL_ACTION). It does not return QC metric values such as
328
+ FRiP or NSC; those are on the experiment's page at encodeproject.org.
326
329
 
327
330
  WHEN TO USE: Use when you have a specific accession and need full details
328
- including files, quality metrics, and audit status.
331
+ including files, controls, and audit counts.
329
332
  RELATED TOOLS: encode_list_files, encode_track_experiment, encode_compare_experiments
330
333
 
331
334
  Args:
@@ -471,6 +474,7 @@ async def encode_search_files(
471
474
  """
472
475
  client = await _get_client()
473
476
  limit = clamp_limit(limit)
477
+ offset = max(0, offset)
474
478
  filter_warnings = _validate_filters(assay_title, organ, biosample_type)
475
479
  result = await client.search_files(
476
480
  file_format=file_format,
@@ -648,6 +652,7 @@ async def encode_batch_download(
648
652
  verify_md5: bool = True,
649
653
  limit: int = 100,
650
654
  dry_run: bool = True,
655
+ offset: int = 0,
651
656
  ) -> str:
652
657
  """Search for files and download them all in batch.
653
658
 
@@ -686,6 +691,8 @@ async def encode_batch_download(
686
691
  verify_md5: Verify downloads with MD5 checksums (default True)
687
692
  limit: Max files to download (default 100, safety limit)
688
693
  dry_run: If True (default), only preview what would be downloaded. Set False to download.
694
+ offset: Skip the first N matching files; pass the next_offset of the previous reply to
695
+ continue a search that has more files than limit
689
696
 
690
697
  Returns:
691
698
  JSON with download preview (dry_run=True) or download results (dry_run=False).
@@ -694,6 +701,7 @@ async def encode_batch_download(
694
701
  downloader = _get_downloader()
695
702
  validate_organize_by(organize_by)
696
703
  limit = clamp_limit(limit)
704
+ offset = max(0, offset)
697
705
  filter_warnings = _validate_filters(assay_title, organ, biosample_type)
698
706
 
699
707
  # Search for files
@@ -710,18 +718,34 @@ async def encode_batch_download(
710
718
  status="released",
711
719
  preferred_default=preferred_default,
712
720
  limit=limit,
721
+ offset=offset,
713
722
  )
714
723
 
715
724
  files = search_result["results"]
725
+ # the file search may stop before it has read every experiment; keep its note in every reply
726
+ total_note = search_result.get("total_note")
716
727
 
717
728
  if not files:
729
+ # with an offset, an empty page means "past the last match", not "nothing matches"
730
+ search_total = search_result.get("total", 0)
731
+ if search_total:
732
+ message = f"No files at offset {offset}: the search matches {search_total} file(s). Use a smaller offset."
733
+ suggestion = "Call again with offset=0 to start from the first matching file."
734
+ else:
735
+ message = "No files found matching the search criteria."
736
+ suggestion = (
737
+ "Try broadening your search filters. Use encode_get_facets to see what data is "
738
+ "available for your criteria."
739
+ )
718
740
  empty_result = {
719
- "message": "No files found matching the search criteria.",
720
- "total": 0,
741
+ "message": message,
742
+ "total": search_total,
721
743
  "has_more": False,
722
744
  "next_offset": None,
723
- "suggestion": "Try broadening your search filters. Use encode_get_facets to see what data is available for your criteria.",
745
+ "suggestion": suggestion,
724
746
  }
747
+ if total_note:
748
+ empty_result["total_note"] = total_note
725
749
  if filter_warnings:
726
750
  empty_result["filter_warnings"] = filter_warnings
727
751
  return json.dumps(empty_result, indent=2)
@@ -734,8 +758,10 @@ async def encode_batch_download(
734
758
  f"Found {preview['file_count']} files ({preview['total_size_human']}). Set dry_run=False to download."
735
759
  )
736
760
  preview["search_total"] = search_total
737
- preview["has_more"] = search_total > limit
738
- preview["next_offset"] = limit if search_total > limit else None
761
+ preview["has_more"] = search_total > offset + limit
762
+ preview["next_offset"] = offset + limit if search_total > offset + limit else None
763
+ if total_note:
764
+ preview["total_note"] = total_note
739
765
  if filter_warnings:
740
766
  preview["filter_warnings"] = filter_warnings
741
767
  return json.dumps(_serialize(preview), indent=2)
@@ -754,9 +780,11 @@ async def encode_batch_download(
754
780
  "total_size": sum(r.file_size for r in results if r.success),
755
781
  "total_size_human": _human_size(sum(r.file_size for r in results if r.success)),
756
782
  },
757
- "has_more": search_total > limit,
758
- "next_offset": limit if search_total > limit else None,
783
+ "has_more": search_total > offset + limit,
784
+ "next_offset": offset + limit if search_total > offset + limit else None,
759
785
  }
786
+ if total_note:
787
+ output["total_note"] = total_note
760
788
  if filter_warnings:
761
789
  output["filter_warnings"] = filter_warnings
762
790
  return json.dumps(output, indent=2)
@@ -1124,7 +1152,7 @@ async def encode_list_tracked(
1124
1152
  )
1125
1153
 
1126
1154
  # Build metadata table
1127
- table = tracker.get_metadata_table([e["accession"] for e in experiments] if experiments else None)
1155
+ table = tracker.get_metadata_table([e["accession"] for e in experiments])
1128
1156
 
1129
1157
  # Remove raw_metadata from output
1130
1158
  for row in table:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: encode-toolkit
3
- Version: 0.3.2
3
+ Version: 0.3.4
4
4
  Summary: MCP server for querying and downloading ENCODE Project genomics data directly from Claude
5
5
  Project-URL: Homepage, https://github.com/ammawla/encode-toolkit
6
6
  Project-URL: Repository, https://github.com/ammawla/encode-toolkit
@@ -37,7 +37,7 @@ Description-Content-Type: text/markdown
37
37
 
38
38
  [![License: AGPL-3.0](https://img.shields.io/badge/license-AGPL--3.0-green.svg)](LICENSE)
39
39
  [![Python 3.10+](https://img.shields.io/badge/python-3.10+-blue.svg)](https://www.python.org/downloads/)
40
- [![Version](https://img.shields.io/badge/version-0.3.2-green)](CHANGELOG.md)
40
+ [![Version](https://img.shields.io/badge/version-0.3.4-green)](CHANGELOG.md)
41
41
  [![Status](https://img.shields.io/badge/status-beta-yellow)]()
42
42
  [![Skills](https://img.shields.io/badge/skills-47-orange)](docs/skill-vignettes/)
43
43
  [![Tools](https://img.shields.io/badge/MCP_tools-20-purple)](src/encode_connector/server/main.py)
@@ -331,7 +331,7 @@ Search ENCODE experiments with 20+ filters.
331
331
 
332
332
  | Parameter | Type | Description |
333
333
  |-----------|------|-------------|
334
- | `assay_title` | string | Assay type: "Histone ChIP-seq", "ATAC-seq", "RNA-seq", "Hi-C", etc. |
334
+ | `assay_title` | string | Assay type: "Histone ChIP-seq", "ATAC-seq", "total RNA-seq", "Hi-C", etc. |
335
335
  | `organism` | string | Species (default: "Homo sapiens") |
336
336
  | `organ` | string | Organ: "pancreas", "brain", "liver", "heart", "kidney", etc. |
337
337
  | `biosample_type` | string | "tissue", "cell line", "primary cell", "organoid" |
@@ -496,7 +496,7 @@ Get grouped statistics of your tracked experiment collection.
496
496
  </details>
497
497
 
498
498
  <details>
499
- <summary><strong>Provenance and export tools (4)</strong></summary>
499
+ <summary><strong>Provenance and export tools (5)</strong></summary>
500
500
 
501
501
  ### `encode_log_derived_file`
502
502
 
@@ -624,7 +624,7 @@ When installed as a Claude Code plugin, ENCODE Toolkit includes 47 literature-ba
624
624
  </details>
625
625
 
626
626
  <details>
627
- <summary><strong>Workflow skills (7)</strong></summary>
627
+ <summary><strong>Workflow skills (10)</strong></summary>
628
628
 
629
629
  | Skill | Description |
630
630
  |-------|-------------|
@@ -681,7 +681,7 @@ Each pipeline includes a SKILL.md overview, 5-stage reference files (preprocessi
681
681
  | File | Description |
682
682
  |-------|-------------|
683
683
  | `skills/histone-aggregation/references/histone-marks-reference.md` | Comprehensive chromatin biology catalog (1,442 lines) — 21 histone marks with writers/erasers/readers, 5 novel acylation marks, ChromHMM state models (5 to 51 states), TF co-binding patterns, chromatin remodeling complexes, DNA methylation-chromatin interplay, nucleosome dynamics, 3D genome organization, chromatin in disease. 74 primary references |
684
- | `skills/*/references/literature.md` | 33 per-skill literature reference documents — ~250 papers cataloged with DOI, PMID, citation counts, and skill-relevant key findings |
684
+ | `skills/*/references/literature.md` | 47 per-skill literature reference documents — about 240 papers cataloged with DOI, PMID, citation counts, and skill-relevant key findings |
685
685
 
686
686
  </details>
687
687
 
@@ -710,12 +710,12 @@ Most genomics tools give you one thing. ENCODE Toolkit gives you the full resear
710
710
  | Category | Assays |
711
711
  |----------|--------|
712
712
  | **Histone/Chromatin** | Histone ChIP-seq, TF ChIP-seq, ATAC-seq, DNase-seq, CUT&RUN, CUT&Tag, MNase-seq |
713
- | **Transcription** | RNA-seq, total RNA-seq, small RNA-seq, long read RNA-seq, CAGE, RAMPAGE, PRO-seq, GRO-seq |
713
+ | **Transcription** | total RNA-seq, polyA plus RNA-seq, small RNA-seq, long read RNA-seq, CAGE, RAMPAGE, PRO-seq, GRO-seq |
714
714
  | **3D Genome** | Hi-C, intact Hi-C, Micro-C, ChIA-PET, HiChIP, PLAC-seq, 5C |
715
715
  | **DNA Methylation** | WGBS, RRBS, MeDIP-seq, MRE-seq |
716
716
  | **Functional** | STARR-seq, MPRA, CRISPR screen, eCLIP, iCLIP |
717
- | **Single Cell** | scRNA-seq, snATAC-seq, 10x multiome, SHARE-seq, Parse SPLiT-seq |
718
- | **Perturbation** | CRISPRi + RNA-seq, shRNA + RNA-seq, siRNA + RNA-seq |
717
+ | **Single Cell** | scRNA-seq, snATAC-seq, snRNA-seq, long read scRNA-seq |
718
+ | **Perturbation** | CRISPRi RNA-seq, shRNA RNA-seq, siRNA RNA-seq, CRISPR RNA-seq |
719
719
 
720
720
  **Supported file formats**: `fastq` `bam` `bed` `bigWig` `bigBed` `tsv` `csv` `hic` `tagAlign` `bedpe` `pairs` `fasta` `vcf` `tar`
721
721
 
@@ -0,0 +1,18 @@
1
+ encode_connector/__init__.py,sha256=m2jzlw8g0WFGPmR0OLkpNcOo5ao8U4Bwzp9g2nQnFac,311
2
+ encode_connector/__main__.py,sha256=hX7Hv1cKkwjnZrQkgsV-Kkx-60ywaNmobOuLuRfLZ-o,106
3
+ encode_connector/client/__init__.py,sha256=wyIOQXMUZoUqGdv8_h5lW7vfWZ6Ohq_ttz2aLN7Mvo8,216
4
+ encode_connector/client/auth.py,sha256=SnQZ3Hkq7BhpQD6vBZudAi6Lr2zCB47WXEWeFy_4xZg,9576
5
+ encode_connector/client/constants.py,sha256=8l5kUaNPEoVC0bDbQW4IaTEQJfsAfit8IVqX-ZUHmUo,11259
6
+ encode_connector/client/downloader.py,sha256=hDPLkGs8MO3Jr5gxJnPxhl211I5iZkoytZd7m7EeTFk,11877
7
+ encode_connector/client/encode_client.py,sha256=yom0wAPQogRMF0euLN-ItczgRjDleLLMfF5zqqaROPM,23303
8
+ encode_connector/client/models.py,sha256=lkt-2lr0dqBCPDBlSb7ozvSvvrlp-0JH7KtSpyjZil8,14830
9
+ encode_connector/client/tracker.py,sha256=ts_B4fOuGtL4Xz1GrCph7XyPjfcyeUNGJRO_rFOyAIs,46777
10
+ encode_connector/client/validation.py,sha256=vssV9tu9iTQes98jxCUvfSNsABtn_o8AzGKEJ2RqWf0,7814
11
+ encode_connector/server/__init__.py,sha256=qE1SX7iMPz3bwjVuuvMlXNN1g_iHB-sIUQY-LtGgAPw,33
12
+ encode_connector/server/__main__.py,sha256=gVilW3ql6i1R9fJ34xUi5RmGFDy2E_fyy9j0wcQ3tUg,124
13
+ encode_connector/server/main.py,sha256=dWFNjaX-eNbKeV4BwcbycA-0fkftypmkRtUWHl2G72I,59200
14
+ encode_toolkit-0.3.4.dist-info/METADATA,sha256=c7VqORxOgAQTex5aSHzeBmrFAtm7lyFZgSX1yDkmN-c,34459
15
+ encode_toolkit-0.3.4.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
16
+ encode_toolkit-0.3.4.dist-info/entry_points.txt,sha256=ZNqGYIpnig7ZLZKSP1xZTRZ1FdTSRLt8BcY9cqmCOW8,69
17
+ encode_toolkit-0.3.4.dist-info/licenses/LICENSE,sha256=hIahDEOTzuHCU5J2nd07LWwkLW7Hko4UFO__ffsvB-8,34523
18
+ encode_toolkit-0.3.4.dist-info/RECORD,,
@@ -1,4 +1,4 @@
1
1
  Wheel-Version: 1.0
2
- Generator: hatchling 1.32.3
2
+ Generator: hatchling 1.32.4
3
3
  Root-Is-Purelib: true
4
4
  Tag: py3-none-any
@@ -1,18 +0,0 @@
1
- encode_connector/__init__.py,sha256=A1xhCKU_XGnBkePtv1NmXWFX-A0WTcm62YQfMgaYwE8,124
2
- encode_connector/__main__.py,sha256=hX7Hv1cKkwjnZrQkgsV-Kkx-60ywaNmobOuLuRfLZ-o,106
3
- encode_connector/client/__init__.py,sha256=wyIOQXMUZoUqGdv8_h5lW7vfWZ6Ohq_ttz2aLN7Mvo8,216
4
- encode_connector/client/auth.py,sha256=SnQZ3Hkq7BhpQD6vBZudAi6Lr2zCB47WXEWeFy_4xZg,9576
5
- encode_connector/client/constants.py,sha256=x3XWhgHICZv_osWAyQrvR8GiYA8byZ3a9F0qscZ8GKc,9709
6
- encode_connector/client/downloader.py,sha256=hDPLkGs8MO3Jr5gxJnPxhl211I5iZkoytZd7m7EeTFk,11877
7
- encode_connector/client/encode_client.py,sha256=PiSGepIWoaxryX1YhcYfLTuKhJxrm4fgACmZNglAUs8,21314
8
- encode_connector/client/models.py,sha256=hNukcSisnM2Dw9xms-trt72Ymj1H1_yhi2iAxn3jfn0,13804
9
- encode_connector/client/tracker.py,sha256=8x6JNlla6UIpzgVDj8HOGnssfjQPRp-WdpmiirjLDI8,44804
10
- encode_connector/client/validation.py,sha256=vssV9tu9iTQes98jxCUvfSNsABtn_o8AzGKEJ2RqWf0,7814
11
- encode_connector/server/__init__.py,sha256=qE1SX7iMPz3bwjVuuvMlXNN1g_iHB-sIUQY-LtGgAPw,33
12
- encode_connector/server/__main__.py,sha256=gVilW3ql6i1R9fJ34xUi5RmGFDy2E_fyy9j0wcQ3tUg,124
13
- encode_connector/server/main.py,sha256=2oYZKnDM_XxW-z_NOXxlGdGOmauw5NuQdu7d3-aAbow,57826
14
- encode_toolkit-0.3.2.dist-info/METADATA,sha256=rZIBBJ6Yazo_ogyJb6tNCO9raY15xjMCrUgHfFIPdiY,34436
15
- encode_toolkit-0.3.2.dist-info/WHEEL,sha256=THafob7ofN-NsuMN7Mg4qZyHaQI7KkD-QlcQatYhXPo,87
16
- encode_toolkit-0.3.2.dist-info/entry_points.txt,sha256=ZNqGYIpnig7ZLZKSP1xZTRZ1FdTSRLt8BcY9cqmCOW8,69
17
- encode_toolkit-0.3.2.dist-info/licenses/LICENSE,sha256=hIahDEOTzuHCU5J2nd07LWwkLW7Hko4UFO__ffsvB-8,34523
18
- encode_toolkit-0.3.2.dist-info/RECORD,,