@biomate/mcp-server 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,2335 @@
1
+ {
2
+ "version": "2.0.0",
3
+ "generated_from": "mcp/tools_manifest.py",
4
+ "server": {
5
+ "name": "BioMate",
6
+ "vendor": "BioMate AI",
7
+ "documentation_url": "https://biomate.ai/connectors",
8
+ "privacy_policy_url": "https://biomate.ai/legal/privacy",
9
+ "terms_of_service_url": "https://biomate.ai/legal/terms",
10
+ "support_email": "support@biomate.ai",
11
+ "support_url": "https://biomate.ai/support"
12
+ },
13
+ "mcp": [
14
+ {
15
+ "name": "biomate_session",
16
+ "title": "Run BioMate Session",
17
+ "description": "**Primary entry point \u2014 use this for 90% of requests.** Run a complete BioMate scientific session from a natural-language goal. BioMate selects the right workflow from 2,455 indexed pipelines, pre-fills parameters from your goal text, executes on BioMate cloud, handles QC gates with auto-loop remediation, and produces structured findings. While running, the tool streams real-time progress (phase started, step completed, QC gate, auto-loop remediation, finding) back to the host. Returns a final run summary, a deep link to the live results panel, and the report URL. \n\n**How to write the `goal` parameter** \u2014 plain English, one to three sentences:\n\u2022 Include the *what*: analysis type + subject (e.g. 'ADMET screening', 'RNA-seq DE', 'variant calling')\n\u2022 Include *data location*: inline SMILES/sequences, S3 paths, accession numbers, or upload first with upload_file\n\u2022 Include key *parameters* that matter: organism, library type, comparisons, thresholds\n\u2022 You can omit anything BioMate can infer (it will ask if genuinely ambiguous)\n\n**Good examples:**\n 'Screen aspirin (CC(=O)Oc1ccccc1C(=O)O) and caffeine (Cn1cnc2c1c(=O)n(c(=O)n2C)C) for hERG inhibition, CYP3A4, and oral bioavailability'\n 'RNA-seq differential expression on s3://lab-bucket/exp42/fastqs/ \u2014 human GRCh38, dUTP strand-specific, treated (n=3) vs control (n=3), FDR 0.05'\n 'Whole-genome variant calling on the uploaded FASTQ pair, GRCh38, GATK HaplotypeCaller, germline mode'\n 'Run homogeneous 3D refinement in CryoSPARC on s3://cryo/job042/, C2 symmetry, box size 256'\n 'Fetch GSE183947 from GEO and run the same RNA-seq pipeline'\n\nUse run_workflow instead when the user wants to call a specific workflow by ID with explicit parameter control.",
18
+ "inputSchema": {
19
+ "type": "object",
20
+ "properties": {
21
+ "goal": {
22
+ "type": "string",
23
+ "description": "Natural-language scientific goal. Include: analysis type, data location (inline SMILES/sequences, s3:// paths, or GEO/SRA accession numbers), and key parameters (organism, comparisons, thresholds). BioMate infers the rest. Examples: 'Screen these 5 SMILES for hERG IC50 and CYP3A4 inhibition', 'RNA-seq DE on s3://bucket/exp1/ human GRCh38 paired-end treated vs control', 'Fetch GSE183947 and run differential expression'."
24
+ },
25
+ "inputs": {
26
+ "type": "object",
27
+ "description": "Optional structured inputs passed alongside `goal`. BioMate's inner AI sees these as a JSON block appended to the goal text and merges them with any parameters it extracts from the goal string. Use this to pass: S3 keys from upload_file, sequences, SMILES lists, accession numbers, or explicit parameter overrides. Examples:\n After upload_file: {\"fastq_file\": \"users/42/uploads/uuid-sample.fastq.gz\"} (use the s3_key value returned by upload_file)\n SMILES list: {\"smiles_list\": [\"CC(=O)Oc1ccccc1C(=O)O\"], \"organism\": \"human\"}\n Parameter override: {\"genome\": \"GRCh38\", \"aligner\": \"STAR\", \"fdr\": 0.05}",
28
+ "additionalProperties": true
29
+ },
30
+ "experiment_id": {
31
+ "type": "string",
32
+ "description": "Optional experiment to attach this run to (from recall_memory)."
33
+ },
34
+ "stream": {
35
+ "type": "boolean",
36
+ "description": "Emit progress notifications during execution. Default true. Set false for hosts without MCP notification support \u2014 then poll with get_run.",
37
+ "default": true
38
+ }
39
+ },
40
+ "required": [
41
+ "goal"
42
+ ]
43
+ },
44
+ "annotations": {
45
+ "title": "Run BioMate Session",
46
+ "readOnlyHint": false,
47
+ "destructiveHint": false,
48
+ "idempotentHint": false,
49
+ "openWorldHint": true
50
+ }
51
+ },
52
+ {
53
+ "name": "search_workflow",
54
+ "title": "Search Workflows",
55
+ "description": "Search the BioMate workflow catalog (2,455 indexed workflows across 34 domains) by natural language. Returns ranked workflow cards with id, name, domain, one-line description, and estimated BioMate cloud cost. Use this when the user wants to pick a workflow explicitly; otherwise prefer biomate_session.",
56
+ "inputSchema": {
57
+ "type": "object",
58
+ "properties": {
59
+ "query": {
60
+ "type": "string",
61
+ "description": "Natural language description of the analysis."
62
+ },
63
+ "limit": {
64
+ "type": "integer",
65
+ "description": "Max results (default 5, max 20).",
66
+ "default": 5
67
+ },
68
+ "domain": {
69
+ "type": "string",
70
+ "description": "Optional domain filter: transcriptomics, genomics, proteomics, drug_discovery, cryo_em, etc."
71
+ }
72
+ },
73
+ "required": [
74
+ "query"
75
+ ]
76
+ },
77
+ "annotations": {
78
+ "title": "Search Workflows",
79
+ "readOnlyHint": true,
80
+ "destructiveHint": false,
81
+ "idempotentHint": false,
82
+ "openWorldHint": true
83
+ }
84
+ },
85
+ {
86
+ "name": "get_workflow_spec",
87
+ "title": "Get Workflow Spec",
88
+ "description": "Return the full specification for a workflow: required + optional parameters (with types and allowed values), default QC profile and thresholds, expected input files, estimated cost and runtime, and any license requirements. Call this before run_workflow when the user wants explicit parameter control.",
89
+ "inputSchema": {
90
+ "type": "object",
91
+ "properties": {
92
+ "workflow_id": {
93
+ "type": "string",
94
+ "description": "Workflow ID from search_workflow."
95
+ }
96
+ },
97
+ "required": [
98
+ "workflow_id"
99
+ ]
100
+ },
101
+ "annotations": {
102
+ "title": "Get Workflow Spec",
103
+ "readOnlyHint": true,
104
+ "destructiveHint": false,
105
+ "idempotentHint": false,
106
+ "openWorldHint": false
107
+ }
108
+ },
109
+ {
110
+ "name": "run_workflow",
111
+ "title": "Run Workflow",
112
+ "description": "Execute a specific BioMate workflow on BioMate cloud with explicit parameters. Returns a run_id immediately. If stream=true, also emits progress notifications until the run terminates (same events as biomate_session). Use biomate_session instead when the user gave you a natural-language goal.",
113
+ "inputSchema": {
114
+ "type": "object",
115
+ "properties": {
116
+ "workflow_id": {
117
+ "type": "string",
118
+ "description": "Workflow ID from search_workflow."
119
+ },
120
+ "params": {
121
+ "type": "object",
122
+ "description": "Parameter dict. Required params come from get_workflow_spec.",
123
+ "additionalProperties": true
124
+ },
125
+ "experiment_id": {
126
+ "type": "string",
127
+ "description": "Optional experiment to attach to."
128
+ },
129
+ "stream": {
130
+ "type": "boolean",
131
+ "description": "Emit progress notifications while running. Default false.",
132
+ "default": false
133
+ }
134
+ },
135
+ "required": [
136
+ "workflow_id"
137
+ ]
138
+ },
139
+ "annotations": {
140
+ "title": "Run Workflow",
141
+ "readOnlyHint": false,
142
+ "destructiveHint": false,
143
+ "idempotentHint": false,
144
+ "openWorldHint": true
145
+ }
146
+ },
147
+ {
148
+ "name": "watch_run",
149
+ "title": "",
150
+ "description": "Stream real-time progress for a running BioMate workflow and return full results when done. Emits MCP notifications/progress for every phase start/complete, step update, and QC gate. When the run finishes, automatically fetches output files (with download URLs) and AI findings. Use after run_workflow (non-streaming) to watch a submitted run. Does not require re-submitting \u2014 takes an existing run_id.",
151
+ "inputSchema": {
152
+ "type": "object",
153
+ "properties": {
154
+ "run_id": {
155
+ "type": "string",
156
+ "description": "Run ID returned by run_workflow."
157
+ }
158
+ },
159
+ "required": [
160
+ "run_id"
161
+ ]
162
+ },
163
+ "annotations": {
164
+ "title": "",
165
+ "readOnlyHint": false,
166
+ "destructiveHint": false,
167
+ "idempotentHint": false,
168
+ "openWorldHint": false
169
+ }
170
+ },
171
+ {
172
+ "name": "get_run",
173
+ "title": "Get Run Details",
174
+ "description": "Return everything about a run in one call: status (pending|running|completed|failed), per-phase and per-step progress with timestamps, output files with download URLs, structured findings, QC gate results, and any auto-loop remediations applied. Replaces get_run_status + get_run_results + step-level findings polling.",
175
+ "inputSchema": {
176
+ "type": "object",
177
+ "properties": {
178
+ "run_id": {
179
+ "type": "string",
180
+ "description": "Run ID."
181
+ },
182
+ "include_findings": {
183
+ "type": "boolean",
184
+ "description": "Include structured findings cards (default true).",
185
+ "default": true
186
+ }
187
+ },
188
+ "required": [
189
+ "run_id"
190
+ ]
191
+ },
192
+ "annotations": {
193
+ "title": "Get Run Details",
194
+ "readOnlyHint": true,
195
+ "destructiveHint": false,
196
+ "idempotentHint": false,
197
+ "openWorldHint": false
198
+ }
199
+ },
200
+ {
201
+ "name": "cancel_run",
202
+ "title": "Cancel Run",
203
+ "description": "Cancel a running or queued BioMate workflow on BioMate cloud.",
204
+ "inputSchema": {
205
+ "type": "object",
206
+ "properties": {
207
+ "run_id": {
208
+ "type": "string",
209
+ "description": "Run ID to cancel."
210
+ }
211
+ },
212
+ "required": [
213
+ "run_id"
214
+ ]
215
+ },
216
+ "annotations": {
217
+ "title": "Cancel Run",
218
+ "readOnlyHint": false,
219
+ "destructiveHint": true,
220
+ "idempotentHint": true,
221
+ "openWorldHint": false
222
+ }
223
+ },
224
+ {
225
+ "name": "list_runs",
226
+ "title": "List Runs",
227
+ "description": "List the user's recent runs with status and timestamps. Filter by status or experiment.",
228
+ "inputSchema": {
229
+ "type": "object",
230
+ "properties": {
231
+ "limit": {
232
+ "type": "integer",
233
+ "description": "Max runs to return (default 10).",
234
+ "default": 10
235
+ },
236
+ "status": {
237
+ "type": "string",
238
+ "enum": [
239
+ "all",
240
+ "running",
241
+ "completed",
242
+ "failed",
243
+ "pending"
244
+ ],
245
+ "default": "all"
246
+ },
247
+ "experiment_id": {
248
+ "type": "string",
249
+ "description": "Optional experiment filter."
250
+ }
251
+ }
252
+ },
253
+ "annotations": {
254
+ "title": "List Runs",
255
+ "readOnlyHint": true,
256
+ "destructiveHint": false,
257
+ "idempotentHint": false,
258
+ "openWorldHint": false
259
+ }
260
+ },
261
+ {
262
+ "name": "preview_file",
263
+ "title": "Preview Output File",
264
+ "description": "Render a server-side preview of an output file (FASTA, VCF, CSV/TSV, image, PDF). Returns markdown + optional thumbnail PNG. Use this to show the user what an output looks like without downloading multi-GB files. For input QC of files the user is about to upload, prefer biomate_session (it runs a real QC workflow).",
265
+ "inputSchema": {
266
+ "type": "object",
267
+ "properties": {
268
+ "s3_key": {
269
+ "type": "string",
270
+ "description": "The full s3_uri value (must start with 's3://') copied verbatim from a get_run output_files entry \u2014 e.g. 's3://bucket/path/file.json'. NOT a bare key; the s3:// prefix is required."
271
+ },
272
+ "run_id": {
273
+ "type": "string",
274
+ "description": "Optional run_id for context-aware parsing."
275
+ },
276
+ "max_rows": {
277
+ "type": "integer",
278
+ "description": "For tabular files (default 100).",
279
+ "default": 100
280
+ }
281
+ },
282
+ "required": [
283
+ "s3_key"
284
+ ]
285
+ },
286
+ "annotations": {
287
+ "title": "Preview Output File",
288
+ "readOnlyHint": true,
289
+ "destructiveHint": false,
290
+ "idempotentHint": false,
291
+ "openWorldHint": false
292
+ }
293
+ },
294
+ {
295
+ "name": "export_report",
296
+ "title": "Export Report",
297
+ "description": "Render a publication-ready report for a completed run as PDF or markdown. Includes the methods section, QC audit trail, structured findings, and figures. This is what users need for IND submissions, CRO compliance packages, and publication supplementary materials.",
298
+ "inputSchema": {
299
+ "type": "object",
300
+ "properties": {
301
+ "run_id": {
302
+ "type": "string",
303
+ "description": "Completed run ID."
304
+ },
305
+ "format": {
306
+ "type": "string",
307
+ "enum": [
308
+ "pdf",
309
+ "markdown",
310
+ "docx"
311
+ ],
312
+ "default": "pdf"
313
+ },
314
+ "sections": {
315
+ "type": "array",
316
+ "items": {
317
+ "type": "string",
318
+ "enum": [
319
+ "methods",
320
+ "qc",
321
+ "findings",
322
+ "figures",
323
+ "appendix"
324
+ ]
325
+ },
326
+ "description": "Which sections to include (default: all)."
327
+ }
328
+ },
329
+ "required": [
330
+ "run_id"
331
+ ]
332
+ },
333
+ "annotations": {
334
+ "title": "Export Report",
335
+ "readOnlyHint": false,
336
+ "destructiveHint": false,
337
+ "idempotentHint": true,
338
+ "openWorldHint": false
339
+ }
340
+ },
341
+ {
342
+ "name": "analyze_results",
343
+ "title": "Analyze Results",
344
+ "description": "Ask BioMate's AI to interpret a completed run. Returns natural-language analysis: key findings, quality assessment, scientific interpretation, recommended next steps. Use after get_run when the user asks 'what does this mean?'.",
345
+ "inputSchema": {
346
+ "type": "object",
347
+ "properties": {
348
+ "run_id": {
349
+ "type": "string",
350
+ "description": "Completed run ID."
351
+ },
352
+ "question": {
353
+ "type": "string",
354
+ "description": "Optional focused question."
355
+ }
356
+ },
357
+ "required": [
358
+ "run_id"
359
+ ]
360
+ },
361
+ "annotations": {
362
+ "title": "Analyze Results",
363
+ "readOnlyHint": true,
364
+ "destructiveHint": false,
365
+ "idempotentHint": false,
366
+ "openWorldHint": true
367
+ }
368
+ },
369
+ {
370
+ "name": "explain_error",
371
+ "title": "Explain Run Error",
372
+ "description": "Diagnose a failed run. Returns the likely root cause (genome mismatch, missing input, OOM, container pull failure, etc.) and the specific fix. Often the next step is run_workflow with corrected params.",
373
+ "inputSchema": {
374
+ "type": "object",
375
+ "properties": {
376
+ "run_id": {
377
+ "type": "string",
378
+ "description": "Failed run ID."
379
+ },
380
+ "error_log": {
381
+ "type": "string",
382
+ "description": "Optional pasted error log."
383
+ }
384
+ },
385
+ "required": [
386
+ "run_id"
387
+ ]
388
+ },
389
+ "annotations": {
390
+ "title": "Explain Run Error",
391
+ "readOnlyHint": true,
392
+ "destructiveHint": false,
393
+ "idempotentHint": false,
394
+ "openWorldHint": true
395
+ }
396
+ },
397
+ {
398
+ "name": "query_database",
399
+ "title": "Query Biological Database",
400
+ "description": "Query a biological/chemical database by accession or name. Use database='federated' to fan out across all sources simultaneously and merge results. Single-source options: uniprot, pdb, pdbe, alphafold, ncbi_gene, dbsnp, clinvar, gnomad, kegg, reactome, chebi, chembl, pubchem, bindingdb, pharos, hpo, string, pubmed.",
401
+ "inputSchema": {
402
+ "type": "object",
403
+ "properties": {
404
+ "database": {
405
+ "type": "string",
406
+ "description": "Database to query. Use 'federated' to query all sources in parallel.",
407
+ "enum": [
408
+ "federated",
409
+ "uniprot",
410
+ "pdb",
411
+ "pdbe",
412
+ "alphafold",
413
+ "ncbi_gene",
414
+ "dbsnp",
415
+ "clinvar",
416
+ "gnomad",
417
+ "kegg",
418
+ "reactome",
419
+ "chebi",
420
+ "chembl",
421
+ "pubchem",
422
+ "bindingdb",
423
+ "pharos",
424
+ "hpo",
425
+ "string",
426
+ "pubmed"
427
+ ],
428
+ "default": "federated"
429
+ },
430
+ "query": {
431
+ "type": "string",
432
+ "description": "Accession, gene symbol, compound name, or free text."
433
+ },
434
+ "operation": {
435
+ "type": "string",
436
+ "description": "Search mode chosen from user intent: 'lookup' (by ID/name, default), 'similarity' (chembl/pubchem=chemical similarity over SMILES; pdb=sequence homologs over the whole PDB), or 'text' (keyword search).",
437
+ "enum": [
438
+ "lookup",
439
+ "similarity",
440
+ "text"
441
+ ],
442
+ "default": "lookup"
443
+ },
444
+ "entity_type": {
445
+ "type": "string",
446
+ "description": "Hint for federated routing: gene, protein, variant, compound, pathway, disease, or omit to auto-detect.",
447
+ "enum": [
448
+ "gene",
449
+ "protein",
450
+ "variant",
451
+ "compound",
452
+ "pathway",
453
+ "disease"
454
+ ]
455
+ }
456
+ },
457
+ "required": [
458
+ "query"
459
+ ]
460
+ },
461
+ "annotations": {
462
+ "title": "Query Biological Database",
463
+ "readOnlyHint": true,
464
+ "destructiveHint": false,
465
+ "idempotentHint": false,
466
+ "openWorldHint": true
467
+ }
468
+ },
469
+ {
470
+ "name": "resolve_accession",
471
+ "title": "Resolve Accession",
472
+ "description": "Identify a public archive accession (GEO, SRA, ENA, DDBJ) and return the best BioMate workflow to fetch it plus pre-filled params ready for run_workflow. Examples: GSE183947 \u2192 Geo Data Connector; SRR12345 \u2192 nf-core/fetchngs. Call this before run_workflow whenever the user provides an accession instead of a file.",
473
+ "inputSchema": {
474
+ "type": "object",
475
+ "properties": {
476
+ "accession": {
477
+ "type": "string",
478
+ "description": "Public archive accession: GSE*, GSM*, GDS*, SRR*, SRX*, SRP*, ERR*, ERX*, ERP*, DRR*, PRJNA*, PRJEB*, E-MTAB-*, E-GEOD-*"
479
+ }
480
+ },
481
+ "required": [
482
+ "accession"
483
+ ]
484
+ },
485
+ "annotations": {
486
+ "title": "Resolve Accession",
487
+ "readOnlyHint": true,
488
+ "destructiveHint": false,
489
+ "idempotentHint": false,
490
+ "openWorldHint": true
491
+ }
492
+ },
493
+ {
494
+ "name": "browse_data",
495
+ "title": "Browse Data Repository",
496
+ "description": "Browse a biological data repository or S3 workspace by listing files and directories at a given path. Public sources (EBI, NCBI, Ensembl, UCSC) require a path. S3 sources (biomate_workspace, user_s3) accept an optional prefix. Use before fetch_public_data to navigate to the exact file you need.",
497
+ "inputSchema": {
498
+ "type": "object",
499
+ "properties": {
500
+ "source_id": {
501
+ "type": "string",
502
+ "enum": [
503
+ "ebi_ftp",
504
+ "ncbi_ftp",
505
+ "ensembl_ftp",
506
+ "ucsc_downloads",
507
+ "http_public",
508
+ "biomate_workspace",
509
+ "user_s3"
510
+ ],
511
+ "description": "Data source to browse. biomate_workspace = BioMate's S3 work bucket; user_s3 = user's own S3 bucket (if configured)."
512
+ },
513
+ "path": {
514
+ "type": "string",
515
+ "description": "Directory path or S3 prefix to list (e.g. '/pub/databases/uniprot/' or 'results/run-xyz/')."
516
+ }
517
+ },
518
+ "required": [
519
+ "source_id"
520
+ ]
521
+ },
522
+ "annotations": {
523
+ "title": "Browse Data Repository",
524
+ "readOnlyHint": true,
525
+ "destructiveHint": false,
526
+ "idempotentHint": false,
527
+ "openWorldHint": true
528
+ }
529
+ },
530
+ {
531
+ "name": "fetch_public_data",
532
+ "title": "Fetch Public Data",
533
+ "description": "Download a file from a public biological repository (EBI, NCBI, Ensembl, UCSC) into BioMate's S3 workspace and return a presigned URL and S3 URI. Use the S3 URI as a workflow input parameter. Call browse_data first to find the exact path.",
534
+ "inputSchema": {
535
+ "type": "object",
536
+ "properties": {
537
+ "source_id": {
538
+ "type": "string",
539
+ "enum": [
540
+ "ebi_ftp",
541
+ "ncbi_ftp",
542
+ "ensembl_ftp",
543
+ "ucsc_downloads",
544
+ "http_public"
545
+ ],
546
+ "description": "Data source the file comes from."
547
+ },
548
+ "remote_path": {
549
+ "type": "string",
550
+ "description": "Path on the FTP server (e.g. '/pub/databases/uniprot/.../uniprot_sprot.fasta.gz')."
551
+ },
552
+ "url": {
553
+ "type": "string",
554
+ "description": "For http_public source: full HTTPS URL of the file to download."
555
+ }
556
+ },
557
+ "required": [
558
+ "source_id"
559
+ ]
560
+ },
561
+ "annotations": {
562
+ "title": "Fetch Public Data",
563
+ "readOnlyHint": false,
564
+ "destructiveHint": false,
565
+ "idempotentHint": true,
566
+ "openWorldHint": true
567
+ }
568
+ },
569
+ {
570
+ "name": "recall_memory",
571
+ "title": "Recall Memory",
572
+ "description": "Retrieve relevant prior context for the current goal: past runs on similar inputs, validated procedures, findings tagged by the user, learned parameter preferences. Call before biomate_session for repeat users \u2014 it dramatically improves param auto-fill and avoids re-running work.",
573
+ "inputSchema": {
574
+ "type": "object",
575
+ "properties": {
576
+ "query": {
577
+ "type": "string",
578
+ "description": "What to recall \u2014 usually the user's goal."
579
+ },
580
+ "scope": {
581
+ "type": "string",
582
+ "enum": [
583
+ "runs",
584
+ "findings",
585
+ "procedures",
586
+ "all"
587
+ ],
588
+ "default": "all"
589
+ },
590
+ "limit": {
591
+ "type": "integer",
592
+ "default": 5
593
+ }
594
+ },
595
+ "required": [
596
+ "query"
597
+ ]
598
+ },
599
+ "annotations": {
600
+ "title": "Recall Memory",
601
+ "readOnlyHint": true,
602
+ "destructiveHint": false,
603
+ "idempotentHint": false,
604
+ "openWorldHint": false
605
+ }
606
+ },
607
+ {
608
+ "name": "search_literature",
609
+ "title": "Search Literature",
610
+ "description": "Search published scientific literature across PubMed, EuropePMC, Semantic Scholar, and OpenAlex. Uses an iterative depth-loop: starts broad, then drills into related topics until min_results are found or max_depth is reached. Returns ranked papers with title, authors, year, DOI, abstract snippet, and citation count. Best for: finding prior art, understanding a disease mechanism, locating validation datasets, or reviewing methods before running a workflow.",
611
+ "inputSchema": {
612
+ "type": "object",
613
+ "properties": {
614
+ "query": {
615
+ "type": "string",
616
+ "description": "Natural-language search query. Be specific: include organism, assay type, or drug name. E.g. 'DESeq2 bulk RNA-seq breast cancer ER+ 2020-2024'."
617
+ },
618
+ "max_depth": {
619
+ "type": "integer",
620
+ "default": 2,
621
+ "description": "How many iterative refinement rounds to run (1\u20134). Higher = more thorough, slower."
622
+ },
623
+ "min_results": {
624
+ "type": "integer",
625
+ "default": 10,
626
+ "description": "Stop when at least this many papers are found."
627
+ }
628
+ },
629
+ "required": [
630
+ "query"
631
+ ]
632
+ },
633
+ "annotations": {
634
+ "title": "Search Literature",
635
+ "readOnlyHint": true,
636
+ "destructiveHint": false,
637
+ "idempotentHint": false,
638
+ "openWorldHint": true
639
+ }
640
+ },
641
+ {
642
+ "name": "upload_file",
643
+ "title": "Upload File",
644
+ "description": "Get a one-shot signed S3 PUT URL so the host can upload a local file directly to BioMate's data plane without proxying bytes through the chat transport. For files >5MB this is mandatory; for small text payloads, inline strings are fine.\n\n**Returns** `{upload_url, s3_key}` \u2014 two fields:\n- `upload_url`: a presigned HTTPS URL. You MUST upload the file bytes to it with an HTTP PUT before calling any other tool: `curl -X PUT -T <local_path> \"<upload_url>\"`\n- `s3_key`: a bare S3 object key (e.g. `users/42/uploads/uuid-sample.fastq.gz`). Pass this value directly into the `inputs` dict of `biomate_session` or as a `params` value in `run_workflow` \u2014 BioMate resolves the bucket internally. Example: `inputs={\"fastq_file\": s3_key}` or `params={\"input\": s3_key}`.\n\n**Full upload workflow:**\n1. `upload_file(filename='sample.fastq.gz', size_bytes=52000000)` \u2192 get `upload_url` + `s3_key`\n2. `curl -X PUT -T sample.fastq.gz \"<upload_url>\"` (or equivalent HTTP PUT \u2014 no auth header needed)\n3. `biomate_session(goal='Run RNA-seq DE on the uploaded FASTQs, human GRCh38', inputs={\"fastq_file\": s3_key})`",
645
+ "inputSchema": {
646
+ "type": "object",
647
+ "properties": {
648
+ "filename": {
649
+ "type": "string",
650
+ "description": "Original filename (sets content-type and extension)."
651
+ },
652
+ "size_bytes": {
653
+ "type": "integer",
654
+ "description": "File size for quota check."
655
+ },
656
+ "content_type": {
657
+ "type": "string",
658
+ "description": "MIME type (auto-detected from filename if omitted)."
659
+ }
660
+ },
661
+ "required": [
662
+ "filename"
663
+ ]
664
+ },
665
+ "annotations": {
666
+ "title": "Upload File",
667
+ "readOnlyHint": false,
668
+ "destructiveHint": false,
669
+ "idempotentHint": false,
670
+ "openWorldHint": false
671
+ }
672
+ },
673
+ {
674
+ "name": "list_instruments",
675
+ "title": "List Lab Instruments",
676
+ "description": "Discover lab instruments reachable on the local network (currently OpenTrons OT-2/Flex liquid-handling robots). Returns each robot's IP, name, model, and health. Runs locally \u2014 only works when the MCP server is on the same network as the instruments. Set the OPENTRONS_ROBOT_IPS env var or pass `ips` to skip network scanning.",
677
+ "inputSchema": {
678
+ "type": "object",
679
+ "properties": {
680
+ "ips": {
681
+ "type": "array",
682
+ "items": {
683
+ "type": "string"
684
+ },
685
+ "description": "Explicit robot IPs to probe (e.g. ['192.168.1.42']). Omit to use OPENTRONS_ROBOT_IPS."
686
+ }
687
+ }
688
+ },
689
+ "annotations": {
690
+ "title": "List Lab Instruments",
691
+ "readOnlyHint": true,
692
+ "destructiveHint": false,
693
+ "idempotentHint": false,
694
+ "openWorldHint": true
695
+ }
696
+ },
697
+ {
698
+ "name": "get_instrument_status",
699
+ "title": "Get Instrument Status",
700
+ "description": "Get the status of a run on a lab instrument (OpenTrons). Returns status ('idle'|'running'|'paused'|'succeeded'|'failed'|'stopped'), timestamps, and any errors. Poll this after run_instrument_protocol.",
701
+ "inputSchema": {
702
+ "type": "object",
703
+ "properties": {
704
+ "robot_ip": {
705
+ "type": "string",
706
+ "description": "IP of the robot running the protocol."
707
+ },
708
+ "run_id": {
709
+ "type": "string",
710
+ "description": "Run ID returned by run_instrument_protocol."
711
+ }
712
+ },
713
+ "required": [
714
+ "robot_ip",
715
+ "run_id"
716
+ ]
717
+ },
718
+ "annotations": {
719
+ "title": "Get Instrument Status",
720
+ "readOnlyHint": true,
721
+ "destructiveHint": false,
722
+ "idempotentHint": false,
723
+ "openWorldHint": true
724
+ }
725
+ },
726
+ {
727
+ "name": "run_instrument_protocol",
728
+ "title": "Run Instrument Protocol",
729
+ "description": "Execute a liquid-handling protocol on an OpenTrons robot. **Physical action \u2014 actuates real hardware.** Two-step guardrail: called with `confirm=false` (the DEFAULT) it performs a DRY RUN \u2014 compiles and returns the exact protocol that WOULD run, WITHOUT contacting or actuating the robot. Review it, then call again with `confirm=true` to verify reachability, upload, and physically start the run. Provide either raw `protocol_code` (a valid OT-2/Flex Python protocol) or a structured `transfers` spec (source/dest labware + pipette + well transfers), which is compiled into a protocol for you. Returns {protocol_id, run_id, status}; poll get_instrument_status to follow the run.",
730
+ "inputSchema": {
731
+ "type": "object",
732
+ "properties": {
733
+ "robot_ip": {
734
+ "type": "string",
735
+ "description": "Target robot IP (from list_instruments)."
736
+ },
737
+ "confirm": {
738
+ "type": "boolean",
739
+ "default": false,
740
+ "description": "MUST be true to physically run. false (default) = dry-run preview only, no actuation."
741
+ },
742
+ "protocol_code": {
743
+ "type": "string",
744
+ "description": "Full OT-2/Flex Python protocol source. Omit if using `transfers`."
745
+ },
746
+ "protocol_name": {
747
+ "type": "string",
748
+ "default": "biomate_protocol.py",
749
+ "description": "Filename on the robot (must end in .py)."
750
+ },
751
+ "transfers": {
752
+ "type": "array",
753
+ "description": "Structured well-to-well transfers; compiled to a protocol when protocol_code is omitted.",
754
+ "items": {
755
+ "type": "object",
756
+ "properties": {
757
+ "from_well": {
758
+ "type": "string"
759
+ },
760
+ "to_well": {
761
+ "type": "string"
762
+ },
763
+ "volume_ul": {
764
+ "type": "number"
765
+ }
766
+ },
767
+ "required": [
768
+ "from_well",
769
+ "to_well",
770
+ "volume_ul"
771
+ ]
772
+ }
773
+ },
774
+ "source_labware": {
775
+ "type": "string",
776
+ "description": "OT-2 labware for the source plate (required with `transfers`)."
777
+ },
778
+ "dest_labware": {
779
+ "type": "string",
780
+ "description": "OT-2 labware for the destination plate (required with `transfers`)."
781
+ },
782
+ "pipette": {
783
+ "type": "string",
784
+ "default": "p300_single_gen2",
785
+ "description": "Pipette model (used with `transfers`)."
786
+ }
787
+ },
788
+ "required": [
789
+ "robot_ip"
790
+ ]
791
+ },
792
+ "annotations": {
793
+ "title": "Run Instrument Protocol",
794
+ "readOnlyHint": false,
795
+ "destructiveHint": true,
796
+ "idempotentHint": false,
797
+ "openWorldHint": true
798
+ }
799
+ }
800
+ ],
801
+ "anthropic": [
802
+ {
803
+ "name": "biomate_session",
804
+ "description": "**Primary entry point \u2014 use this for 90% of requests.** Run a complete BioMate scientific session from a natural-language goal. BioMate selects the right workflow from 2,455 indexed pipelines, pre-fills parameters from your goal text, executes on BioMate cloud, handles QC gates with auto-loop remediation, and produces structured findings. While running, the tool streams real-time progress (phase started, step completed, QC gate, auto-loop remediation, finding) back to the host. Returns a final run summary, a deep link to the live results panel, and the report URL. \n\n**How to write the `goal` parameter** \u2014 plain English, one to three sentences:\n\u2022 Include the *what*: analysis type + subject (e.g. 'ADMET screening', 'RNA-seq DE', 'variant calling')\n\u2022 Include *data location*: inline SMILES/sequences, S3 paths, accession numbers, or upload first with upload_file\n\u2022 Include key *parameters* that matter: organism, library type, comparisons, thresholds\n\u2022 You can omit anything BioMate can infer (it will ask if genuinely ambiguous)\n\n**Good examples:**\n 'Screen aspirin (CC(=O)Oc1ccccc1C(=O)O) and caffeine (Cn1cnc2c1c(=O)n(c(=O)n2C)C) for hERG inhibition, CYP3A4, and oral bioavailability'\n 'RNA-seq differential expression on s3://lab-bucket/exp42/fastqs/ \u2014 human GRCh38, dUTP strand-specific, treated (n=3) vs control (n=3), FDR 0.05'\n 'Whole-genome variant calling on the uploaded FASTQ pair, GRCh38, GATK HaplotypeCaller, germline mode'\n 'Run homogeneous 3D refinement in CryoSPARC on s3://cryo/job042/, C2 symmetry, box size 256'\n 'Fetch GSE183947 from GEO and run the same RNA-seq pipeline'\n\nUse run_workflow instead when the user wants to call a specific workflow by ID with explicit parameter control.",
805
+ "input_schema": {
806
+ "type": "object",
807
+ "properties": {
808
+ "goal": {
809
+ "type": "string",
810
+ "description": "Natural-language scientific goal. Include: analysis type, data location (inline SMILES/sequences, s3:// paths, or GEO/SRA accession numbers), and key parameters (organism, comparisons, thresholds). BioMate infers the rest. Examples: 'Screen these 5 SMILES for hERG IC50 and CYP3A4 inhibition', 'RNA-seq DE on s3://bucket/exp1/ human GRCh38 paired-end treated vs control', 'Fetch GSE183947 and run differential expression'."
811
+ },
812
+ "inputs": {
813
+ "type": "object",
814
+ "description": "Optional structured inputs passed alongside `goal`. BioMate's inner AI sees these as a JSON block appended to the goal text and merges them with any parameters it extracts from the goal string. Use this to pass: S3 keys from upload_file, sequences, SMILES lists, accession numbers, or explicit parameter overrides. Examples:\n After upload_file: {\"fastq_file\": \"users/42/uploads/uuid-sample.fastq.gz\"} (use the s3_key value returned by upload_file)\n SMILES list: {\"smiles_list\": [\"CC(=O)Oc1ccccc1C(=O)O\"], \"organism\": \"human\"}\n Parameter override: {\"genome\": \"GRCh38\", \"aligner\": \"STAR\", \"fdr\": 0.05}",
815
+ "additionalProperties": true
816
+ },
817
+ "experiment_id": {
818
+ "type": "string",
819
+ "description": "Optional experiment to attach this run to (from recall_memory)."
820
+ },
821
+ "stream": {
822
+ "type": "boolean",
823
+ "description": "Emit progress notifications during execution. Default true. Set false for hosts without MCP notification support \u2014 then poll with get_run.",
824
+ "default": true
825
+ }
826
+ },
827
+ "required": [
828
+ "goal"
829
+ ]
830
+ }
831
+ },
832
+ {
833
+ "name": "search_workflow",
834
+ "description": "Search the BioMate workflow catalog (2,455 indexed workflows across 34 domains) by natural language. Returns ranked workflow cards with id, name, domain, one-line description, and estimated BioMate cloud cost. Use this when the user wants to pick a workflow explicitly; otherwise prefer biomate_session.",
835
+ "input_schema": {
836
+ "type": "object",
837
+ "properties": {
838
+ "query": {
839
+ "type": "string",
840
+ "description": "Natural language description of the analysis."
841
+ },
842
+ "limit": {
843
+ "type": "integer",
844
+ "description": "Max results (default 5, max 20).",
845
+ "default": 5
846
+ },
847
+ "domain": {
848
+ "type": "string",
849
+ "description": "Optional domain filter: transcriptomics, genomics, proteomics, drug_discovery, cryo_em, etc."
850
+ }
851
+ },
852
+ "required": [
853
+ "query"
854
+ ]
855
+ }
856
+ },
857
+ {
858
+ "name": "get_workflow_spec",
859
+ "description": "Return the full specification for a workflow: required + optional parameters (with types and allowed values), default QC profile and thresholds, expected input files, estimated cost and runtime, and any license requirements. Call this before run_workflow when the user wants explicit parameter control.",
860
+ "input_schema": {
861
+ "type": "object",
862
+ "properties": {
863
+ "workflow_id": {
864
+ "type": "string",
865
+ "description": "Workflow ID from search_workflow."
866
+ }
867
+ },
868
+ "required": [
869
+ "workflow_id"
870
+ ]
871
+ }
872
+ },
873
+ {
874
+ "name": "run_workflow",
875
+ "description": "Execute a specific BioMate workflow on BioMate cloud with explicit parameters. Returns a run_id immediately. If stream=true, also emits progress notifications until the run terminates (same events as biomate_session). Use biomate_session instead when the user gave you a natural-language goal.",
876
+ "input_schema": {
877
+ "type": "object",
878
+ "properties": {
879
+ "workflow_id": {
880
+ "type": "string",
881
+ "description": "Workflow ID from search_workflow."
882
+ },
883
+ "params": {
884
+ "type": "object",
885
+ "description": "Parameter dict. Required params come from get_workflow_spec.",
886
+ "additionalProperties": true
887
+ },
888
+ "experiment_id": {
889
+ "type": "string",
890
+ "description": "Optional experiment to attach to."
891
+ },
892
+ "stream": {
893
+ "type": "boolean",
894
+ "description": "Emit progress notifications while running. Default false.",
895
+ "default": false
896
+ }
897
+ },
898
+ "required": [
899
+ "workflow_id"
900
+ ]
901
+ }
902
+ },
903
+ {
904
+ "name": "watch_run",
905
+ "description": "Stream real-time progress for a running BioMate workflow and return full results when done. Emits MCP notifications/progress for every phase start/complete, step update, and QC gate. When the run finishes, automatically fetches output files (with download URLs) and AI findings. Use after run_workflow (non-streaming) to watch a submitted run. Does not require re-submitting \u2014 takes an existing run_id.",
906
+ "input_schema": {
907
+ "type": "object",
908
+ "properties": {
909
+ "run_id": {
910
+ "type": "string",
911
+ "description": "Run ID returned by run_workflow."
912
+ }
913
+ },
914
+ "required": [
915
+ "run_id"
916
+ ]
917
+ }
918
+ },
919
+ {
920
+ "name": "get_run",
921
+ "description": "Return everything about a run in one call: status (pending|running|completed|failed), per-phase and per-step progress with timestamps, output files with download URLs, structured findings, QC gate results, and any auto-loop remediations applied. Replaces get_run_status + get_run_results + step-level findings polling.",
922
+ "input_schema": {
923
+ "type": "object",
924
+ "properties": {
925
+ "run_id": {
926
+ "type": "string",
927
+ "description": "Run ID."
928
+ },
929
+ "include_findings": {
930
+ "type": "boolean",
931
+ "description": "Include structured findings cards (default true).",
932
+ "default": true
933
+ }
934
+ },
935
+ "required": [
936
+ "run_id"
937
+ ]
938
+ }
939
+ },
940
+ {
941
+ "name": "cancel_run",
942
+ "description": "Cancel a running or queued BioMate workflow on BioMate cloud.",
943
+ "input_schema": {
944
+ "type": "object",
945
+ "properties": {
946
+ "run_id": {
947
+ "type": "string",
948
+ "description": "Run ID to cancel."
949
+ }
950
+ },
951
+ "required": [
952
+ "run_id"
953
+ ]
954
+ }
955
+ },
956
+ {
957
+ "name": "list_runs",
958
+ "description": "List the user's recent runs with status and timestamps. Filter by status or experiment.",
959
+ "input_schema": {
960
+ "type": "object",
961
+ "properties": {
962
+ "limit": {
963
+ "type": "integer",
964
+ "description": "Max runs to return (default 10).",
965
+ "default": 10
966
+ },
967
+ "status": {
968
+ "type": "string",
969
+ "enum": [
970
+ "all",
971
+ "running",
972
+ "completed",
973
+ "failed",
974
+ "pending"
975
+ ],
976
+ "default": "all"
977
+ },
978
+ "experiment_id": {
979
+ "type": "string",
980
+ "description": "Optional experiment filter."
981
+ }
982
+ }
983
+ }
984
+ },
985
+ {
986
+ "name": "preview_file",
987
+ "description": "Render a server-side preview of an output file (FASTA, VCF, CSV/TSV, image, PDF). Returns markdown + optional thumbnail PNG. Use this to show the user what an output looks like without downloading multi-GB files. For input QC of files the user is about to upload, prefer biomate_session (it runs a real QC workflow).",
988
+ "input_schema": {
989
+ "type": "object",
990
+ "properties": {
991
+ "s3_key": {
992
+ "type": "string",
993
+ "description": "The full s3_uri value (must start with 's3://') copied verbatim from a get_run output_files entry \u2014 e.g. 's3://bucket/path/file.json'. NOT a bare key; the s3:// prefix is required."
994
+ },
995
+ "run_id": {
996
+ "type": "string",
997
+ "description": "Optional run_id for context-aware parsing."
998
+ },
999
+ "max_rows": {
1000
+ "type": "integer",
1001
+ "description": "For tabular files (default 100).",
1002
+ "default": 100
1003
+ }
1004
+ },
1005
+ "required": [
1006
+ "s3_key"
1007
+ ]
1008
+ }
1009
+ },
1010
+ {
1011
+ "name": "export_report",
1012
+ "description": "Render a publication-ready report for a completed run as PDF or markdown. Includes the methods section, QC audit trail, structured findings, and figures. This is what users need for IND submissions, CRO compliance packages, and publication supplementary materials.",
1013
+ "input_schema": {
1014
+ "type": "object",
1015
+ "properties": {
1016
+ "run_id": {
1017
+ "type": "string",
1018
+ "description": "Completed run ID."
1019
+ },
1020
+ "format": {
1021
+ "type": "string",
1022
+ "enum": [
1023
+ "pdf",
1024
+ "markdown",
1025
+ "docx"
1026
+ ],
1027
+ "default": "pdf"
1028
+ },
1029
+ "sections": {
1030
+ "type": "array",
1031
+ "items": {
1032
+ "type": "string",
1033
+ "enum": [
1034
+ "methods",
1035
+ "qc",
1036
+ "findings",
1037
+ "figures",
1038
+ "appendix"
1039
+ ]
1040
+ },
1041
+ "description": "Which sections to include (default: all)."
1042
+ }
1043
+ },
1044
+ "required": [
1045
+ "run_id"
1046
+ ]
1047
+ }
1048
+ },
1049
+ {
1050
+ "name": "analyze_results",
1051
+ "description": "Ask BioMate's AI to interpret a completed run. Returns natural-language analysis: key findings, quality assessment, scientific interpretation, recommended next steps. Use after get_run when the user asks 'what does this mean?'.",
1052
+ "input_schema": {
1053
+ "type": "object",
1054
+ "properties": {
1055
+ "run_id": {
1056
+ "type": "string",
1057
+ "description": "Completed run ID."
1058
+ },
1059
+ "question": {
1060
+ "type": "string",
1061
+ "description": "Optional focused question."
1062
+ }
1063
+ },
1064
+ "required": [
1065
+ "run_id"
1066
+ ]
1067
+ }
1068
+ },
1069
+ {
1070
+ "name": "explain_error",
1071
+ "description": "Diagnose a failed run. Returns the likely root cause (genome mismatch, missing input, OOM, container pull failure, etc.) and the specific fix. Often the next step is run_workflow with corrected params.",
1072
+ "input_schema": {
1073
+ "type": "object",
1074
+ "properties": {
1075
+ "run_id": {
1076
+ "type": "string",
1077
+ "description": "Failed run ID."
1078
+ },
1079
+ "error_log": {
1080
+ "type": "string",
1081
+ "description": "Optional pasted error log."
1082
+ }
1083
+ },
1084
+ "required": [
1085
+ "run_id"
1086
+ ]
1087
+ }
1088
+ },
1089
+ {
1090
+ "name": "query_database",
1091
+ "description": "Query a biological/chemical database by accession or name. Use database='federated' to fan out across all sources simultaneously and merge results. Single-source options: uniprot, pdb, pdbe, alphafold, ncbi_gene, dbsnp, clinvar, gnomad, kegg, reactome, chebi, chembl, pubchem, bindingdb, pharos, hpo, string, pubmed.",
1092
+ "input_schema": {
1093
+ "type": "object",
1094
+ "properties": {
1095
+ "database": {
1096
+ "type": "string",
1097
+ "description": "Database to query. Use 'federated' to query all sources in parallel.",
1098
+ "enum": [
1099
+ "federated",
1100
+ "uniprot",
1101
+ "pdb",
1102
+ "pdbe",
1103
+ "alphafold",
1104
+ "ncbi_gene",
1105
+ "dbsnp",
1106
+ "clinvar",
1107
+ "gnomad",
1108
+ "kegg",
1109
+ "reactome",
1110
+ "chebi",
1111
+ "chembl",
1112
+ "pubchem",
1113
+ "bindingdb",
1114
+ "pharos",
1115
+ "hpo",
1116
+ "string",
1117
+ "pubmed"
1118
+ ],
1119
+ "default": "federated"
1120
+ },
1121
+ "query": {
1122
+ "type": "string",
1123
+ "description": "Accession, gene symbol, compound name, or free text."
1124
+ },
1125
+ "operation": {
1126
+ "type": "string",
1127
+ "description": "Search mode chosen from user intent: 'lookup' (by ID/name, default), 'similarity' (chembl/pubchem=chemical similarity over SMILES; pdb=sequence homologs over the whole PDB), or 'text' (keyword search).",
1128
+ "enum": [
1129
+ "lookup",
1130
+ "similarity",
1131
+ "text"
1132
+ ],
1133
+ "default": "lookup"
1134
+ },
1135
+ "entity_type": {
1136
+ "type": "string",
1137
+ "description": "Hint for federated routing: gene, protein, variant, compound, pathway, disease, or omit to auto-detect.",
1138
+ "enum": [
1139
+ "gene",
1140
+ "protein",
1141
+ "variant",
1142
+ "compound",
1143
+ "pathway",
1144
+ "disease"
1145
+ ]
1146
+ }
1147
+ },
1148
+ "required": [
1149
+ "query"
1150
+ ]
1151
+ }
1152
+ },
1153
+ {
1154
+ "name": "resolve_accession",
1155
+ "description": "Identify a public archive accession (GEO, SRA, ENA, DDBJ) and return the best BioMate workflow to fetch it plus pre-filled params ready for run_workflow. Examples: GSE183947 \u2192 Geo Data Connector; SRR12345 \u2192 nf-core/fetchngs. Call this before run_workflow whenever the user provides an accession instead of a file.",
1156
+ "input_schema": {
1157
+ "type": "object",
1158
+ "properties": {
1159
+ "accession": {
1160
+ "type": "string",
1161
+ "description": "Public archive accession: GSE*, GSM*, GDS*, SRR*, SRX*, SRP*, ERR*, ERX*, ERP*, DRR*, PRJNA*, PRJEB*, E-MTAB-*, E-GEOD-*"
1162
+ }
1163
+ },
1164
+ "required": [
1165
+ "accession"
1166
+ ]
1167
+ }
1168
+ },
1169
+ {
1170
+ "name": "browse_data",
1171
+ "description": "Browse a biological data repository or S3 workspace by listing files and directories at a given path. Public sources (EBI, NCBI, Ensembl, UCSC) require a path. S3 sources (biomate_workspace, user_s3) accept an optional prefix. Use before fetch_public_data to navigate to the exact file you need.",
1172
+ "input_schema": {
1173
+ "type": "object",
1174
+ "properties": {
1175
+ "source_id": {
1176
+ "type": "string",
1177
+ "enum": [
1178
+ "ebi_ftp",
1179
+ "ncbi_ftp",
1180
+ "ensembl_ftp",
1181
+ "ucsc_downloads",
1182
+ "http_public",
1183
+ "biomate_workspace",
1184
+ "user_s3"
1185
+ ],
1186
+ "description": "Data source to browse. biomate_workspace = BioMate's S3 work bucket; user_s3 = user's own S3 bucket (if configured)."
1187
+ },
1188
+ "path": {
1189
+ "type": "string",
1190
+ "description": "Directory path or S3 prefix to list (e.g. '/pub/databases/uniprot/' or 'results/run-xyz/')."
1191
+ }
1192
+ },
1193
+ "required": [
1194
+ "source_id"
1195
+ ]
1196
+ }
1197
+ },
1198
+ {
1199
+ "name": "fetch_public_data",
1200
+ "description": "Download a file from a public biological repository (EBI, NCBI, Ensembl, UCSC) into BioMate's S3 workspace and return a presigned URL and S3 URI. Use the S3 URI as a workflow input parameter. Call browse_data first to find the exact path.",
1201
+ "input_schema": {
1202
+ "type": "object",
1203
+ "properties": {
1204
+ "source_id": {
1205
+ "type": "string",
1206
+ "enum": [
1207
+ "ebi_ftp",
1208
+ "ncbi_ftp",
1209
+ "ensembl_ftp",
1210
+ "ucsc_downloads",
1211
+ "http_public"
1212
+ ],
1213
+ "description": "Data source the file comes from."
1214
+ },
1215
+ "remote_path": {
1216
+ "type": "string",
1217
+ "description": "Path on the FTP server (e.g. '/pub/databases/uniprot/.../uniprot_sprot.fasta.gz')."
1218
+ },
1219
+ "url": {
1220
+ "type": "string",
1221
+ "description": "For http_public source: full HTTPS URL of the file to download."
1222
+ }
1223
+ },
1224
+ "required": [
1225
+ "source_id"
1226
+ ]
1227
+ }
1228
+ },
1229
+ {
1230
+ "name": "recall_memory",
1231
+ "description": "Retrieve relevant prior context for the current goal: past runs on similar inputs, validated procedures, findings tagged by the user, learned parameter preferences. Call before biomate_session for repeat users \u2014 it dramatically improves param auto-fill and avoids re-running work.",
1232
+ "input_schema": {
1233
+ "type": "object",
1234
+ "properties": {
1235
+ "query": {
1236
+ "type": "string",
1237
+ "description": "What to recall \u2014 usually the user's goal."
1238
+ },
1239
+ "scope": {
1240
+ "type": "string",
1241
+ "enum": [
1242
+ "runs",
1243
+ "findings",
1244
+ "procedures",
1245
+ "all"
1246
+ ],
1247
+ "default": "all"
1248
+ },
1249
+ "limit": {
1250
+ "type": "integer",
1251
+ "default": 5
1252
+ }
1253
+ },
1254
+ "required": [
1255
+ "query"
1256
+ ]
1257
+ }
1258
+ },
1259
+ {
1260
+ "name": "search_literature",
1261
+ "description": "Search published scientific literature across PubMed, EuropePMC, Semantic Scholar, and OpenAlex. Uses an iterative depth-loop: starts broad, then drills into related topics until min_results are found or max_depth is reached. Returns ranked papers with title, authors, year, DOI, abstract snippet, and citation count. Best for: finding prior art, understanding a disease mechanism, locating validation datasets, or reviewing methods before running a workflow.",
1262
+ "input_schema": {
1263
+ "type": "object",
1264
+ "properties": {
1265
+ "query": {
1266
+ "type": "string",
1267
+ "description": "Natural-language search query. Be specific: include organism, assay type, or drug name. E.g. 'DESeq2 bulk RNA-seq breast cancer ER+ 2020-2024'."
1268
+ },
1269
+ "max_depth": {
1270
+ "type": "integer",
1271
+ "default": 2,
1272
+ "description": "How many iterative refinement rounds to run (1\u20134). Higher = more thorough, slower."
1273
+ },
1274
+ "min_results": {
1275
+ "type": "integer",
1276
+ "default": 10,
1277
+ "description": "Stop when at least this many papers are found."
1278
+ }
1279
+ },
1280
+ "required": [
1281
+ "query"
1282
+ ]
1283
+ }
1284
+ },
1285
+ {
1286
+ "name": "upload_file",
1287
+ "description": "Get a one-shot signed S3 PUT URL so the host can upload a local file directly to BioMate's data plane without proxying bytes through the chat transport. For files >5MB this is mandatory; for small text payloads, inline strings are fine.\n\n**Returns** `{upload_url, s3_key}` \u2014 two fields:\n- `upload_url`: a presigned HTTPS URL. You MUST upload the file bytes to it with an HTTP PUT before calling any other tool: `curl -X PUT -T <local_path> \"<upload_url>\"`\n- `s3_key`: a bare S3 object key (e.g. `users/42/uploads/uuid-sample.fastq.gz`). Pass this value directly into the `inputs` dict of `biomate_session` or as a `params` value in `run_workflow` \u2014 BioMate resolves the bucket internally. Example: `inputs={\"fastq_file\": s3_key}` or `params={\"input\": s3_key}`.\n\n**Full upload workflow:**\n1. `upload_file(filename='sample.fastq.gz', size_bytes=52000000)` \u2192 get `upload_url` + `s3_key`\n2. `curl -X PUT -T sample.fastq.gz \"<upload_url>\"` (or equivalent HTTP PUT \u2014 no auth header needed)\n3. `biomate_session(goal='Run RNA-seq DE on the uploaded FASTQs, human GRCh38', inputs={\"fastq_file\": s3_key})`",
1288
+ "input_schema": {
1289
+ "type": "object",
1290
+ "properties": {
1291
+ "filename": {
1292
+ "type": "string",
1293
+ "description": "Original filename (sets content-type and extension)."
1294
+ },
1295
+ "size_bytes": {
1296
+ "type": "integer",
1297
+ "description": "File size for quota check."
1298
+ },
1299
+ "content_type": {
1300
+ "type": "string",
1301
+ "description": "MIME type (auto-detected from filename if omitted)."
1302
+ }
1303
+ },
1304
+ "required": [
1305
+ "filename"
1306
+ ]
1307
+ }
1308
+ }
1309
+ ],
1310
+ "openai": [
1311
+ {
1312
+ "type": "function",
1313
+ "function": {
1314
+ "name": "biomate_session",
1315
+ "description": "**Primary entry point \u2014 use this for 90% of requests.** Run a complete BioMate scientific session from a natural-language goal. BioMate selects the right workflow from 2,455 indexed pipelines, pre-fills parameters from your goal text, executes on BioMate cloud, handles QC gates with auto-loop remediation, and produces structured findings. While running, the tool streams real-time progress (phase started, step completed, QC gate, auto-loop remediation, finding) back to the host. Returns a final run summary, a deep link to the live results panel, and the report URL. \n\n**How to write the `goal` parameter** \u2014 plain English, one to three sentences:\n\u2022 Include the *what*: analysis type + subject (e.g. 'ADMET screening', 'RNA-seq DE', 'variant calling')\n\u2022 Include *data location*: inline SMILES/sequences, S3 paths, accession numbers, or upload first with upload_file\n\u2022 Include key *parameters* that matter: organism, library type, comparisons, thresholds\n\u2022 You can omit anything BioMate can infer (it will ask if genuinely ambiguous)\n\n**Good examples:**\n 'Screen aspirin (CC(=O)Oc1ccccc1C(=O)O) and caffeine (Cn1cnc2c1c(=O)n(c(=O)n2C)C) for hERG inhibition, CYP3A4, and oral bioavailability'\n 'RNA-seq differential expression on s3://lab-bucket/exp42/fastqs/ \u2014 human GRCh38, dUTP strand-specific, treated (n=3) vs control (n=3), FDR 0.05'\n 'Whole-genome variant calling on the uploaded FASTQ pair, GRCh38, GATK HaplotypeCaller, germline mode'\n 'Run homogeneous 3D refinement in CryoSPARC on s3://cryo/job042/, C2 symmetry, box size 256'\n 'Fetch GSE183947 from GEO and run the same RNA-seq pipeline'\n\nUse run_workflow instead when the user wants to call a specific workflow by ID with explicit parameter control.",
1316
+ "parameters": {
1317
+ "type": "object",
1318
+ "properties": {
1319
+ "goal": {
1320
+ "type": "string",
1321
+ "description": "Natural-language scientific goal. Include: analysis type, data location (inline SMILES/sequences, s3:// paths, or GEO/SRA accession numbers), and key parameters (organism, comparisons, thresholds). BioMate infers the rest. Examples: 'Screen these 5 SMILES for hERG IC50 and CYP3A4 inhibition', 'RNA-seq DE on s3://bucket/exp1/ human GRCh38 paired-end treated vs control', 'Fetch GSE183947 and run differential expression'."
1322
+ },
1323
+ "inputs": {
1324
+ "type": "object",
1325
+ "description": "Optional structured inputs passed alongside `goal`. BioMate's inner AI sees these as a JSON block appended to the goal text and merges them with any parameters it extracts from the goal string. Use this to pass: S3 keys from upload_file, sequences, SMILES lists, accession numbers, or explicit parameter overrides. Examples:\n After upload_file: {\"fastq_file\": \"users/42/uploads/uuid-sample.fastq.gz\"} (use the s3_key value returned by upload_file)\n SMILES list: {\"smiles_list\": [\"CC(=O)Oc1ccccc1C(=O)O\"], \"organism\": \"human\"}\n Parameter override: {\"genome\": \"GRCh38\", \"aligner\": \"STAR\", \"fdr\": 0.05}",
1326
+ "additionalProperties": true
1327
+ },
1328
+ "experiment_id": {
1329
+ "type": "string",
1330
+ "description": "Optional experiment to attach this run to (from recall_memory)."
1331
+ },
1332
+ "stream": {
1333
+ "type": "boolean",
1334
+ "description": "Emit progress notifications during execution. Default true. Set false for hosts without MCP notification support \u2014 then poll with get_run.",
1335
+ "default": true
1336
+ }
1337
+ },
1338
+ "required": [
1339
+ "goal"
1340
+ ]
1341
+ }
1342
+ }
1343
+ },
1344
+ {
1345
+ "type": "function",
1346
+ "function": {
1347
+ "name": "search_workflow",
1348
+ "description": "Search the BioMate workflow catalog (2,455 indexed workflows across 34 domains) by natural language. Returns ranked workflow cards with id, name, domain, one-line description, and estimated BioMate cloud cost. Use this when the user wants to pick a workflow explicitly; otherwise prefer biomate_session.",
1349
+ "parameters": {
1350
+ "type": "object",
1351
+ "properties": {
1352
+ "query": {
1353
+ "type": "string",
1354
+ "description": "Natural language description of the analysis."
1355
+ },
1356
+ "limit": {
1357
+ "type": "integer",
1358
+ "description": "Max results (default 5, max 20).",
1359
+ "default": 5
1360
+ },
1361
+ "domain": {
1362
+ "type": "string",
1363
+ "description": "Optional domain filter: transcriptomics, genomics, proteomics, drug_discovery, cryo_em, etc."
1364
+ }
1365
+ },
1366
+ "required": [
1367
+ "query"
1368
+ ]
1369
+ }
1370
+ }
1371
+ },
1372
+ {
1373
+ "type": "function",
1374
+ "function": {
1375
+ "name": "get_workflow_spec",
1376
+ "description": "Return the full specification for a workflow: required + optional parameters (with types and allowed values), default QC profile and thresholds, expected input files, estimated cost and runtime, and any license requirements. Call this before run_workflow when the user wants explicit parameter control.",
1377
+ "parameters": {
1378
+ "type": "object",
1379
+ "properties": {
1380
+ "workflow_id": {
1381
+ "type": "string",
1382
+ "description": "Workflow ID from search_workflow."
1383
+ }
1384
+ },
1385
+ "required": [
1386
+ "workflow_id"
1387
+ ]
1388
+ }
1389
+ }
1390
+ },
1391
+ {
1392
+ "type": "function",
1393
+ "function": {
1394
+ "name": "run_workflow",
1395
+ "description": "Execute a specific BioMate workflow on BioMate cloud with explicit parameters. Returns a run_id immediately. If stream=true, also emits progress notifications until the run terminates (same events as biomate_session). Use biomate_session instead when the user gave you a natural-language goal.",
1396
+ "parameters": {
1397
+ "type": "object",
1398
+ "properties": {
1399
+ "workflow_id": {
1400
+ "type": "string",
1401
+ "description": "Workflow ID from search_workflow."
1402
+ },
1403
+ "params": {
1404
+ "type": "object",
1405
+ "description": "Parameter dict. Required params come from get_workflow_spec.",
1406
+ "additionalProperties": true
1407
+ },
1408
+ "experiment_id": {
1409
+ "type": "string",
1410
+ "description": "Optional experiment to attach to."
1411
+ },
1412
+ "stream": {
1413
+ "type": "boolean",
1414
+ "description": "Emit progress notifications while running. Default false.",
1415
+ "default": false
1416
+ }
1417
+ },
1418
+ "required": [
1419
+ "workflow_id"
1420
+ ]
1421
+ }
1422
+ }
1423
+ },
1424
+ {
1425
+ "type": "function",
1426
+ "function": {
1427
+ "name": "watch_run",
1428
+ "description": "Stream real-time progress for a running BioMate workflow and return full results when done. Emits MCP notifications/progress for every phase start/complete, step update, and QC gate. When the run finishes, automatically fetches output files (with download URLs) and AI findings. Use after run_workflow (non-streaming) to watch a submitted run. Does not require re-submitting \u2014 takes an existing run_id.",
1429
+ "parameters": {
1430
+ "type": "object",
1431
+ "properties": {
1432
+ "run_id": {
1433
+ "type": "string",
1434
+ "description": "Run ID returned by run_workflow."
1435
+ }
1436
+ },
1437
+ "required": [
1438
+ "run_id"
1439
+ ]
1440
+ }
1441
+ }
1442
+ },
1443
+ {
1444
+ "type": "function",
1445
+ "function": {
1446
+ "name": "get_run",
1447
+ "description": "Return everything about a run in one call: status (pending|running|completed|failed), per-phase and per-step progress with timestamps, output files with download URLs, structured findings, QC gate results, and any auto-loop remediations applied. Replaces get_run_status + get_run_results + step-level findings polling.",
1448
+ "parameters": {
1449
+ "type": "object",
1450
+ "properties": {
1451
+ "run_id": {
1452
+ "type": "string",
1453
+ "description": "Run ID."
1454
+ },
1455
+ "include_findings": {
1456
+ "type": "boolean",
1457
+ "description": "Include structured findings cards (default true).",
1458
+ "default": true
1459
+ }
1460
+ },
1461
+ "required": [
1462
+ "run_id"
1463
+ ]
1464
+ }
1465
+ }
1466
+ },
1467
+ {
1468
+ "type": "function",
1469
+ "function": {
1470
+ "name": "cancel_run",
1471
+ "description": "Cancel a running or queued BioMate workflow on BioMate cloud.",
1472
+ "parameters": {
1473
+ "type": "object",
1474
+ "properties": {
1475
+ "run_id": {
1476
+ "type": "string",
1477
+ "description": "Run ID to cancel."
1478
+ }
1479
+ },
1480
+ "required": [
1481
+ "run_id"
1482
+ ]
1483
+ }
1484
+ }
1485
+ },
1486
+ {
1487
+ "type": "function",
1488
+ "function": {
1489
+ "name": "list_runs",
1490
+ "description": "List the user's recent runs with status and timestamps. Filter by status or experiment.",
1491
+ "parameters": {
1492
+ "type": "object",
1493
+ "properties": {
1494
+ "limit": {
1495
+ "type": "integer",
1496
+ "description": "Max runs to return (default 10).",
1497
+ "default": 10
1498
+ },
1499
+ "status": {
1500
+ "type": "string",
1501
+ "enum": [
1502
+ "all",
1503
+ "running",
1504
+ "completed",
1505
+ "failed",
1506
+ "pending"
1507
+ ],
1508
+ "default": "all"
1509
+ },
1510
+ "experiment_id": {
1511
+ "type": "string",
1512
+ "description": "Optional experiment filter."
1513
+ }
1514
+ }
1515
+ }
1516
+ }
1517
+ },
1518
+ {
1519
+ "type": "function",
1520
+ "function": {
1521
+ "name": "preview_file",
1522
+ "description": "Render a server-side preview of an output file (FASTA, VCF, CSV/TSV, image, PDF). Returns markdown + optional thumbnail PNG. Use this to show the user what an output looks like without downloading multi-GB files. For input QC of files the user is about to upload, prefer biomate_session (it runs a real QC workflow).",
1523
+ "parameters": {
1524
+ "type": "object",
1525
+ "properties": {
1526
+ "s3_key": {
1527
+ "type": "string",
1528
+ "description": "The full s3_uri value (must start with 's3://') copied verbatim from a get_run output_files entry \u2014 e.g. 's3://bucket/path/file.json'. NOT a bare key; the s3:// prefix is required."
1529
+ },
1530
+ "run_id": {
1531
+ "type": "string",
1532
+ "description": "Optional run_id for context-aware parsing."
1533
+ },
1534
+ "max_rows": {
1535
+ "type": "integer",
1536
+ "description": "For tabular files (default 100).",
1537
+ "default": 100
1538
+ }
1539
+ },
1540
+ "required": [
1541
+ "s3_key"
1542
+ ]
1543
+ }
1544
+ }
1545
+ },
1546
+ {
1547
+ "type": "function",
1548
+ "function": {
1549
+ "name": "export_report",
1550
+ "description": "Render a publication-ready report for a completed run as PDF or markdown. Includes the methods section, QC audit trail, structured findings, and figures. This is what users need for IND submissions, CRO compliance packages, and publication supplementary materials.",
1551
+ "parameters": {
1552
+ "type": "object",
1553
+ "properties": {
1554
+ "run_id": {
1555
+ "type": "string",
1556
+ "description": "Completed run ID."
1557
+ },
1558
+ "format": {
1559
+ "type": "string",
1560
+ "enum": [
1561
+ "pdf",
1562
+ "markdown",
1563
+ "docx"
1564
+ ],
1565
+ "default": "pdf"
1566
+ },
1567
+ "sections": {
1568
+ "type": "array",
1569
+ "items": {
1570
+ "type": "string",
1571
+ "enum": [
1572
+ "methods",
1573
+ "qc",
1574
+ "findings",
1575
+ "figures",
1576
+ "appendix"
1577
+ ]
1578
+ },
1579
+ "description": "Which sections to include (default: all)."
1580
+ }
1581
+ },
1582
+ "required": [
1583
+ "run_id"
1584
+ ]
1585
+ }
1586
+ }
1587
+ },
1588
+ {
1589
+ "type": "function",
1590
+ "function": {
1591
+ "name": "analyze_results",
1592
+ "description": "Ask BioMate's AI to interpret a completed run. Returns natural-language analysis: key findings, quality assessment, scientific interpretation, recommended next steps. Use after get_run when the user asks 'what does this mean?'.",
1593
+ "parameters": {
1594
+ "type": "object",
1595
+ "properties": {
1596
+ "run_id": {
1597
+ "type": "string",
1598
+ "description": "Completed run ID."
1599
+ },
1600
+ "question": {
1601
+ "type": "string",
1602
+ "description": "Optional focused question."
1603
+ }
1604
+ },
1605
+ "required": [
1606
+ "run_id"
1607
+ ]
1608
+ }
1609
+ }
1610
+ },
1611
+ {
1612
+ "type": "function",
1613
+ "function": {
1614
+ "name": "explain_error",
1615
+ "description": "Diagnose a failed run. Returns the likely root cause (genome mismatch, missing input, OOM, container pull failure, etc.) and the specific fix. Often the next step is run_workflow with corrected params.",
1616
+ "parameters": {
1617
+ "type": "object",
1618
+ "properties": {
1619
+ "run_id": {
1620
+ "type": "string",
1621
+ "description": "Failed run ID."
1622
+ },
1623
+ "error_log": {
1624
+ "type": "string",
1625
+ "description": "Optional pasted error log."
1626
+ }
1627
+ },
1628
+ "required": [
1629
+ "run_id"
1630
+ ]
1631
+ }
1632
+ }
1633
+ },
1634
+ {
1635
+ "type": "function",
1636
+ "function": {
1637
+ "name": "query_database",
1638
+ "description": "Query a biological/chemical database by accession or name. Use database='federated' to fan out across all sources simultaneously and merge results. Single-source options: uniprot, pdb, pdbe, alphafold, ncbi_gene, dbsnp, clinvar, gnomad, kegg, reactome, chebi, chembl, pubchem, bindingdb, pharos, hpo, string, pubmed.",
1639
+ "parameters": {
1640
+ "type": "object",
1641
+ "properties": {
1642
+ "database": {
1643
+ "type": "string",
1644
+ "description": "Database to query. Use 'federated' to query all sources in parallel.",
1645
+ "enum": [
1646
+ "federated",
1647
+ "uniprot",
1648
+ "pdb",
1649
+ "pdbe",
1650
+ "alphafold",
1651
+ "ncbi_gene",
1652
+ "dbsnp",
1653
+ "clinvar",
1654
+ "gnomad",
1655
+ "kegg",
1656
+ "reactome",
1657
+ "chebi",
1658
+ "chembl",
1659
+ "pubchem",
1660
+ "bindingdb",
1661
+ "pharos",
1662
+ "hpo",
1663
+ "string",
1664
+ "pubmed"
1665
+ ],
1666
+ "default": "federated"
1667
+ },
1668
+ "query": {
1669
+ "type": "string",
1670
+ "description": "Accession, gene symbol, compound name, or free text."
1671
+ },
1672
+ "operation": {
1673
+ "type": "string",
1674
+ "description": "Search mode chosen from user intent: 'lookup' (by ID/name, default), 'similarity' (chembl/pubchem=chemical similarity over SMILES; pdb=sequence homologs over the whole PDB), or 'text' (keyword search).",
1675
+ "enum": [
1676
+ "lookup",
1677
+ "similarity",
1678
+ "text"
1679
+ ],
1680
+ "default": "lookup"
1681
+ },
1682
+ "entity_type": {
1683
+ "type": "string",
1684
+ "description": "Hint for federated routing: gene, protein, variant, compound, pathway, disease, or omit to auto-detect.",
1685
+ "enum": [
1686
+ "gene",
1687
+ "protein",
1688
+ "variant",
1689
+ "compound",
1690
+ "pathway",
1691
+ "disease"
1692
+ ]
1693
+ }
1694
+ },
1695
+ "required": [
1696
+ "query"
1697
+ ]
1698
+ }
1699
+ }
1700
+ },
1701
+ {
1702
+ "type": "function",
1703
+ "function": {
1704
+ "name": "resolve_accession",
1705
+ "description": "Identify a public archive accession (GEO, SRA, ENA, DDBJ) and return the best BioMate workflow to fetch it plus pre-filled params ready for run_workflow. Examples: GSE183947 \u2192 Geo Data Connector; SRR12345 \u2192 nf-core/fetchngs. Call this before run_workflow whenever the user provides an accession instead of a file.",
1706
+ "parameters": {
1707
+ "type": "object",
1708
+ "properties": {
1709
+ "accession": {
1710
+ "type": "string",
1711
+ "description": "Public archive accession: GSE*, GSM*, GDS*, SRR*, SRX*, SRP*, ERR*, ERX*, ERP*, DRR*, PRJNA*, PRJEB*, E-MTAB-*, E-GEOD-*"
1712
+ }
1713
+ },
1714
+ "required": [
1715
+ "accession"
1716
+ ]
1717
+ }
1718
+ }
1719
+ },
1720
+ {
1721
+ "type": "function",
1722
+ "function": {
1723
+ "name": "browse_data",
1724
+ "description": "Browse a biological data repository or S3 workspace by listing files and directories at a given path. Public sources (EBI, NCBI, Ensembl, UCSC) require a path. S3 sources (biomate_workspace, user_s3) accept an optional prefix. Use before fetch_public_data to navigate to the exact file you need.",
1725
+ "parameters": {
1726
+ "type": "object",
1727
+ "properties": {
1728
+ "source_id": {
1729
+ "type": "string",
1730
+ "enum": [
1731
+ "ebi_ftp",
1732
+ "ncbi_ftp",
1733
+ "ensembl_ftp",
1734
+ "ucsc_downloads",
1735
+ "http_public",
1736
+ "biomate_workspace",
1737
+ "user_s3"
1738
+ ],
1739
+ "description": "Data source to browse. biomate_workspace = BioMate's S3 work bucket; user_s3 = user's own S3 bucket (if configured)."
1740
+ },
1741
+ "path": {
1742
+ "type": "string",
1743
+ "description": "Directory path or S3 prefix to list (e.g. '/pub/databases/uniprot/' or 'results/run-xyz/')."
1744
+ }
1745
+ },
1746
+ "required": [
1747
+ "source_id"
1748
+ ]
1749
+ }
1750
+ }
1751
+ },
1752
+ {
1753
+ "type": "function",
1754
+ "function": {
1755
+ "name": "fetch_public_data",
1756
+ "description": "Download a file from a public biological repository (EBI, NCBI, Ensembl, UCSC) into BioMate's S3 workspace and return a presigned URL and S3 URI. Use the S3 URI as a workflow input parameter. Call browse_data first to find the exact path.",
1757
+ "parameters": {
1758
+ "type": "object",
1759
+ "properties": {
1760
+ "source_id": {
1761
+ "type": "string",
1762
+ "enum": [
1763
+ "ebi_ftp",
1764
+ "ncbi_ftp",
1765
+ "ensembl_ftp",
1766
+ "ucsc_downloads",
1767
+ "http_public"
1768
+ ],
1769
+ "description": "Data source the file comes from."
1770
+ },
1771
+ "remote_path": {
1772
+ "type": "string",
1773
+ "description": "Path on the FTP server (e.g. '/pub/databases/uniprot/.../uniprot_sprot.fasta.gz')."
1774
+ },
1775
+ "url": {
1776
+ "type": "string",
1777
+ "description": "For http_public source: full HTTPS URL of the file to download."
1778
+ }
1779
+ },
1780
+ "required": [
1781
+ "source_id"
1782
+ ]
1783
+ }
1784
+ }
1785
+ },
1786
+ {
1787
+ "type": "function",
1788
+ "function": {
1789
+ "name": "recall_memory",
1790
+ "description": "Retrieve relevant prior context for the current goal: past runs on similar inputs, validated procedures, findings tagged by the user, learned parameter preferences. Call before biomate_session for repeat users \u2014 it dramatically improves param auto-fill and avoids re-running work.",
1791
+ "parameters": {
1792
+ "type": "object",
1793
+ "properties": {
1794
+ "query": {
1795
+ "type": "string",
1796
+ "description": "What to recall \u2014 usually the user's goal."
1797
+ },
1798
+ "scope": {
1799
+ "type": "string",
1800
+ "enum": [
1801
+ "runs",
1802
+ "findings",
1803
+ "procedures",
1804
+ "all"
1805
+ ],
1806
+ "default": "all"
1807
+ },
1808
+ "limit": {
1809
+ "type": "integer",
1810
+ "default": 5
1811
+ }
1812
+ },
1813
+ "required": [
1814
+ "query"
1815
+ ]
1816
+ }
1817
+ }
1818
+ },
1819
+ {
1820
+ "type": "function",
1821
+ "function": {
1822
+ "name": "search_literature",
1823
+ "description": "Search published scientific literature across PubMed, EuropePMC, Semantic Scholar, and OpenAlex. Uses an iterative depth-loop: starts broad, then drills into related topics until min_results are found or max_depth is reached. Returns ranked papers with title, authors, year, DOI, abstract snippet, and citation count. Best for: finding prior art, understanding a disease mechanism, locating validation datasets, or reviewing methods before running a workflow.",
1824
+ "parameters": {
1825
+ "type": "object",
1826
+ "properties": {
1827
+ "query": {
1828
+ "type": "string",
1829
+ "description": "Natural-language search query. Be specific: include organism, assay type, or drug name. E.g. 'DESeq2 bulk RNA-seq breast cancer ER+ 2020-2024'."
1830
+ },
1831
+ "max_depth": {
1832
+ "type": "integer",
1833
+ "default": 2,
1834
+ "description": "How many iterative refinement rounds to run (1\u20134). Higher = more thorough, slower."
1835
+ },
1836
+ "min_results": {
1837
+ "type": "integer",
1838
+ "default": 10,
1839
+ "description": "Stop when at least this many papers are found."
1840
+ }
1841
+ },
1842
+ "required": [
1843
+ "query"
1844
+ ]
1845
+ }
1846
+ }
1847
+ },
1848
+ {
1849
+ "type": "function",
1850
+ "function": {
1851
+ "name": "upload_file",
1852
+ "description": "Get a one-shot signed S3 PUT URL so the host can upload a local file directly to BioMate's data plane without proxying bytes through the chat transport. For files >5MB this is mandatory; for small text payloads, inline strings are fine.\n\n**Returns** `{upload_url, s3_key}` \u2014 two fields:\n- `upload_url`: a presigned HTTPS URL. You MUST upload the file bytes to it with an HTTP PUT before calling any other tool: `curl -X PUT -T <local_path> \"<upload_url>\"`\n- `s3_key`: a bare S3 object key (e.g. `users/42/uploads/uuid-sample.fastq.gz`). Pass this value directly into the `inputs` dict of `biomate_session` or as a `params` value in `run_workflow` \u2014 BioMate resolves the bucket internally. Example: `inputs={\"fastq_file\": s3_key}` or `params={\"input\": s3_key}`.\n\n**Full upload workflow:**\n1. `upload_file(filename='sample.fastq.gz', size_bytes=52000000)` \u2192 get `upload_url` + `s3_key`\n2. `curl -X PUT -T sample.fastq.gz \"<upload_url>\"` (or equivalent HTTP PUT \u2014 no auth header needed)\n3. `biomate_session(goal='Run RNA-seq DE on the uploaded FASTQs, human GRCh38', inputs={\"fastq_file\": s3_key})`",
1853
+ "parameters": {
1854
+ "type": "object",
1855
+ "properties": {
1856
+ "filename": {
1857
+ "type": "string",
1858
+ "description": "Original filename (sets content-type and extension)."
1859
+ },
1860
+ "size_bytes": {
1861
+ "type": "integer",
1862
+ "description": "File size for quota check."
1863
+ },
1864
+ "content_type": {
1865
+ "type": "string",
1866
+ "description": "MIME type (auto-detected from filename if omitted)."
1867
+ }
1868
+ },
1869
+ "required": [
1870
+ "filename"
1871
+ ]
1872
+ }
1873
+ }
1874
+ }
1875
+ ],
1876
+ "lite": {
1877
+ "mcp": [
1878
+ {
1879
+ "name": "biomate_session",
1880
+ "title": "Run BioMate Session",
1881
+ "description": "**Primary entry point \u2014 use this for 90% of requests.** Run a complete BioMate scientific session from a natural-language goal. BioMate selects the right workflow from 2,455 indexed pipelines, pre-fills parameters from your goal text, executes on BioMate cloud, handles QC gates with auto-loop remediation, and produces structured findings. While running, the tool streams real-time progress (phase started, step completed, QC gate, auto-loop remediation, finding) back to the host. Returns a final run summary, a deep link to the live results panel, and the report URL. \n\n**How to write the `goal` parameter** \u2014 plain English, one to three sentences:\n\u2022 Include the *what*: analysis type + subject (e.g. 'ADMET screening', 'RNA-seq DE', 'variant calling')\n\u2022 Include *data location*: inline SMILES/sequences, S3 paths, accession numbers, or upload first with upload_file\n\u2022 Include key *parameters* that matter: organism, library type, comparisons, thresholds\n\u2022 You can omit anything BioMate can infer (it will ask if genuinely ambiguous)\n\n**Good examples:**\n 'Screen aspirin (CC(=O)Oc1ccccc1C(=O)O) and caffeine (Cn1cnc2c1c(=O)n(c(=O)n2C)C) for hERG inhibition, CYP3A4, and oral bioavailability'\n 'RNA-seq differential expression on s3://lab-bucket/exp42/fastqs/ \u2014 human GRCh38, dUTP strand-specific, treated (n=3) vs control (n=3), FDR 0.05'\n 'Whole-genome variant calling on the uploaded FASTQ pair, GRCh38, GATK HaplotypeCaller, germline mode'\n 'Run homogeneous 3D refinement in CryoSPARC on s3://cryo/job042/, C2 symmetry, box size 256'\n 'Fetch GSE183947 from GEO and run the same RNA-seq pipeline'\n\nUse run_workflow instead when the user wants to call a specific workflow by ID with explicit parameter control.",
1882
+ "inputSchema": {
1883
+ "type": "object",
1884
+ "properties": {
1885
+ "goal": {
1886
+ "type": "string",
1887
+ "description": "Natural-language scientific goal. Include: analysis type, data location (inline SMILES/sequences, s3:// paths, or GEO/SRA accession numbers), and key parameters (organism, comparisons, thresholds). BioMate infers the rest. Examples: 'Screen these 5 SMILES for hERG IC50 and CYP3A4 inhibition', 'RNA-seq DE on s3://bucket/exp1/ human GRCh38 paired-end treated vs control', 'Fetch GSE183947 and run differential expression'."
1888
+ },
1889
+ "inputs": {
1890
+ "type": "object",
1891
+ "description": "Optional structured inputs passed alongside `goal`. BioMate's inner AI sees these as a JSON block appended to the goal text and merges them with any parameters it extracts from the goal string. Use this to pass: S3 keys from upload_file, sequences, SMILES lists, accession numbers, or explicit parameter overrides. Examples:\n After upload_file: {\"fastq_file\": \"users/42/uploads/uuid-sample.fastq.gz\"} (use the s3_key value returned by upload_file)\n SMILES list: {\"smiles_list\": [\"CC(=O)Oc1ccccc1C(=O)O\"], \"organism\": \"human\"}\n Parameter override: {\"genome\": \"GRCh38\", \"aligner\": \"STAR\", \"fdr\": 0.05}",
1892
+ "additionalProperties": true
1893
+ },
1894
+ "experiment_id": {
1895
+ "type": "string",
1896
+ "description": "Optional experiment to attach this run to (from recall_memory)."
1897
+ },
1898
+ "stream": {
1899
+ "type": "boolean",
1900
+ "description": "Emit progress notifications during execution. Default true. Set false for hosts without MCP notification support \u2014 then poll with get_run.",
1901
+ "default": true
1902
+ }
1903
+ },
1904
+ "required": [
1905
+ "goal"
1906
+ ]
1907
+ },
1908
+ "annotations": {
1909
+ "title": "Run BioMate Session",
1910
+ "readOnlyHint": false,
1911
+ "destructiveHint": false,
1912
+ "idempotentHint": false,
1913
+ "openWorldHint": true
1914
+ }
1915
+ },
1916
+ {
1917
+ "name": "export_report",
1918
+ "title": "Export Report",
1919
+ "description": "Render a publication-ready report for a completed run as PDF or markdown. Includes the methods section, QC audit trail, structured findings, and figures. This is what users need for IND submissions, CRO compliance packages, and publication supplementary materials.",
1920
+ "inputSchema": {
1921
+ "type": "object",
1922
+ "properties": {
1923
+ "run_id": {
1924
+ "type": "string",
1925
+ "description": "Completed run ID."
1926
+ },
1927
+ "format": {
1928
+ "type": "string",
1929
+ "enum": [
1930
+ "pdf",
1931
+ "markdown",
1932
+ "docx"
1933
+ ],
1934
+ "default": "pdf"
1935
+ },
1936
+ "sections": {
1937
+ "type": "array",
1938
+ "items": {
1939
+ "type": "string",
1940
+ "enum": [
1941
+ "methods",
1942
+ "qc",
1943
+ "findings",
1944
+ "figures",
1945
+ "appendix"
1946
+ ]
1947
+ },
1948
+ "description": "Which sections to include (default: all)."
1949
+ }
1950
+ },
1951
+ "required": [
1952
+ "run_id"
1953
+ ]
1954
+ },
1955
+ "annotations": {
1956
+ "title": "Export Report",
1957
+ "readOnlyHint": false,
1958
+ "destructiveHint": false,
1959
+ "idempotentHint": true,
1960
+ "openWorldHint": false
1961
+ }
1962
+ },
1963
+ {
1964
+ "name": "upload_file",
1965
+ "title": "Upload File",
1966
+ "description": "Get a one-shot signed S3 PUT URL so the host can upload a local file directly to BioMate's data plane without proxying bytes through the chat transport. For files >5MB this is mandatory; for small text payloads, inline strings are fine.\n\n**Returns** `{upload_url, s3_key}` \u2014 two fields:\n- `upload_url`: a presigned HTTPS URL. You MUST upload the file bytes to it with an HTTP PUT before calling any other tool: `curl -X PUT -T <local_path> \"<upload_url>\"`\n- `s3_key`: a bare S3 object key (e.g. `users/42/uploads/uuid-sample.fastq.gz`). Pass this value directly into the `inputs` dict of `biomate_session` or as a `params` value in `run_workflow` \u2014 BioMate resolves the bucket internally. Example: `inputs={\"fastq_file\": s3_key}` or `params={\"input\": s3_key}`.\n\n**Full upload workflow:**\n1. `upload_file(filename='sample.fastq.gz', size_bytes=52000000)` \u2192 get `upload_url` + `s3_key`\n2. `curl -X PUT -T sample.fastq.gz \"<upload_url>\"` (or equivalent HTTP PUT \u2014 no auth header needed)\n3. `biomate_session(goal='Run RNA-seq DE on the uploaded FASTQs, human GRCh38', inputs={\"fastq_file\": s3_key})`",
1967
+ "inputSchema": {
1968
+ "type": "object",
1969
+ "properties": {
1970
+ "filename": {
1971
+ "type": "string",
1972
+ "description": "Original filename (sets content-type and extension)."
1973
+ },
1974
+ "size_bytes": {
1975
+ "type": "integer",
1976
+ "description": "File size for quota check."
1977
+ },
1978
+ "content_type": {
1979
+ "type": "string",
1980
+ "description": "MIME type (auto-detected from filename if omitted)."
1981
+ }
1982
+ },
1983
+ "required": [
1984
+ "filename"
1985
+ ]
1986
+ },
1987
+ "annotations": {
1988
+ "title": "Upload File",
1989
+ "readOnlyHint": false,
1990
+ "destructiveHint": false,
1991
+ "idempotentHint": false,
1992
+ "openWorldHint": false
1993
+ }
1994
+ }
1995
+ ],
1996
+ "anthropic": [
1997
+ {
1998
+ "name": "biomate_session",
1999
+ "description": "**Primary entry point \u2014 use this for 90% of requests.** Run a complete BioMate scientific session from a natural-language goal. BioMate selects the right workflow from 2,455 indexed pipelines, pre-fills parameters from your goal text, executes on BioMate cloud, handles QC gates with auto-loop remediation, and produces structured findings. While running, the tool streams real-time progress (phase started, step completed, QC gate, auto-loop remediation, finding) back to the host. Returns a final run summary, a deep link to the live results panel, and the report URL. \n\n**How to write the `goal` parameter** \u2014 plain English, one to three sentences:\n\u2022 Include the *what*: analysis type + subject (e.g. 'ADMET screening', 'RNA-seq DE', 'variant calling')\n\u2022 Include *data location*: inline SMILES/sequences, S3 paths, accession numbers, or upload first with upload_file\n\u2022 Include key *parameters* that matter: organism, library type, comparisons, thresholds\n\u2022 You can omit anything BioMate can infer (it will ask if genuinely ambiguous)\n\n**Good examples:**\n 'Screen aspirin (CC(=O)Oc1ccccc1C(=O)O) and caffeine (Cn1cnc2c1c(=O)n(c(=O)n2C)C) for hERG inhibition, CYP3A4, and oral bioavailability'\n 'RNA-seq differential expression on s3://lab-bucket/exp42/fastqs/ \u2014 human GRCh38, dUTP strand-specific, treated (n=3) vs control (n=3), FDR 0.05'\n 'Whole-genome variant calling on the uploaded FASTQ pair, GRCh38, GATK HaplotypeCaller, germline mode'\n 'Run homogeneous 3D refinement in CryoSPARC on s3://cryo/job042/, C2 symmetry, box size 256'\n 'Fetch GSE183947 from GEO and run the same RNA-seq pipeline'\n\nUse run_workflow instead when the user wants to call a specific workflow by ID with explicit parameter control.",
2000
+ "input_schema": {
2001
+ "type": "object",
2002
+ "properties": {
2003
+ "goal": {
2004
+ "type": "string",
2005
+ "description": "Natural-language scientific goal. Include: analysis type, data location (inline SMILES/sequences, s3:// paths, or GEO/SRA accession numbers), and key parameters (organism, comparisons, thresholds). BioMate infers the rest. Examples: 'Screen these 5 SMILES for hERG IC50 and CYP3A4 inhibition', 'RNA-seq DE on s3://bucket/exp1/ human GRCh38 paired-end treated vs control', 'Fetch GSE183947 and run differential expression'."
2006
+ },
2007
+ "inputs": {
2008
+ "type": "object",
2009
+ "description": "Optional structured inputs passed alongside `goal`. BioMate's inner AI sees these as a JSON block appended to the goal text and merges them with any parameters it extracts from the goal string. Use this to pass: S3 keys from upload_file, sequences, SMILES lists, accession numbers, or explicit parameter overrides. Examples:\n After upload_file: {\"fastq_file\": \"users/42/uploads/uuid-sample.fastq.gz\"} (use the s3_key value returned by upload_file)\n SMILES list: {\"smiles_list\": [\"CC(=O)Oc1ccccc1C(=O)O\"], \"organism\": \"human\"}\n Parameter override: {\"genome\": \"GRCh38\", \"aligner\": \"STAR\", \"fdr\": 0.05}",
2010
+ "additionalProperties": true
2011
+ },
2012
+ "experiment_id": {
2013
+ "type": "string",
2014
+ "description": "Optional experiment to attach this run to (from recall_memory)."
2015
+ },
2016
+ "stream": {
2017
+ "type": "boolean",
2018
+ "description": "Emit progress notifications during execution. Default true. Set false for hosts without MCP notification support \u2014 then poll with get_run.",
2019
+ "default": true
2020
+ }
2021
+ },
2022
+ "required": [
2023
+ "goal"
2024
+ ]
2025
+ }
2026
+ },
2027
+ {
2028
+ "name": "export_report",
2029
+ "description": "Render a publication-ready report for a completed run as PDF or markdown. Includes the methods section, QC audit trail, structured findings, and figures. This is what users need for IND submissions, CRO compliance packages, and publication supplementary materials.",
2030
+ "input_schema": {
2031
+ "type": "object",
2032
+ "properties": {
2033
+ "run_id": {
2034
+ "type": "string",
2035
+ "description": "Completed run ID."
2036
+ },
2037
+ "format": {
2038
+ "type": "string",
2039
+ "enum": [
2040
+ "pdf",
2041
+ "markdown",
2042
+ "docx"
2043
+ ],
2044
+ "default": "pdf"
2045
+ },
2046
+ "sections": {
2047
+ "type": "array",
2048
+ "items": {
2049
+ "type": "string",
2050
+ "enum": [
2051
+ "methods",
2052
+ "qc",
2053
+ "findings",
2054
+ "figures",
2055
+ "appendix"
2056
+ ]
2057
+ },
2058
+ "description": "Which sections to include (default: all)."
2059
+ }
2060
+ },
2061
+ "required": [
2062
+ "run_id"
2063
+ ]
2064
+ }
2065
+ },
2066
+ {
2067
+ "name": "upload_file",
2068
+ "description": "Get a one-shot signed S3 PUT URL so the host can upload a local file directly to BioMate's data plane without proxying bytes through the chat transport. For files >5MB this is mandatory; for small text payloads, inline strings are fine.\n\n**Returns** `{upload_url, s3_key}` \u2014 two fields:\n- `upload_url`: a presigned HTTPS URL. You MUST upload the file bytes to it with an HTTP PUT before calling any other tool: `curl -X PUT -T <local_path> \"<upload_url>\"`\n- `s3_key`: a bare S3 object key (e.g. `users/42/uploads/uuid-sample.fastq.gz`). Pass this value directly into the `inputs` dict of `biomate_session` or as a `params` value in `run_workflow` \u2014 BioMate resolves the bucket internally. Example: `inputs={\"fastq_file\": s3_key}` or `params={\"input\": s3_key}`.\n\n**Full upload workflow:**\n1. `upload_file(filename='sample.fastq.gz', size_bytes=52000000)` \u2192 get `upload_url` + `s3_key`\n2. `curl -X PUT -T sample.fastq.gz \"<upload_url>\"` (or equivalent HTTP PUT \u2014 no auth header needed)\n3. `biomate_session(goal='Run RNA-seq DE on the uploaded FASTQs, human GRCh38', inputs={\"fastq_file\": s3_key})`",
2069
+ "input_schema": {
2070
+ "type": "object",
2071
+ "properties": {
2072
+ "filename": {
2073
+ "type": "string",
2074
+ "description": "Original filename (sets content-type and extension)."
2075
+ },
2076
+ "size_bytes": {
2077
+ "type": "integer",
2078
+ "description": "File size for quota check."
2079
+ },
2080
+ "content_type": {
2081
+ "type": "string",
2082
+ "description": "MIME type (auto-detected from filename if omitted)."
2083
+ }
2084
+ },
2085
+ "required": [
2086
+ "filename"
2087
+ ]
2088
+ }
2089
+ }
2090
+ ],
2091
+ "openai": [
2092
+ {
2093
+ "type": "function",
2094
+ "function": {
2095
+ "name": "biomate_session",
2096
+ "description": "**Primary entry point \u2014 use this for 90% of requests.** Run a complete BioMate scientific session from a natural-language goal. BioMate selects the right workflow from 2,455 indexed pipelines, pre-fills parameters from your goal text, executes on BioMate cloud, handles QC gates with auto-loop remediation, and produces structured findings. While running, the tool streams real-time progress (phase started, step completed, QC gate, auto-loop remediation, finding) back to the host. Returns a final run summary, a deep link to the live results panel, and the report URL. \n\n**How to write the `goal` parameter** \u2014 plain English, one to three sentences:\n\u2022 Include the *what*: analysis type + subject (e.g. 'ADMET screening', 'RNA-seq DE', 'variant calling')\n\u2022 Include *data location*: inline SMILES/sequences, S3 paths, accession numbers, or upload first with upload_file\n\u2022 Include key *parameters* that matter: organism, library type, comparisons, thresholds\n\u2022 You can omit anything BioMate can infer (it will ask if genuinely ambiguous)\n\n**Good examples:**\n 'Screen aspirin (CC(=O)Oc1ccccc1C(=O)O) and caffeine (Cn1cnc2c1c(=O)n(c(=O)n2C)C) for hERG inhibition, CYP3A4, and oral bioavailability'\n 'RNA-seq differential expression on s3://lab-bucket/exp42/fastqs/ \u2014 human GRCh38, dUTP strand-specific, treated (n=3) vs control (n=3), FDR 0.05'\n 'Whole-genome variant calling on the uploaded FASTQ pair, GRCh38, GATK HaplotypeCaller, germline mode'\n 'Run homogeneous 3D refinement in CryoSPARC on s3://cryo/job042/, C2 symmetry, box size 256'\n 'Fetch GSE183947 from GEO and run the same RNA-seq pipeline'\n\nUse run_workflow instead when the user wants to call a specific workflow by ID with explicit parameter control.",
2097
+ "parameters": {
2098
+ "type": "object",
2099
+ "properties": {
2100
+ "goal": {
2101
+ "type": "string",
2102
+ "description": "Natural-language scientific goal. Include: analysis type, data location (inline SMILES/sequences, s3:// paths, or GEO/SRA accession numbers), and key parameters (organism, comparisons, thresholds). BioMate infers the rest. Examples: 'Screen these 5 SMILES for hERG IC50 and CYP3A4 inhibition', 'RNA-seq DE on s3://bucket/exp1/ human GRCh38 paired-end treated vs control', 'Fetch GSE183947 and run differential expression'."
2103
+ },
2104
+ "inputs": {
2105
+ "type": "object",
2106
+ "description": "Optional structured inputs passed alongside `goal`. BioMate's inner AI sees these as a JSON block appended to the goal text and merges them with any parameters it extracts from the goal string. Use this to pass: S3 keys from upload_file, sequences, SMILES lists, accession numbers, or explicit parameter overrides. Examples:\n After upload_file: {\"fastq_file\": \"users/42/uploads/uuid-sample.fastq.gz\"} (use the s3_key value returned by upload_file)\n SMILES list: {\"smiles_list\": [\"CC(=O)Oc1ccccc1C(=O)O\"], \"organism\": \"human\"}\n Parameter override: {\"genome\": \"GRCh38\", \"aligner\": \"STAR\", \"fdr\": 0.05}",
2107
+ "additionalProperties": true
2108
+ },
2109
+ "experiment_id": {
2110
+ "type": "string",
2111
+ "description": "Optional experiment to attach this run to (from recall_memory)."
2112
+ },
2113
+ "stream": {
2114
+ "type": "boolean",
2115
+ "description": "Emit progress notifications during execution. Default true. Set false for hosts without MCP notification support \u2014 then poll with get_run.",
2116
+ "default": true
2117
+ }
2118
+ },
2119
+ "required": [
2120
+ "goal"
2121
+ ]
2122
+ }
2123
+ }
2124
+ },
2125
+ {
2126
+ "type": "function",
2127
+ "function": {
2128
+ "name": "export_report",
2129
+ "description": "Render a publication-ready report for a completed run as PDF or markdown. Includes the methods section, QC audit trail, structured findings, and figures. This is what users need for IND submissions, CRO compliance packages, and publication supplementary materials.",
2130
+ "parameters": {
2131
+ "type": "object",
2132
+ "properties": {
2133
+ "run_id": {
2134
+ "type": "string",
2135
+ "description": "Completed run ID."
2136
+ },
2137
+ "format": {
2138
+ "type": "string",
2139
+ "enum": [
2140
+ "pdf",
2141
+ "markdown",
2142
+ "docx"
2143
+ ],
2144
+ "default": "pdf"
2145
+ },
2146
+ "sections": {
2147
+ "type": "array",
2148
+ "items": {
2149
+ "type": "string",
2150
+ "enum": [
2151
+ "methods",
2152
+ "qc",
2153
+ "findings",
2154
+ "figures",
2155
+ "appendix"
2156
+ ]
2157
+ },
2158
+ "description": "Which sections to include (default: all)."
2159
+ }
2160
+ },
2161
+ "required": [
2162
+ "run_id"
2163
+ ]
2164
+ }
2165
+ }
2166
+ },
2167
+ {
2168
+ "type": "function",
2169
+ "function": {
2170
+ "name": "upload_file",
2171
+ "description": "Get a one-shot signed S3 PUT URL so the host can upload a local file directly to BioMate's data plane without proxying bytes through the chat transport. For files >5MB this is mandatory; for small text payloads, inline strings are fine.\n\n**Returns** `{upload_url, s3_key}` \u2014 two fields:\n- `upload_url`: a presigned HTTPS URL. You MUST upload the file bytes to it with an HTTP PUT before calling any other tool: `curl -X PUT -T <local_path> \"<upload_url>\"`\n- `s3_key`: a bare S3 object key (e.g. `users/42/uploads/uuid-sample.fastq.gz`). Pass this value directly into the `inputs` dict of `biomate_session` or as a `params` value in `run_workflow` \u2014 BioMate resolves the bucket internally. Example: `inputs={\"fastq_file\": s3_key}` or `params={\"input\": s3_key}`.\n\n**Full upload workflow:**\n1. `upload_file(filename='sample.fastq.gz', size_bytes=52000000)` \u2192 get `upload_url` + `s3_key`\n2. `curl -X PUT -T sample.fastq.gz \"<upload_url>\"` (or equivalent HTTP PUT \u2014 no auth header needed)\n3. `biomate_session(goal='Run RNA-seq DE on the uploaded FASTQs, human GRCh38', inputs={\"fastq_file\": s3_key})`",
2172
+ "parameters": {
2173
+ "type": "object",
2174
+ "properties": {
2175
+ "filename": {
2176
+ "type": "string",
2177
+ "description": "Original filename (sets content-type and extension)."
2178
+ },
2179
+ "size_bytes": {
2180
+ "type": "integer",
2181
+ "description": "File size for quota check."
2182
+ },
2183
+ "content_type": {
2184
+ "type": "string",
2185
+ "description": "MIME type (auto-detected from filename if omitted)."
2186
+ }
2187
+ },
2188
+ "required": [
2189
+ "filename"
2190
+ ]
2191
+ }
2192
+ }
2193
+ }
2194
+ ],
2195
+ "tool_names": [
2196
+ "biomate_session",
2197
+ "export_report",
2198
+ "upload_file"
2199
+ ]
2200
+ },
2201
+ "backend_routes": [
2202
+ {
2203
+ "name": "biomate_session",
2204
+ "method": "POST",
2205
+ "path": "/api/open-claw/stream",
2206
+ "streaming": true
2207
+ },
2208
+ {
2209
+ "name": "search_workflow",
2210
+ "method": "POST",
2211
+ "path": "/api/workflows/search",
2212
+ "streaming": false
2213
+ },
2214
+ {
2215
+ "name": "get_workflow_spec",
2216
+ "method": "GET",
2217
+ "path": "/api/workflows/spec",
2218
+ "streaming": false
2219
+ },
2220
+ {
2221
+ "name": "run_workflow",
2222
+ "method": "POST",
2223
+ "path": "/api/workflows/execute",
2224
+ "streaming": true
2225
+ },
2226
+ {
2227
+ "name": "watch_run",
2228
+ "method": "GET",
2229
+ "path": "/api/workflows/runs/{run_id}",
2230
+ "streaming": true
2231
+ },
2232
+ {
2233
+ "name": "get_run",
2234
+ "method": "GET",
2235
+ "path": "/api/workflows/runs/{run_id}",
2236
+ "streaming": false
2237
+ },
2238
+ {
2239
+ "name": "cancel_run",
2240
+ "method": "POST",
2241
+ "path": "/api/workflows/runs/{run_id}/cancel",
2242
+ "streaming": false
2243
+ },
2244
+ {
2245
+ "name": "list_runs",
2246
+ "method": "GET",
2247
+ "path": "/api/workflows/runs",
2248
+ "streaming": false
2249
+ },
2250
+ {
2251
+ "name": "preview_file",
2252
+ "method": "POST",
2253
+ "path": "/api/files/preview",
2254
+ "streaming": false
2255
+ },
2256
+ {
2257
+ "name": "export_report",
2258
+ "method": "POST",
2259
+ "path": "/api/workflows/runs/{run_id}/findings/report",
2260
+ "streaming": false
2261
+ },
2262
+ {
2263
+ "name": "analyze_results",
2264
+ "method": "POST",
2265
+ "path": "/api/workflows/runs/{run_id}/ai/analyze",
2266
+ "streaming": false
2267
+ },
2268
+ {
2269
+ "name": "explain_error",
2270
+ "method": "POST",
2271
+ "path": "/api/workflows/explain_error",
2272
+ "streaming": false
2273
+ },
2274
+ {
2275
+ "name": "query_database",
2276
+ "method": "POST",
2277
+ "path": "/api/databases/query",
2278
+ "streaming": false
2279
+ },
2280
+ {
2281
+ "name": "resolve_accession",
2282
+ "method": "POST",
2283
+ "path": "/api/accession/resolve",
2284
+ "streaming": false
2285
+ },
2286
+ {
2287
+ "name": "browse_data",
2288
+ "method": "POST",
2289
+ "path": "/api/data/browse",
2290
+ "streaming": false
2291
+ },
2292
+ {
2293
+ "name": "fetch_public_data",
2294
+ "method": "POST",
2295
+ "path": "/api/data/fetch",
2296
+ "streaming": false
2297
+ },
2298
+ {
2299
+ "name": "recall_memory",
2300
+ "method": "POST",
2301
+ "path": "/api/memory/relevant",
2302
+ "streaming": false
2303
+ },
2304
+ {
2305
+ "name": "search_literature",
2306
+ "method": "POST",
2307
+ "path": "/api/literature/search",
2308
+ "streaming": false
2309
+ },
2310
+ {
2311
+ "name": "upload_file",
2312
+ "method": "POST",
2313
+ "path": "/api/uploads/signed_url",
2314
+ "streaming": false
2315
+ },
2316
+ {
2317
+ "name": "list_instruments",
2318
+ "method": "GET",
2319
+ "path": "local:opentrons/discover",
2320
+ "streaming": false
2321
+ },
2322
+ {
2323
+ "name": "get_instrument_status",
2324
+ "method": "GET",
2325
+ "path": "local:opentrons/run_status",
2326
+ "streaming": false
2327
+ },
2328
+ {
2329
+ "name": "run_instrument_protocol",
2330
+ "method": "POST",
2331
+ "path": "local:opentrons/run",
2332
+ "streaming": false
2333
+ }
2334
+ ]
2335
+ }