smartapi-mcp 0.3.2__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/PKG-INFO +51 -14
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/README.md +47 -10
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/pyproject.toml +18 -10
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/smartapi_mcp/__init__.py +1 -1
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/smartapi_mcp/biothings.py +198 -11
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/smartapi_mcp/cli.py +45 -4
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/smartapi_mcp/config.py +18 -0
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/smartapi_mcp/server.py +240 -22
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/smartapi_mcp/smartapi.py +21 -20
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/LICENSE +0 -0
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/MANIFEST.in +0 -0
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/requirements-dev.txt +0 -0
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/requirements.txt +0 -0
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/setup.cfg +0 -0
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/setup.py +0 -0
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/smartapi_mcp/__main__.py +0 -0
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/smartapi_mcp/py.typed +0 -0
- {smartapi_mcp-0.3.2 → smartapi_mcp-0.4.0}/smartapi_mcp.egg-info/SOURCES.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: smartapi-mcp
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: Create MCP servers for one or multiple APIs registered in SmartAPI registry
|
|
5
5
|
Author-email: BioThings Team <help@biothings.io>
|
|
6
6
|
Maintainer-email: BioThings Team <help@biothings.io>
|
|
@@ -25,13 +25,13 @@ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
|
25
25
|
Requires-Python: >=3.10
|
|
26
26
|
Description-Content-Type: text/markdown
|
|
27
27
|
License-File: LICENSE
|
|
28
|
-
Requires-Dist: awslabs_openapi_mcp_server<1
|
|
29
|
-
Requires-Dist: fastmcp<3
|
|
28
|
+
Requires-Dist: awslabs_openapi_mcp_server<2,>=1.1.5
|
|
29
|
+
Requires-Dist: fastmcp<4,>=3.3.1
|
|
30
30
|
Provides-Extra: dev
|
|
31
31
|
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
32
32
|
Requires-Dist: pytest-asyncio>=0.21.0; extra == "dev"
|
|
33
33
|
Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
|
|
34
|
-
Requires-Dist: ruff
|
|
34
|
+
Requires-Dist: ruff<0.17,>=0.14; extra == "dev"
|
|
35
35
|
Requires-Dist: build>=0.8.0; extra == "dev"
|
|
36
36
|
Requires-Dist: twine>=4.0.0; extra == "dev"
|
|
37
37
|
Provides-Extra: test
|
|
@@ -62,8 +62,9 @@ Built on top of the [AWS Labs OpenAPI MCP Server](https://github.com/awslabs/ope
|
|
|
62
62
|
|
|
63
63
|
- Python 3.10 or higher
|
|
64
64
|
- Network access to SmartAPI registry (https://smart-api.info)
|
|
65
|
-
- Dependencies: `awslabs_openapi_mcp_server>=
|
|
66
|
-
(
|
|
65
|
+
- Dependencies: `awslabs_openapi_mcp_server>=1.1.5,<2` and `fastmcp>=3.3.1,<4`
|
|
66
|
+
(fastmcp 4.x is not yet supported — it moves to the MCP 2.x SDK and `httpx2`,
|
|
67
|
+
and no awslabs release targets it yet; see the dependency notes in
|
|
67
68
|
`pyproject.toml`)
|
|
68
69
|
|
|
69
70
|
## Features
|
|
@@ -247,6 +248,41 @@ inspect each BioThings spec at startup and serve any API that has extra
|
|
|
247
248
|
endpoints with faithful per-API tools instead — at the cost of slower startup
|
|
248
249
|
(it downloads the specs upfront).
|
|
249
250
|
|
|
251
|
+
#### Tool search (`--tool-search`)
|
|
252
|
+
|
|
253
|
+
The facade solves the tool explosion for BioThings APIs. `--tool-search` solves
|
|
254
|
+
it for everything else, including per-API tools in a hybrid server: instead of
|
|
255
|
+
listing every tool, the server lists two synthetic tools — `search_tools` and
|
|
256
|
+
`call_tool` — and the model discovers what it needs on demand.
|
|
257
|
+
|
|
258
|
+
```bash
|
|
259
|
+
smartapi-mcp --api_set biothings_all --tool-search bm25
|
|
260
|
+
```
|
|
261
|
+
|
|
262
|
+
Every tool stays *callable* through `call_tool`; only the listing changes. Facade
|
|
263
|
+
tools (`biothings_query`, `biothings_get`, …) stay listed, so the common path
|
|
264
|
+
remains directly callable and only the per-API long tail is collapsed:
|
|
265
|
+
|
|
266
|
+
```
|
|
267
|
+
Tool search (bm25) enabled: 13 tools collapsed to 7 listed
|
|
268
|
+
(5 pinned + search_tools/call_tool); max_results=5.
|
|
269
|
+
All 13 tools stay callable via call_tool.
|
|
270
|
+
```
|
|
271
|
+
|
|
272
|
+
| Mode | Behaviour |
|
|
273
|
+
| --- | --- |
|
|
274
|
+
| `off` (default) | List every tool |
|
|
275
|
+
| `bm25` | Rank matches by keyword relevance |
|
|
276
|
+
| `regex` | Match tool names/descriptions by pattern |
|
|
277
|
+
|
|
278
|
+
`--tool-search-max-results` (default 5) caps the hits per search. Both options
|
|
279
|
+
have environment equivalents: `SMARTAPI_TOOL_SEARCH` and
|
|
280
|
+
`TOOL_SEARCH_MAX_RESULTS`.
|
|
281
|
+
|
|
282
|
+
Note that discovery-on-demand costs the model an extra round trip per unfamiliar
|
|
283
|
+
tool, so leave it `off` for small sets where the full list already fits
|
|
284
|
+
comfortably.
|
|
285
|
+
|
|
250
286
|
#### Development/Testing Setup
|
|
251
287
|
|
|
252
288
|
```json
|
|
@@ -451,9 +487,10 @@ from smartapi_mcp import (
|
|
|
451
487
|
load_api_spec,
|
|
452
488
|
get_mcp_server,
|
|
453
489
|
get_merged_mcp_server,
|
|
454
|
-
PREDEFINED_API_SETS
|
|
490
|
+
PREDEFINED_API_SETS,
|
|
455
491
|
)
|
|
456
492
|
|
|
493
|
+
|
|
457
494
|
async def main():
|
|
458
495
|
# Get SmartAPI IDs using a query
|
|
459
496
|
smartapi_ids = await get_smartapi_ids("tags.name=biothings")
|
|
@@ -463,16 +500,15 @@ async def main():
|
|
|
463
500
|
api_spec = load_api_spec("59dce17363dce279d389100834e43648") # MyGene.info
|
|
464
501
|
print(f"Loaded API: {api_spec.get('info', {}).get('title', 'Unknown')}")
|
|
465
502
|
|
|
466
|
-
# Create MCP server for a single API
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
)
|
|
503
|
+
# Create MCP server for a single API. The server is named after the
|
|
504
|
+
# spec's info.title; pass server_name to the merged/`build_server_for_set`
|
|
505
|
+
# entry points below if you want to choose the name yourself.
|
|
506
|
+
server = await get_mcp_server("59dce17363dce279d389100834e43648")
|
|
471
507
|
|
|
472
508
|
# Create merged MCP server for multiple APIs (recommended approach)
|
|
473
509
|
merged_server = await get_merged_mcp_server(
|
|
474
510
|
api_set="biothings_core", # Use predefined set
|
|
475
|
-
server_name="BioThings Core MCP Server"
|
|
511
|
+
server_name="BioThings Core MCP Server",
|
|
476
512
|
)
|
|
477
513
|
|
|
478
514
|
# Or with specific SmartAPI IDs
|
|
@@ -481,7 +517,7 @@ async def main():
|
|
|
481
517
|
"59dce17363dce279d389100834e43648", # MyGene.info
|
|
482
518
|
"09c8782d9f4027712e65b95424adba79", # MyVariant.info
|
|
483
519
|
],
|
|
484
|
-
server_name="Custom MCP Server"
|
|
520
|
+
server_name="Custom MCP Server",
|
|
485
521
|
)
|
|
486
522
|
|
|
487
523
|
# Show available predefined API sets
|
|
@@ -493,6 +529,7 @@ async def main():
|
|
|
493
529
|
# Or run with HTTP transport
|
|
494
530
|
# merged_server.run(transport="http", host="localhost", port=8000)
|
|
495
531
|
|
|
532
|
+
|
|
496
533
|
if __name__ == "__main__":
|
|
497
534
|
asyncio.run(main())
|
|
498
535
|
```
|
|
@@ -17,8 +17,9 @@ Built on top of the [AWS Labs OpenAPI MCP Server](https://github.com/awslabs/ope
|
|
|
17
17
|
|
|
18
18
|
- Python 3.10 or higher
|
|
19
19
|
- Network access to SmartAPI registry (https://smart-api.info)
|
|
20
|
-
- Dependencies: `awslabs_openapi_mcp_server>=
|
|
21
|
-
(
|
|
20
|
+
- Dependencies: `awslabs_openapi_mcp_server>=1.1.5,<2` and `fastmcp>=3.3.1,<4`
|
|
21
|
+
(fastmcp 4.x is not yet supported — it moves to the MCP 2.x SDK and `httpx2`,
|
|
22
|
+
and no awslabs release targets it yet; see the dependency notes in
|
|
22
23
|
`pyproject.toml`)
|
|
23
24
|
|
|
24
25
|
## Features
|
|
@@ -202,6 +203,41 @@ inspect each BioThings spec at startup and serve any API that has extra
|
|
|
202
203
|
endpoints with faithful per-API tools instead — at the cost of slower startup
|
|
203
204
|
(it downloads the specs upfront).
|
|
204
205
|
|
|
206
|
+
#### Tool search (`--tool-search`)
|
|
207
|
+
|
|
208
|
+
The facade solves the tool explosion for BioThings APIs. `--tool-search` solves
|
|
209
|
+
it for everything else, including per-API tools in a hybrid server: instead of
|
|
210
|
+
listing every tool, the server lists two synthetic tools — `search_tools` and
|
|
211
|
+
`call_tool` — and the model discovers what it needs on demand.
|
|
212
|
+
|
|
213
|
+
```bash
|
|
214
|
+
smartapi-mcp --api_set biothings_all --tool-search bm25
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
Every tool stays *callable* through `call_tool`; only the listing changes. Facade
|
|
218
|
+
tools (`biothings_query`, `biothings_get`, …) stay listed, so the common path
|
|
219
|
+
remains directly callable and only the per-API long tail is collapsed:
|
|
220
|
+
|
|
221
|
+
```
|
|
222
|
+
Tool search (bm25) enabled: 13 tools collapsed to 7 listed
|
|
223
|
+
(5 pinned + search_tools/call_tool); max_results=5.
|
|
224
|
+
All 13 tools stay callable via call_tool.
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
| Mode | Behaviour |
|
|
228
|
+
| --- | --- |
|
|
229
|
+
| `off` (default) | List every tool |
|
|
230
|
+
| `bm25` | Rank matches by keyword relevance |
|
|
231
|
+
| `regex` | Match tool names/descriptions by pattern |
|
|
232
|
+
|
|
233
|
+
`--tool-search-max-results` (default 5) caps the hits per search. Both options
|
|
234
|
+
have environment equivalents: `SMARTAPI_TOOL_SEARCH` and
|
|
235
|
+
`TOOL_SEARCH_MAX_RESULTS`.
|
|
236
|
+
|
|
237
|
+
Note that discovery-on-demand costs the model an extra round trip per unfamiliar
|
|
238
|
+
tool, so leave it `off` for small sets where the full list already fits
|
|
239
|
+
comfortably.
|
|
240
|
+
|
|
205
241
|
#### Development/Testing Setup
|
|
206
242
|
|
|
207
243
|
```json
|
|
@@ -406,9 +442,10 @@ from smartapi_mcp import (
|
|
|
406
442
|
load_api_spec,
|
|
407
443
|
get_mcp_server,
|
|
408
444
|
get_merged_mcp_server,
|
|
409
|
-
PREDEFINED_API_SETS
|
|
445
|
+
PREDEFINED_API_SETS,
|
|
410
446
|
)
|
|
411
447
|
|
|
448
|
+
|
|
412
449
|
async def main():
|
|
413
450
|
# Get SmartAPI IDs using a query
|
|
414
451
|
smartapi_ids = await get_smartapi_ids("tags.name=biothings")
|
|
@@ -418,16 +455,15 @@ async def main():
|
|
|
418
455
|
api_spec = load_api_spec("59dce17363dce279d389100834e43648") # MyGene.info
|
|
419
456
|
print(f"Loaded API: {api_spec.get('info', {}).get('title', 'Unknown')}")
|
|
420
457
|
|
|
421
|
-
# Create MCP server for a single API
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
)
|
|
458
|
+
# Create MCP server for a single API. The server is named after the
|
|
459
|
+
# spec's info.title; pass server_name to the merged/`build_server_for_set`
|
|
460
|
+
# entry points below if you want to choose the name yourself.
|
|
461
|
+
server = await get_mcp_server("59dce17363dce279d389100834e43648")
|
|
426
462
|
|
|
427
463
|
# Create merged MCP server for multiple APIs (recommended approach)
|
|
428
464
|
merged_server = await get_merged_mcp_server(
|
|
429
465
|
api_set="biothings_core", # Use predefined set
|
|
430
|
-
server_name="BioThings Core MCP Server"
|
|
466
|
+
server_name="BioThings Core MCP Server",
|
|
431
467
|
)
|
|
432
468
|
|
|
433
469
|
# Or with specific SmartAPI IDs
|
|
@@ -436,7 +472,7 @@ async def main():
|
|
|
436
472
|
"59dce17363dce279d389100834e43648", # MyGene.info
|
|
437
473
|
"09c8782d9f4027712e65b95424adba79", # MyVariant.info
|
|
438
474
|
],
|
|
439
|
-
server_name="Custom MCP Server"
|
|
475
|
+
server_name="Custom MCP Server",
|
|
440
476
|
)
|
|
441
477
|
|
|
442
478
|
# Show available predefined API sets
|
|
@@ -448,6 +484,7 @@ async def main():
|
|
|
448
484
|
# Or run with HTTP transport
|
|
449
485
|
# merged_server.run(transport="http", host="localhost", port=8000)
|
|
450
486
|
|
|
487
|
+
|
|
451
488
|
if __name__ == "__main__":
|
|
452
489
|
asyncio.run(main())
|
|
453
490
|
```
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "smartapi-mcp"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.4.0"
|
|
8
8
|
description = "Create MCP servers for one or multiple APIs registered in SmartAPI registry"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "Apache-2.0"
|
|
@@ -32,12 +32,14 @@ classifiers = [
|
|
|
32
32
|
keywords = ["mcp", "smartapi", "api", "server", "bioinformatics"]
|
|
33
33
|
requires-python = ">=3.10"
|
|
34
34
|
dependencies = [
|
|
35
|
-
# awslabs 1.x
|
|
36
|
-
#
|
|
37
|
-
#
|
|
38
|
-
#
|
|
39
|
-
|
|
40
|
-
|
|
35
|
+
# awslabs 1.x is the release line built against fastmcp 3.x; both are pinned
|
|
36
|
+
# together because the two migrated in lockstep (fastmcp 3 replaced
|
|
37
|
+
# FastMCP.get_tools()/get_prompts() with list_tools()/list_prompts(), and
|
|
38
|
+
# awslabs 1.x switched to the new names).
|
|
39
|
+
# Upper bounds: fastmcp 4.x moves to the MCP 2.x SDK and httpx2, and no
|
|
40
|
+
# awslabs release supports it yet (awslabs 1.1.5 itself pins fastmcp<4).
|
|
41
|
+
"awslabs_openapi_mcp_server>=1.1.5,<2",
|
|
42
|
+
"fastmcp>=3.3.1,<4",
|
|
41
43
|
]
|
|
42
44
|
|
|
43
45
|
[project.optional-dependencies]
|
|
@@ -45,7 +47,10 @@ dev = [
|
|
|
45
47
|
"pytest>=7.0.0",
|
|
46
48
|
"pytest-asyncio>=0.21.0",
|
|
47
49
|
"pytest-cov>=4.0.0",
|
|
48
|
-
|
|
50
|
+
# Upper bound deliberately: an unpinned linter makes CI results drift
|
|
51
|
+
# with upstream releases, failing on unchanged code. Raise it
|
|
52
|
+
# intentionally, with a lint/format pass, rather than by surprise.
|
|
53
|
+
"ruff>=0.14,<0.17",
|
|
49
54
|
"build>=0.8.0",
|
|
50
55
|
"twine>=4.0.0",
|
|
51
56
|
]
|
|
@@ -199,8 +204,11 @@ ignore = [
|
|
|
199
204
|
"B027",
|
|
200
205
|
# Allow boolean positional values in function calls
|
|
201
206
|
"FBT003",
|
|
202
|
-
# Ignore complexity
|
|
203
|
-
|
|
207
|
+
# Ignore complexity. PLR0917 (too-many-positional-arguments) is the same
|
|
208
|
+
# complaint as PLR0913 for the build entry points, which take a set of
|
|
209
|
+
# mutually exclusive API selectors; making them keyword-only would be a
|
|
210
|
+
# breaking change to the public API.
|
|
211
|
+
"C901", "PLR0911", "PLR0912", "PLR0913", "PLR0915", "PLR0917",
|
|
204
212
|
# Allow print statements (useful for CLI tools)
|
|
205
213
|
"T201",
|
|
206
214
|
# Allow assert statements
|
|
@@ -12,6 +12,7 @@ the set, so the server works on every MCP client without depending on runtime
|
|
|
12
12
|
"""
|
|
13
13
|
|
|
14
14
|
import asyncio
|
|
15
|
+
import math
|
|
15
16
|
import re
|
|
16
17
|
from dataclasses import dataclass, field
|
|
17
18
|
|
|
@@ -21,6 +22,7 @@ from fastmcp import FastMCP
|
|
|
21
22
|
from fastmcp.tools import Tool
|
|
22
23
|
|
|
23
24
|
from .smartapi import (
|
|
25
|
+
CORE_BIOTHINGS_API_IDS,
|
|
24
26
|
HTTP_TIMEOUT,
|
|
25
27
|
get_base_server_url,
|
|
26
28
|
get_smartapi_registry,
|
|
@@ -37,6 +39,20 @@ _MAX_DESC_LEN = 200
|
|
|
37
39
|
# HTTP methods recognized when enumerating spec operations.
|
|
38
40
|
_HTTP_METHODS = {"get", "post", "put", "delete", "patch", "head", "options"}
|
|
39
41
|
|
|
42
|
+
# Tags that disqualify an API from the BioThings *annotation-API family* even
|
|
43
|
+
# though it carries the "biothings" tag. TRAPI services (BioThings Explorer,
|
|
44
|
+
# Service Provider) are built by the same team and tagged accordingly, but they
|
|
45
|
+
# speak the Translator Reasoner API -- a query-graph protocol -- not the
|
|
46
|
+
# BioThings annotation interface, so none of the generic facade tools apply to
|
|
47
|
+
# them. They are served with faithful per-API tools instead.
|
|
48
|
+
#
|
|
49
|
+
# This is not a cosmetic distinction: the facade infers an entity type from the
|
|
50
|
+
# first ``/{type}/{id}``-shaped path, and BTE's ``GET /asyncquery_status/{id}``
|
|
51
|
+
# matches that shape. Without this exclusion, ``biothings_get`` would request
|
|
52
|
+
# ``/asyncquery_status/<id>`` and return a job status as though it were an
|
|
53
|
+
# annotation record -- a wrong answer with no error.
|
|
54
|
+
NON_FAMILY_TAGS = frozenset({"trapi"})
|
|
55
|
+
|
|
40
56
|
|
|
41
57
|
@dataclass
|
|
42
58
|
class BioThingsAPIEntry:
|
|
@@ -96,11 +112,21 @@ async def build_registry(
|
|
|
96
112
|
return build_registry_from_entries(entries)
|
|
97
113
|
|
|
98
114
|
|
|
115
|
+
def is_biothings_family(entry: BioThingsAPIEntry) -> bool:
|
|
116
|
+
"""Whether ``entry`` is a BioThings annotation API the facade can serve.
|
|
117
|
+
|
|
118
|
+
Requires the ``biothings`` tag and the absence of any
|
|
119
|
+
:data:`NON_FAMILY_TAGS`.
|
|
120
|
+
"""
|
|
121
|
+
tags = {str(tag).strip().lower() for tag in entry.tags}
|
|
122
|
+
return "biothings" in tags and not (tags & NON_FAMILY_TAGS)
|
|
123
|
+
|
|
124
|
+
|
|
99
125
|
def is_biothings_registry(registry: dict[str, BioThingsAPIEntry]) -> bool:
|
|
100
|
-
"""True only if every API in the registry is
|
|
126
|
+
"""True only if every API in the registry is a facade-servable BioThings API."""
|
|
101
127
|
if not registry:
|
|
102
128
|
return False
|
|
103
|
-
return all(
|
|
129
|
+
return all(is_biothings_family(entry) for entry in registry.values())
|
|
104
130
|
|
|
105
131
|
|
|
106
132
|
def _resolve_endpoints(entry: BioThingsAPIEntry) -> None:
|
|
@@ -225,12 +251,146 @@ async def partition_biothings(
|
|
|
225
251
|
return facade_entries, extra_ids
|
|
226
252
|
|
|
227
253
|
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
254
|
+
# Words too common to discriminate between APIs. Without this, a natural-language
|
|
255
|
+
# intent like "get a gene annotation by its Entrez gene id" is dominated by
|
|
256
|
+
# "get"/"a"/"by"/"its"/"id" rather than by "gene" and "entrez".
|
|
257
|
+
_STOPWORDS = frozenset(
|
|
258
|
+
{
|
|
259
|
+
"a",
|
|
260
|
+
"an",
|
|
261
|
+
"and",
|
|
262
|
+
"are",
|
|
263
|
+
"as",
|
|
264
|
+
"at",
|
|
265
|
+
"be",
|
|
266
|
+
"by",
|
|
267
|
+
"for",
|
|
268
|
+
"from",
|
|
269
|
+
"get",
|
|
270
|
+
"how",
|
|
271
|
+
"in",
|
|
272
|
+
"into",
|
|
273
|
+
"is",
|
|
274
|
+
"it",
|
|
275
|
+
"its",
|
|
276
|
+
"of",
|
|
277
|
+
"on",
|
|
278
|
+
"or",
|
|
279
|
+
"that",
|
|
280
|
+
"the",
|
|
281
|
+
"their",
|
|
282
|
+
"them",
|
|
283
|
+
"there",
|
|
284
|
+
"these",
|
|
285
|
+
"this",
|
|
286
|
+
"to",
|
|
287
|
+
"what",
|
|
288
|
+
"when",
|
|
289
|
+
"where",
|
|
290
|
+
"which",
|
|
291
|
+
"who",
|
|
292
|
+
"with",
|
|
293
|
+
"api",
|
|
294
|
+
"apis",
|
|
295
|
+
"data",
|
|
296
|
+
"database",
|
|
297
|
+
"dataset",
|
|
298
|
+
"info",
|
|
299
|
+
"information",
|
|
300
|
+
"record",
|
|
301
|
+
"records",
|
|
302
|
+
"service",
|
|
303
|
+
"services",
|
|
304
|
+
}
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def _tokenize(text: str) -> list[str]:
|
|
309
|
+
"""Lower-case word tokens. Word-based, not substring-based, on purpose.
|
|
310
|
+
|
|
311
|
+
The previous scorer used ``str.count`` on the raw text, so a query term
|
|
312
|
+
like "id" also matched inside "identifier", "candidate" and "provide".
|
|
313
|
+
"""
|
|
314
|
+
return re.findall(r"[a-z0-9]+", text.lower())
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
# How much more a term in the API's name/title counts than one buried in its
|
|
318
|
+
# description. An API *named* for a concept is a far stronger answer to a query
|
|
319
|
+
# about that concept than one merely mentioning it in prose; without this,
|
|
320
|
+
# scores tie constantly and the tie-break (alphabetical) decides, which
|
|
321
|
+
# systematically favours the "biothings_*"-prefixed names over MyGene/MyChem.
|
|
322
|
+
_NAME_FIELD_WEIGHT = 3.0
|
|
323
|
+
|
|
324
|
+
# Score multiplier for the core BioThings APIs (:data:`CORE_BIOTHINGS_API_IDS`).
|
|
325
|
+
#
|
|
326
|
+
# These are the canonical broad-coverage services, and they are the *worst*
|
|
327
|
+
# served by pure lexical scoring: being general means their descriptions carry
|
|
328
|
+
# the least distinctive vocabulary, while single-source satellite APIs read as
|
|
329
|
+
# highly specific. So a lexical ranker systematically under-ranks exactly the
|
|
330
|
+
# APIs a user most often wants, and needs a prior to correct for it.
|
|
331
|
+
#
|
|
332
|
+
# Measured on 20 BioThings intents. With the registry descriptions as they were
|
|
333
|
+
# before the core-API enrichment, no boost gave recall@5 16/20 (MRR 0.71) and
|
|
334
|
+
# only 3 of the 6 core-API intents were answered; 1.2 gives 19/20 (MRR 0.82).
|
|
335
|
+
# With the enriched descriptions live, no boost already gives 19/20 (MRR 0.92)
|
|
336
|
+
# and 1.2 gives 20/20 (MRR 0.90). Larger values buy nothing on recall and cost
|
|
337
|
+
# ranking quality once the metadata is good -- at 3.0 the post-enrichment MRR
|
|
338
|
+
# falls to 0.81 -- so this is deliberately the smallest value that captures the
|
|
339
|
+
# benefit, and it stays close to neutral as the metadata improves.
|
|
340
|
+
CORE_API_BOOST = 1.2
|
|
341
|
+
|
|
342
|
+
_CORE_API_IDS = frozenset(CORE_BIOTHINGS_API_IDS)
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def _entry_terms(entry: BioThingsAPIEntry) -> tuple[set[str], set[str]]:
|
|
346
|
+
"""Return ``(name_terms, all_terms)`` for one API.
|
|
347
|
+
|
|
348
|
+
``name_terms`` covers the short name, title and tags -- the curated labels;
|
|
349
|
+
``all_terms`` adds the free-text description.
|
|
350
|
+
"""
|
|
351
|
+
name_terms = set(
|
|
352
|
+
_tokenize(" ".join([entry.name, entry.title, " ".join(map(str, entry.tags))]))
|
|
353
|
+
)
|
|
354
|
+
return name_terms, name_terms | set(_tokenize(entry.description))
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def _score_entry(
|
|
358
|
+
terms: tuple[set[str], set[str]],
|
|
359
|
+
query_terms: list[str],
|
|
360
|
+
idf: dict[str, float],
|
|
361
|
+
) -> float:
|
|
362
|
+
"""Score one API against a query. Higher is more relevant.
|
|
363
|
+
|
|
364
|
+
``terms`` is the ``(name_terms, all_terms)`` pair from :func:`_entry_terms`.
|
|
365
|
+
|
|
366
|
+
Three deliberate properties:
|
|
367
|
+
|
|
368
|
+
* **Binary term frequency.** A term counts once however often it appears, so
|
|
369
|
+
an API with a long, repetitive description no longer outranks a precisely
|
|
370
|
+
matching one. This was the concrete defect in the previous scorer:
|
|
371
|
+
searching "get a gene annotation by its Entrez gene id" ranked MyGeneSet
|
|
372
|
+
above MyGene, because summed substring counts reward verbosity.
|
|
373
|
+
* **IDF weighting.** A term shared by most APIs ("gene", "translator")
|
|
374
|
+
contributes almost nothing, while a rare one ("entrez", "ngd", "taxonomy")
|
|
375
|
+
dominates -- which is what makes an intent select the API it names.
|
|
376
|
+
* **Field weighting.** A hit in the name/title/tags counts
|
|
377
|
+
:data:`_NAME_FIELD_WEIGHT` times one that is only in the description. An
|
|
378
|
+
API *named* for a concept answers a query about it far better than one
|
|
379
|
+
merely mentioning it in prose, and without this, scores tie constantly and
|
|
380
|
+
the alphabetical tie-break decides -- which systematically favoured the
|
|
381
|
+
``biothings_*``-prefixed names over MyGene/MyChem/MyVariant.
|
|
382
|
+
"""
|
|
383
|
+
name_terms, all_terms = terms
|
|
384
|
+
score = 0.0
|
|
385
|
+
for term in query_terms:
|
|
386
|
+
weight = idf.get(term, 0.0)
|
|
387
|
+
if not weight:
|
|
388
|
+
continue
|
|
389
|
+
if term in name_terms:
|
|
390
|
+
score += weight * _NAME_FIELD_WEIGHT
|
|
391
|
+
elif term in all_terms:
|
|
392
|
+
score += weight
|
|
393
|
+
return score
|
|
234
394
|
|
|
235
395
|
|
|
236
396
|
def rank_apis(
|
|
@@ -258,11 +418,38 @@ def rank_apis(
|
|
|
258
418
|
if not keyword or not keyword.strip():
|
|
259
419
|
return [_as_dict(registry[name]) for name in sorted(registry)]
|
|
260
420
|
|
|
261
|
-
|
|
262
|
-
|
|
421
|
+
query_terms = [t for t in _tokenize(keyword) if t not in _STOPWORDS]
|
|
422
|
+
if not query_terms:
|
|
423
|
+
# Nothing discriminating left (e.g. "what data is there?"): fall back to
|
|
424
|
+
# the full catalog rather than returning an arbitrary subset.
|
|
425
|
+
return [_as_dict(registry[name]) for name in sorted(registry)]
|
|
426
|
+
|
|
427
|
+
terms_by_name = {name: _entry_terms(entry) for name, entry in registry.items()}
|
|
428
|
+
total = len(registry)
|
|
429
|
+
idf = {}
|
|
430
|
+
for term in set(query_terms):
|
|
431
|
+
seen_in = sum(1 for _, all_terms in terms_by_name.values() if term in all_terms)
|
|
432
|
+
# Robertson/Sparck-Jones IDF. Chosen over the plain log(N/n) form
|
|
433
|
+
# because it stays strictly positive even for a term present in every
|
|
434
|
+
# API: with a small registry (say two APIs that both mention "gene"),
|
|
435
|
+
# a zero weight would drop every candidate and the search would answer
|
|
436
|
+
# "nothing found" for a query that in fact matches everything.
|
|
437
|
+
idf[term] = math.log(1 + (total - seen_in + 0.5) / (seen_in + 0.5))
|
|
438
|
+
|
|
439
|
+
scored = []
|
|
440
|
+
for name, entry in registry.items():
|
|
441
|
+
score = _score_entry(terms_by_name[name], query_terms, idf)
|
|
442
|
+
if entry.smartapi_id in _CORE_API_IDS:
|
|
443
|
+
# Multiplicative, not additive: a zero score stays zero, so the
|
|
444
|
+
# boost reorders results that already match and never promotes a
|
|
445
|
+
# core API into a query it has nothing to do with.
|
|
446
|
+
score *= CORE_API_BOOST
|
|
447
|
+
scored.append((score, entry))
|
|
263
448
|
scored = [pair for pair in scored if pair[0] > 0]
|
|
264
449
|
scored.sort(key=lambda pair: (-pair[0], pair[1].name))
|
|
265
|
-
return [
|
|
450
|
+
return [
|
|
451
|
+
{**_as_dict(entry), "score": round(score, 3)} for score, entry in scored[:limit]
|
|
452
|
+
]
|
|
266
453
|
|
|
267
454
|
|
|
268
455
|
def build_biothings_facade(
|
|
@@ -15,7 +15,7 @@ from awslabs.openapi_mcp_server.server import get_all_counts
|
|
|
15
15
|
from awslabs.openapi_mcp_server.utils.metrics_provider import metrics
|
|
16
16
|
|
|
17
17
|
from .config import load_config
|
|
18
|
-
from .server import build_server_for_set
|
|
18
|
+
from .server import TOOL_SEARCH_MODES, build_server_for_set
|
|
19
19
|
|
|
20
20
|
|
|
21
21
|
def main():
|
|
@@ -26,7 +26,7 @@ def main():
|
|
|
26
26
|
"--api_set",
|
|
27
27
|
help=(
|
|
28
28
|
"A predefined set of SmartAPI APIs to include. One of: "
|
|
29
|
-
"'biothings_core' (
|
|
29
|
+
"'biothings_core' (the 6 core BioThings APIs), 'biothings_test' "
|
|
30
30
|
"(core + SemmedDB), or 'biothings_all' (all BioThings APIs). "
|
|
31
31
|
"[env: SMARTAPI_API_SET]"
|
|
32
32
|
),
|
|
@@ -88,7 +88,7 @@ def main():
|
|
|
88
88
|
parser.add_argument(
|
|
89
89
|
"--facade",
|
|
90
90
|
choices=["auto", "on", "off"],
|
|
91
|
-
default=
|
|
91
|
+
default=None,
|
|
92
92
|
help=(
|
|
93
93
|
"How to expose large BioThings sets. The facade collapses BioThings "
|
|
94
94
|
"APIs into ~5 generic tools (the target API is a parameter); any "
|
|
@@ -102,7 +102,7 @@ def main():
|
|
|
102
102
|
parser.add_argument(
|
|
103
103
|
"--facade-threshold",
|
|
104
104
|
type=int,
|
|
105
|
-
default=
|
|
105
|
+
default=None,
|
|
106
106
|
help=(
|
|
107
107
|
"Number of BioThings APIs in the set at which 'auto' switches to the "
|
|
108
108
|
"facade (default: 10). [env: FACADE_THRESHOLD]"
|
|
@@ -118,6 +118,45 @@ def main():
|
|
|
118
118
|
"[env: FACADE_STRICT]"
|
|
119
119
|
),
|
|
120
120
|
)
|
|
121
|
+
parser.add_argument(
|
|
122
|
+
"--tool-search",
|
|
123
|
+
choices=list(TOOL_SEARCH_MODES),
|
|
124
|
+
default=None,
|
|
125
|
+
help=(
|
|
126
|
+
"How to expose the tool listing. Serving many APIs produces hundreds "
|
|
127
|
+
"of tools, which crowds out a client's context. When search is on, "
|
|
128
|
+
"clients see 'search_tools' and 'call_tool' (plus any facade tools, "
|
|
129
|
+
"which stay listed) and discover the rest on demand; every tool "
|
|
130
|
+
"remains callable via 'call_tool'. 'auto' (default) turns search on "
|
|
131
|
+
"once the server reaches --tool-search-threshold tools. 'bm25' and "
|
|
132
|
+
"'regex' force it on regardless of size; 'off' always lists "
|
|
133
|
+
"everything. Prefer 'bm25' over 'regex': regex needs a real pattern "
|
|
134
|
+
"and returns nothing if given a natural-language query. CLI "
|
|
135
|
+
"overrides the environment variable, which overrides the default. "
|
|
136
|
+
"[env: SMARTAPI_TOOL_SEARCH]"
|
|
137
|
+
),
|
|
138
|
+
)
|
|
139
|
+
parser.add_argument(
|
|
140
|
+
"--tool-search-threshold",
|
|
141
|
+
type=int,
|
|
142
|
+
default=None,
|
|
143
|
+
help=(
|
|
144
|
+
"Tool count at which --tool-search 'auto' turns search on "
|
|
145
|
+
"(default: 15). A listed tool costs ~300-1000 tokens of client "
|
|
146
|
+
"context, so 15 is roughly a 5-15k-token ceiling on the listing. "
|
|
147
|
+
"[env: TOOL_SEARCH_THRESHOLD]"
|
|
148
|
+
),
|
|
149
|
+
)
|
|
150
|
+
parser.add_argument(
|
|
151
|
+
"--tool-search-max-results",
|
|
152
|
+
type=int,
|
|
153
|
+
default=None,
|
|
154
|
+
help=(
|
|
155
|
+
"Maximum number of tools returned per 'search_tools' call "
|
|
156
|
+
"(default: 10). Only used when tool search is active. "
|
|
157
|
+
"[env: TOOL_SEARCH_MAX_RESULTS]"
|
|
158
|
+
),
|
|
159
|
+
)
|
|
121
160
|
parser.add_argument(
|
|
122
161
|
"--log-level",
|
|
123
162
|
choices=["DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL"],
|
|
@@ -150,6 +189,8 @@ def main():
|
|
|
150
189
|
facade=getattr(config, "facade", "auto"),
|
|
151
190
|
facade_threshold=getattr(config, "facade_threshold", 10),
|
|
152
191
|
facade_strict=getattr(config, "facade_strict", False),
|
|
192
|
+
tool_search=getattr(config, "tool_search", "auto"),
|
|
193
|
+
tool_search_max_results=getattr(config, "tool_search_max_results", 5),
|
|
153
194
|
)
|
|
154
195
|
)
|
|
155
196
|
except ValueError as e:
|
|
@@ -18,6 +18,9 @@ class Config(_config.Config):
|
|
|
18
18
|
facade: str = "auto"
|
|
19
19
|
facade_threshold: int = 10
|
|
20
20
|
facade_strict: bool = False
|
|
21
|
+
tool_search: str = "auto"
|
|
22
|
+
tool_search_max_results: int = 10
|
|
23
|
+
tool_search_threshold: int = 15
|
|
21
24
|
|
|
22
25
|
|
|
23
26
|
def _parse_bool(value: str) -> bool:
|
|
@@ -51,6 +54,15 @@ def load_config(args: Any = None) -> Config:
|
|
|
51
54
|
lambda v: setattr(config, "facade_threshold", _parse_int(v, 10))
|
|
52
55
|
),
|
|
53
56
|
"FACADE_STRICT": (lambda v: setattr(config, "facade_strict", _parse_bool(v))),
|
|
57
|
+
"SMARTAPI_TOOL_SEARCH": (
|
|
58
|
+
lambda v: setattr(config, "tool_search", v.strip().lower())
|
|
59
|
+
),
|
|
60
|
+
"TOOL_SEARCH_MAX_RESULTS": (
|
|
61
|
+
lambda v: setattr(config, "tool_search_max_results", _parse_int(v, 10))
|
|
62
|
+
),
|
|
63
|
+
"TOOL_SEARCH_THRESHOLD": (
|
|
64
|
+
lambda v: setattr(config, "tool_search_threshold", _parse_int(v, 15))
|
|
65
|
+
),
|
|
54
66
|
"SERVER_NAME": (lambda v: setattr(config, "server_name", v)),
|
|
55
67
|
}
|
|
56
68
|
|
|
@@ -108,6 +120,12 @@ def load_config(args: Any = None) -> Config:
|
|
|
108
120
|
config.facade_threshold = int(args.facade_threshold)
|
|
109
121
|
if getattr(args, "facade_strict", False):
|
|
110
122
|
config.facade_strict = True
|
|
123
|
+
if getattr(args, "tool_search", None):
|
|
124
|
+
config.tool_search = str(args.tool_search).strip().lower()
|
|
125
|
+
if getattr(args, "tool_search_max_results", None):
|
|
126
|
+
config.tool_search_max_results = int(args.tool_search_max_results)
|
|
127
|
+
if getattr(args, "tool_search_threshold", None):
|
|
128
|
+
config.tool_search_threshold = int(args.tool_search_threshold)
|
|
111
129
|
if hasattr(args, "transport") and args.transport:
|
|
112
130
|
logger.debug(
|
|
113
131
|
f"Setting MCP Server transport mode from arguments: {args.transport}"
|
|
@@ -6,14 +6,25 @@ Main MCP server implementation for SmartAPI integration.
|
|
|
6
6
|
|
|
7
7
|
import hashlib
|
|
8
8
|
import re
|
|
9
|
+
from collections.abc import Iterable
|
|
9
10
|
|
|
10
11
|
from awslabs.openapi_mcp_server import logger
|
|
11
12
|
from awslabs.openapi_mcp_server.api.config import Config
|
|
12
13
|
from awslabs.openapi_mcp_server.server import create_mcp_server_async
|
|
13
14
|
from fastmcp import FastMCP
|
|
15
|
+
from fastmcp.server.transforms.search import (
|
|
16
|
+
BM25SearchTransform,
|
|
17
|
+
RegexSearchTransform,
|
|
18
|
+
serialize_tools_for_output_markdown,
|
|
19
|
+
)
|
|
14
20
|
|
|
15
21
|
# Import BioThings generic-facade builder
|
|
16
|
-
from .biothings import
|
|
22
|
+
from .biothings import (
|
|
23
|
+
build_biothings_facade,
|
|
24
|
+
build_registry,
|
|
25
|
+
is_biothings_family,
|
|
26
|
+
partition_biothings,
|
|
27
|
+
)
|
|
17
28
|
|
|
18
29
|
# Import from smartapi module - avoiding circular imports
|
|
19
30
|
from .smartapi import (
|
|
@@ -33,6 +44,121 @@ from .smartapi import (
|
|
|
33
44
|
# clients that reuse the tool-name validator.
|
|
34
45
|
MAX_TOOL_NAME_LEN = 64
|
|
35
46
|
|
|
47
|
+
# Ways to expose a large tool catalog. "off" lists every tool; "bm25"/"regex"
|
|
48
|
+
# always replace the catalog with a search interface; "auto" picks between them
|
|
49
|
+
# by tool count (see :func:`apply_tool_search`).
|
|
50
|
+
TOOL_SEARCH_MODES = ("auto", "off", "bm25", "regex")
|
|
51
|
+
|
|
52
|
+
# Mode "auto" resolves to. BM25 handles natural-language queries; regex needs the
|
|
53
|
+
# caller to author a pattern and returns nothing (silently) if handed prose.
|
|
54
|
+
TOOL_SEARCH_AUTO_MODE = "bm25"
|
|
55
|
+
|
|
56
|
+
# Tool count at which "auto" turns search on.
|
|
57
|
+
#
|
|
58
|
+
# Measured over the registry's uptime-passing set (592 tools, 92 APIs), a single
|
|
59
|
+
# entry in `tools/list` -- name plus the enriched description plus the JSON input
|
|
60
|
+
# schema -- averages ~3,900 characters (~975 tokens), median ~1,270 (~320), p90
|
|
61
|
+
# ~7,500, with one TRAPI tool at 84,000 (~21,000 tokens). At this threshold a
|
|
62
|
+
# listing therefore costs roughly 5k tokens of median-sized tools or 15k of
|
|
63
|
+
# mean-sized ones, which is a reasonable ceiling to pay before search is worth
|
|
64
|
+
# its extra round trip.
|
|
65
|
+
#
|
|
66
|
+
# Note the 65x spread: tool *count* is a crude proxy for the thing we actually
|
|
67
|
+
# care about, which is payload size. A byte/token budget would be the better
|
|
68
|
+
# instrument and would make this constant a floor rather than the decision.
|
|
69
|
+
TOOL_SEARCH_AUTO_THRESHOLD = 15
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
async def apply_tool_search(
|
|
73
|
+
server: FastMCP,
|
|
74
|
+
mode: str = "off",
|
|
75
|
+
*,
|
|
76
|
+
max_results: int = 10,
|
|
77
|
+
always_visible: Iterable[str] = (),
|
|
78
|
+
threshold: int = TOOL_SEARCH_AUTO_THRESHOLD,
|
|
79
|
+
) -> FastMCP:
|
|
80
|
+
"""Collapse ``server``'s tool catalog behind a search interface.
|
|
81
|
+
|
|
82
|
+
Serving many APIs from one server makes the tool list long enough to crowd
|
|
83
|
+
out a client's context: the ``biothings_all`` set is ~50 APIs at ~6 tools
|
|
84
|
+
each. A search transform replaces the listed catalog with two synthetic
|
|
85
|
+
tools -- ``search_tools`` and ``call_tool`` -- so a model discovers tools on
|
|
86
|
+
demand instead of receiving every schema upfront. Every real tool remains
|
|
87
|
+
callable through ``call_tool``; only the *listing* changes.
|
|
88
|
+
|
|
89
|
+
Names in ``always_visible`` stay listed alongside the synthetic tools.
|
|
90
|
+
:func:`build_server_for_set` pins the BioThings facade tools this way, so
|
|
91
|
+
the common path stays directly callable and only the per-API long tail is
|
|
92
|
+
collapsed. That combination is the intended arrangement: the facade answers
|
|
93
|
+
BioThings queries directly (where lexical search is weakest, because the
|
|
94
|
+
generated per-API descriptions are near-identical boilerplate), and search
|
|
95
|
+
covers the non-BioThings tail (where it works well).
|
|
96
|
+
|
|
97
|
+
``mode`` is one of :data:`TOOL_SEARCH_MODES`:
|
|
98
|
+
|
|
99
|
+
``"auto"``
|
|
100
|
+
Enable :data:`TOOL_SEARCH_AUTO_MODE` once the server has at least
|
|
101
|
+
``threshold`` tools; leave smaller catalogs listed in full.
|
|
102
|
+
``"off"``
|
|
103
|
+
Leave the catalog alone.
|
|
104
|
+
``"bm25"`` / ``"regex"``
|
|
105
|
+
Always enable that transform, regardless of size.
|
|
106
|
+
|
|
107
|
+
``max_results`` caps the hits per search. ``server`` is mutated in place and
|
|
108
|
+
returned.
|
|
109
|
+
"""
|
|
110
|
+
if mode not in TOOL_SEARCH_MODES:
|
|
111
|
+
err_msg = (
|
|
112
|
+
f"Unknown tool search mode {mode!r}; "
|
|
113
|
+
f"expected one of: {', '.join(TOOL_SEARCH_MODES)}."
|
|
114
|
+
)
|
|
115
|
+
raise ValueError(err_msg)
|
|
116
|
+
if mode == "off":
|
|
117
|
+
return server
|
|
118
|
+
|
|
119
|
+
tool_count = len(await server.list_tools())
|
|
120
|
+
if not tool_count:
|
|
121
|
+
# Leave an empty server alone so the caller's "no tools registered"
|
|
122
|
+
# diagnostics still fire instead of counting the synthetic tools.
|
|
123
|
+
logger.warning(
|
|
124
|
+
f"Tool search ({mode}) requested but the server has no tools; "
|
|
125
|
+
"leaving the catalog unchanged."
|
|
126
|
+
)
|
|
127
|
+
return server
|
|
128
|
+
|
|
129
|
+
if mode == "auto":
|
|
130
|
+
if tool_count < threshold:
|
|
131
|
+
logger.info(
|
|
132
|
+
f"Tool search (auto): {tool_count} tools is below the "
|
|
133
|
+
f"{threshold}-tool threshold; listing them all."
|
|
134
|
+
)
|
|
135
|
+
return server
|
|
136
|
+
logger.info(
|
|
137
|
+
f"Tool search (auto): {tool_count} tools reaches the "
|
|
138
|
+
f"{threshold}-tool threshold; enabling {TOOL_SEARCH_AUTO_MODE}."
|
|
139
|
+
)
|
|
140
|
+
mode = TOOL_SEARCH_AUTO_MODE
|
|
141
|
+
|
|
142
|
+
pinned = sorted(always_visible)
|
|
143
|
+
transform_cls = BM25SearchTransform if mode == "bm25" else RegexSearchTransform
|
|
144
|
+
server.add_transform(
|
|
145
|
+
transform_cls(
|
|
146
|
+
max_results=max_results,
|
|
147
|
+
always_visible=pinned,
|
|
148
|
+
# Markdown results are roughly half the size of the default JSON
|
|
149
|
+
# serialization, which is the point when enabling search at all.
|
|
150
|
+
search_result_serializer=serialize_tools_for_output_markdown,
|
|
151
|
+
)
|
|
152
|
+
)
|
|
153
|
+
exposed = len(await server.list_tools())
|
|
154
|
+
logger.info(
|
|
155
|
+
f"Tool search ({mode}) enabled: {tool_count} tools collapsed to "
|
|
156
|
+
f"{exposed} listed ({len(pinned)} pinned + search_tools/call_tool); "
|
|
157
|
+
f"max_results={max_results}. All {tool_count} tools stay callable "
|
|
158
|
+
"via call_tool."
|
|
159
|
+
)
|
|
160
|
+
return server
|
|
161
|
+
|
|
36
162
|
|
|
37
163
|
async def get_mcp_server(smartapi_id: str) -> FastMCP:
|
|
38
164
|
config = Config(
|
|
@@ -72,6 +198,53 @@ def _fit_name(name: str, used: set[str]) -> str:
|
|
|
72
198
|
return truncated
|
|
73
199
|
|
|
74
200
|
|
|
201
|
+
async def build_api_servers(
|
|
202
|
+
smartapi_ids: list[str],
|
|
203
|
+
) -> tuple[list[FastMCP], list[tuple[str, str]]]:
|
|
204
|
+
"""Build one MCP server per SmartAPI id, skipping the ones that fail.
|
|
205
|
+
|
|
206
|
+
Returns ``(servers, failures)`` where each failure is ``(smartapi_id,
|
|
207
|
+
reason)``.
|
|
208
|
+
|
|
209
|
+
Not every registered spec can be turned into a server: some use external
|
|
210
|
+
``$ref``s (refused by awslabs 1.x as an SSRF guard), some are invalid
|
|
211
|
+
OpenAPI, some have no ``servers`` block. Roughly one in six of the
|
|
212
|
+
registry's uptime-passing APIs fails for one of those reasons, and a single
|
|
213
|
+
one of them used to abort the whole build -- so ``--smartapi_q
|
|
214
|
+
'_status.uptime_status:pass'`` could not start at all. Serving the APIs that
|
|
215
|
+
do work, and reporting the rest, is far more useful than serving none.
|
|
216
|
+
|
|
217
|
+
Kept sequential like the code it replaces: fanning these out concurrently
|
|
218
|
+
makes the SmartAPI registry start refusing DNS/connections partway through,
|
|
219
|
+
which turns working APIs into spurious failures.
|
|
220
|
+
"""
|
|
221
|
+
servers: list[FastMCP] = []
|
|
222
|
+
failures: list[tuple[str, str]] = []
|
|
223
|
+
for sid in smartapi_ids:
|
|
224
|
+
try:
|
|
225
|
+
servers.append(await get_mcp_server(sid))
|
|
226
|
+
# SystemExit is caught alongside Exception on purpose. awslabs'
|
|
227
|
+
# create_mcp_server_async reports *every* spec error by calling
|
|
228
|
+
# sys.exit(1) from inside the library, so a spec that fastmcp itself
|
|
229
|
+
# rejects (e.g. an OpenAPI 3.0 document using "type": "null") raises
|
|
230
|
+
# SystemExit rather than an Exception -- which "except Exception" does
|
|
231
|
+
# not catch, and which therefore still took down every other API in the
|
|
232
|
+
# set. Confirmed on the registry's uptime-passing set, where it killed a
|
|
233
|
+
# 27-API build at API 17. This is scoped tightly to one call, so it
|
|
234
|
+
# cannot swallow a genuine interpreter exit (and Ctrl-C raises
|
|
235
|
+
# KeyboardInterrupt, not SystemExit).
|
|
236
|
+
except (Exception, SystemExit) as exc: # any spec problem is survivable
|
|
237
|
+
reason = f"{type(exc).__name__}: {str(exc)[:200]}"
|
|
238
|
+
failures.append((sid, reason))
|
|
239
|
+
logger.warning(f"Skipping SmartAPI {sid}: {reason}")
|
|
240
|
+
if failures:
|
|
241
|
+
logger.warning(
|
|
242
|
+
f"{len(failures)} of {len(smartapi_ids)} API(s) could not be loaded "
|
|
243
|
+
f"and were skipped; {len(servers)} loaded successfully."
|
|
244
|
+
)
|
|
245
|
+
return servers, failures
|
|
246
|
+
|
|
247
|
+
|
|
75
248
|
async def _merge_servers_into(
|
|
76
249
|
target: FastMCP, list_of_servers: list[FastMCP]
|
|
77
250
|
) -> FastMCP:
|
|
@@ -83,33 +256,41 @@ async def _merge_servers_into(
|
|
|
83
256
|
# Seed with names already in the target (e.g. facade tools in the hybrid
|
|
84
257
|
# path) so merged per-API tools/prompts never collide with them. Tools and
|
|
85
258
|
# prompts have separate namespaces, so each gets its own set.
|
|
86
|
-
used_tool_names: set[str] =
|
|
87
|
-
used_prompt_names: set[str] =
|
|
259
|
+
used_tool_names: set[str] = {tool.name for tool in await target.list_tools()}
|
|
260
|
+
used_prompt_names: set[str] = {
|
|
261
|
+
prompt.name for prompt in await target.list_prompts()
|
|
262
|
+
}
|
|
88
263
|
for server in list_of_servers:
|
|
89
264
|
api_name = re.sub(
|
|
90
265
|
r"[^a-z0-9_-]", "_", getattr(server, "name", "unknown_api").lower()
|
|
91
266
|
)
|
|
92
267
|
|
|
93
|
-
tools = await server.
|
|
268
|
+
tools = await server.list_tools()
|
|
94
269
|
if tools:
|
|
95
|
-
for
|
|
270
|
+
for tool in tools:
|
|
96
271
|
# Rename the tool by prefixing with API name, keeping it within
|
|
97
|
-
# the 64-char limit that MCP clients enforce.
|
|
98
|
-
|
|
272
|
+
# the 64-char limit that MCP clients enforce. Renaming in place
|
|
273
|
+
# is safe: Component.key is a property derived from .name, so
|
|
274
|
+
# the target registers the tool under its new name.
|
|
275
|
+
prefixed = f"{api_name}_{tool.name}"
|
|
99
276
|
tool.name = _fit_name(prefixed, used_tool_names)
|
|
100
277
|
used_tool_names.add(tool.name)
|
|
101
278
|
target.add_tool(tool)
|
|
102
279
|
else:
|
|
103
|
-
|
|
104
|
-
|
|
280
|
+
# A spec that parses but yields no tools is a property of that one
|
|
281
|
+
# API, not a reason to lose every other API in the set.
|
|
282
|
+
logger.warning(
|
|
283
|
+
f"API '{api_name}' contributed no tools; skipping it. Its spec "
|
|
284
|
+
"parsed but produced no callable operations."
|
|
285
|
+
)
|
|
105
286
|
|
|
106
287
|
# Merge prompts
|
|
107
|
-
prompts = await server.
|
|
288
|
+
prompts = await server.list_prompts()
|
|
108
289
|
if prompts:
|
|
109
|
-
for
|
|
290
|
+
for prompt in prompts:
|
|
110
291
|
# Rename the prompt by prefixing with API name, keeping it
|
|
111
292
|
# within the 64-char limit that MCP clients enforce.
|
|
112
|
-
prefixed = f"{api_name}_{
|
|
293
|
+
prefixed = f"{api_name}_{prompt.name}"
|
|
113
294
|
prompt.name = _fit_name(prefixed, used_prompt_names)
|
|
114
295
|
used_prompt_names.add(prompt.name)
|
|
115
296
|
target.add_prompt(prompt)
|
|
@@ -168,13 +349,12 @@ async def get_merged_mcp_server(
|
|
|
168
349
|
err_msg = "No SmartAPI IDs provided or found with the given query."
|
|
169
350
|
raise ValueError(err_msg)
|
|
170
351
|
smartapi_exclude_ids = smartapi_exclude_ids or []
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
for sid in smartapi_ids
|
|
174
|
-
if sid not in smartapi_exclude_ids
|
|
175
|
-
]
|
|
352
|
+
wanted = [sid for sid in smartapi_ids if sid not in smartapi_exclude_ids]
|
|
353
|
+
list_of_servers, _failures = await build_api_servers(wanted)
|
|
176
354
|
merged_server = await merge_mcp_servers(list_of_servers, server_name)
|
|
177
|
-
logger.info(
|
|
355
|
+
logger.info(
|
|
356
|
+
f"Merged {len(list_of_servers)} of {len(wanted)} APIs into one MCP server."
|
|
357
|
+
)
|
|
178
358
|
return merged_server
|
|
179
359
|
|
|
180
360
|
|
|
@@ -221,6 +401,9 @@ async def build_server_for_set(
|
|
|
221
401
|
facade: str = "auto",
|
|
222
402
|
facade_threshold: int = 10,
|
|
223
403
|
facade_strict: bool = False,
|
|
404
|
+
tool_search: str = "auto",
|
|
405
|
+
tool_search_max_results: int = 10,
|
|
406
|
+
tool_search_threshold: int = TOOL_SEARCH_AUTO_THRESHOLD,
|
|
224
407
|
) -> FastMCP:
|
|
225
408
|
"""Build the MCP server for an API set, picking the right strategy.
|
|
226
409
|
|
|
@@ -240,6 +423,15 @@ async def build_server_for_set(
|
|
|
240
423
|
downloads). When ``True``, each BioThings spec is inspected and any API with
|
|
241
424
|
extra endpoints is served with faithful per-API tools instead (slower
|
|
242
425
|
startup; downloads specs upfront).
|
|
426
|
+
|
|
427
|
+
``tool_search`` (see :data:`TOOL_SEARCH_MODES`, default ``"auto"``)
|
|
428
|
+
additionally collapses the tool listing behind a search interface, which is
|
|
429
|
+
the answer to the per-API tool explosion when the facade does not apply.
|
|
430
|
+
Facade tools are pinned so they stay listed, giving a hybrid server whose
|
|
431
|
+
BioThings half is answered by the facade and whose per-API half is
|
|
432
|
+
discovered by search. ``"auto"`` engages only once the merged server has
|
|
433
|
+
``tool_search_threshold`` tools; ``tool_search_max_results`` caps hits per
|
|
434
|
+
search.
|
|
243
435
|
"""
|
|
244
436
|
available_ids = await _resolve_smartapi_ids(
|
|
245
437
|
smartapi_q=smartapi_q,
|
|
@@ -251,8 +443,13 @@ async def build_server_for_set(
|
|
|
251
443
|
|
|
252
444
|
if facade != "off":
|
|
253
445
|
registry = await build_registry(available_ids)
|
|
446
|
+
# TRAPI services carry the "biothings" tag but are not annotation APIs;
|
|
447
|
+
# is_biothings_family excludes them so they fall through to the
|
|
448
|
+
# non_biothings_ids branch below and get faithful per-API tools.
|
|
254
449
|
biothings = {
|
|
255
|
-
name: entry
|
|
450
|
+
name: entry
|
|
451
|
+
for name, entry in registry.items()
|
|
452
|
+
if is_biothings_family(entry)
|
|
256
453
|
}
|
|
257
454
|
if facade == "on" and not biothings:
|
|
258
455
|
logger.warning(
|
|
@@ -279,26 +476,47 @@ async def build_server_for_set(
|
|
|
279
476
|
|
|
280
477
|
if facade_entries:
|
|
281
478
|
server = build_biothings_facade(facade_entries, server_name)
|
|
479
|
+
# Capture the facade tools before merging per-API servers so
|
|
480
|
+
# tool search can pin them and collapse only the long tail.
|
|
481
|
+
# Skipped entirely when tool search is off, to keep the default
|
|
482
|
+
# path free of extra work.
|
|
483
|
+
facade_tool_names: list[str] = (
|
|
484
|
+
[tool.name for tool in await server.list_tools()]
|
|
485
|
+
if tool_search != "off"
|
|
486
|
+
else []
|
|
487
|
+
)
|
|
282
488
|
if per_api_ids:
|
|
283
489
|
logger.info(
|
|
284
490
|
f"Hybrid server: facade over {len(facade_entries)} "
|
|
285
491
|
f"BioThings API(s) + per-API tools for "
|
|
286
492
|
f"{len(per_api_ids)} other API(s)."
|
|
287
493
|
)
|
|
288
|
-
extra_servers =
|
|
494
|
+
extra_servers, _failures = await build_api_servers(per_api_ids)
|
|
289
495
|
await _merge_servers_into(server, extra_servers)
|
|
290
496
|
else:
|
|
291
497
|
logger.info(
|
|
292
498
|
f"Using BioThings facade for {len(facade_entries)} APIs "
|
|
293
499
|
f"(server_name={server_name})."
|
|
294
500
|
)
|
|
295
|
-
return
|
|
501
|
+
return await apply_tool_search(
|
|
502
|
+
server,
|
|
503
|
+
tool_search,
|
|
504
|
+
max_results=tool_search_max_results,
|
|
505
|
+
always_visible=facade_tool_names,
|
|
506
|
+
threshold=tool_search_threshold,
|
|
507
|
+
)
|
|
296
508
|
logger.info(
|
|
297
509
|
"No APIs qualified for the BioThings facade; using per-API tools."
|
|
298
510
|
)
|
|
299
511
|
|
|
300
512
|
logger.info(f"Using per-API tools for {len(available_ids)} APIs.")
|
|
301
|
-
|
|
513
|
+
server = await get_merged_mcp_server(
|
|
302
514
|
smartapi_ids=available_ids,
|
|
303
515
|
server_name=server_name,
|
|
304
516
|
)
|
|
517
|
+
return await apply_tool_search(
|
|
518
|
+
server,
|
|
519
|
+
tool_search,
|
|
520
|
+
max_results=tool_search_max_results,
|
|
521
|
+
threshold=tool_search_threshold,
|
|
522
|
+
)
|
|
@@ -106,33 +106,34 @@ def get_base_server_url(api_spec: dict) -> str:
|
|
|
106
106
|
return base_server_url
|
|
107
107
|
|
|
108
108
|
|
|
109
|
+
# The core BioThings APIs: the canonical, broad-coverage annotation services,
|
|
110
|
+
# as distinct from the ~50 single-source satellite APIs. Named because it serves
|
|
111
|
+
# two purposes that must not drift apart -- the ``biothings_core`` preset (which
|
|
112
|
+
# APIs to serve) and discovery ranking (which APIs to prefer when several match
|
|
113
|
+
# a query, see ``CORE_API_BOOST`` in :mod:`smartapi_mcp.biothings`).
|
|
114
|
+
CORE_BIOTHINGS_API_IDS = [
|
|
115
|
+
"59dce17363dce279d389100834e43648", # MyGene.info
|
|
116
|
+
"09c8782d9f4027712e65b95424adba79", # MyVariant.info
|
|
117
|
+
"8f08d1446e0bb9c2b323713ce83e2bd3", # MyChem.info
|
|
118
|
+
"671b45c0301c8624abbd26ae78449ca2", # MyDisease.info
|
|
119
|
+
"85139f4dccfcefa3ac3042372066916d", # MyGeneSet.info
|
|
120
|
+
"f7943e6167166b3ea9e4b8be08f45fa6", # MyTaxon.info
|
|
121
|
+
]
|
|
122
|
+
|
|
123
|
+
# SemmedDB, added to the "test" set for its non-standard /query/ngd endpoint.
|
|
124
|
+
_SEMMEDDB_ID = "1d288b3a3caf75d541ffaae3aab386c8"
|
|
125
|
+
|
|
109
126
|
PREDEFINED_API_SETS = ["biothings_core", "biothings_test", "biothings_all"]
|
|
110
127
|
|
|
111
128
|
|
|
112
129
|
def get_predefined_api_set(api_set: str) -> dict:
|
|
113
130
|
"""Return the predefined API set for the given set name."""
|
|
114
131
|
if api_set == "biothings_core":
|
|
115
|
-
return {
|
|
116
|
-
"smartapi_ids": [
|
|
117
|
-
"59dce17363dce279d389100834e43648", # MyGene.info
|
|
118
|
-
"09c8782d9f4027712e65b95424adba79", # MyVariant.info
|
|
119
|
-
"8f08d1446e0bb9c2b323713ce83e2bd3", # MyChem.info
|
|
120
|
-
"671b45c0301c8624abbd26ae78449ca2", # MyDisease.info
|
|
121
|
-
"85139f4dccfcefa3ac3042372066916d", # MyGeneSet.info
|
|
122
|
-
]
|
|
123
|
-
}
|
|
132
|
+
return {"smartapi_ids": list(CORE_BIOTHINGS_API_IDS)}
|
|
124
133
|
if api_set == "biothings_test":
|
|
125
|
-
#
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
"59dce17363dce279d389100834e43648", # MyGene.info
|
|
129
|
-
"09c8782d9f4027712e65b95424adba79", # MyVariant.info
|
|
130
|
-
"8f08d1446e0bb9c2b323713ce83e2bd3", # MyChem.info
|
|
131
|
-
"671b45c0301c8624abbd26ae78449ca2", # MyDisease.info
|
|
132
|
-
"85139f4dccfcefa3ac3042372066916d", # MyGeneSet.info
|
|
133
|
-
"1d288b3a3caf75d541ffaae3aab386c8", # SemmedDB
|
|
134
|
-
]
|
|
135
|
-
}
|
|
134
|
+
# The core APIs plus SemmedDB, whose /query/ngd endpoint exercises the
|
|
135
|
+
# non-standard-endpoint handling.
|
|
136
|
+
return {"smartapi_ids": [*CORE_BIOTHINGS_API_IDS, _SEMMEDDB_ID]}
|
|
136
137
|
if api_set == "biothings_all":
|
|
137
138
|
# include all biothings APIs with a few excluded
|
|
138
139
|
return {
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|