mcp-server-knowledgebase 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_server_knowledgebase/__init__.py +1 -0
- mcp_server_knowledgebase/common/__init__.py +0 -0
- mcp_server_knowledgebase/common/auth.py +53 -0
- mcp_server_knowledgebase/config.py +57 -0
- mcp_server_knowledgebase/models.py +66 -0
- mcp_server_knowledgebase/server.py +421 -0
- mcp_server_knowledgebase-0.2.0.dist-info/METADATA +248 -0
- mcp_server_knowledgebase-0.2.0.dist-info/RECORD +10 -0
- mcp_server_knowledgebase-0.2.0.dist-info/WHEEL +4 -0
- mcp_server_knowledgebase-0.2.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# Package initialization
|
|
File without changes
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
import json
|
|
2
|
+
|
|
3
|
+
from volcengine.auth.SignerV4 import SignerV4
|
|
4
|
+
from volcengine.base.Request import Request
|
|
5
|
+
from volcengine.Credentials import Credentials
|
|
6
|
+
from mcp_server_knowledgebase.config import config
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def prepare_request(
|
|
10
|
+
method, path, ak=None, sk=None, params=None, data=None, doseq=0, *, api_key=None
|
|
11
|
+
):
|
|
12
|
+
ak = ak.strip() if isinstance(ak, str) else ak
|
|
13
|
+
sk = sk.strip() if isinstance(sk, str) else sk
|
|
14
|
+
api_key = api_key.strip() if isinstance(api_key, str) else api_key
|
|
15
|
+
|
|
16
|
+
if not api_key:
|
|
17
|
+
if bool(ak) != bool(sk):
|
|
18
|
+
raise ValueError("AK and SK must be configured together")
|
|
19
|
+
if not ak or not sk:
|
|
20
|
+
raise ValueError("Configure an authentication method: VIKING_API_KEY or AK/SK")
|
|
21
|
+
|
|
22
|
+
if params:
|
|
23
|
+
for key in params:
|
|
24
|
+
if (
|
|
25
|
+
type(params[key]) == int
|
|
26
|
+
or type(params[key]) == float
|
|
27
|
+
or type(params[key]) == bool
|
|
28
|
+
):
|
|
29
|
+
params[key] = str(params[key])
|
|
30
|
+
elif type(params[key]) == list:
|
|
31
|
+
if not doseq:
|
|
32
|
+
params[key] = ",".join(params[key])
|
|
33
|
+
r = Request()
|
|
34
|
+
r.set_shema("https")
|
|
35
|
+
r.set_method(method)
|
|
36
|
+
r.set_connection_timeout(10)
|
|
37
|
+
r.set_socket_timeout(10)
|
|
38
|
+
mheaders = {
|
|
39
|
+
"Accept": "application/json",
|
|
40
|
+
"Content-Type": "application/json",
|
|
41
|
+
}
|
|
42
|
+
if api_key:
|
|
43
|
+
mheaders["Authorization"] = f"Bearer {api_key}"
|
|
44
|
+
r.set_headers(mheaders)
|
|
45
|
+
if params:
|
|
46
|
+
r.set_query(params)
|
|
47
|
+
r.set_path(path)
|
|
48
|
+
if data is not None:
|
|
49
|
+
r.set_body(json.dumps(data))
|
|
50
|
+
if not api_key:
|
|
51
|
+
credentials = Credentials(ak, sk, "air", config.region)
|
|
52
|
+
SignerV4.sign(r, credentials)
|
|
53
|
+
return r
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import os
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Optional
|
|
5
|
+
|
|
6
|
+
logger = logging.getLogger(__name__)
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass
|
|
10
|
+
class KnowledgeBaseConfig:
|
|
11
|
+
"""Configuration for Viking Knowledge Base MCP Server."""
|
|
12
|
+
|
|
13
|
+
ak: Optional[str] = None
|
|
14
|
+
sk: Optional[str] = None
|
|
15
|
+
project: Optional[str] = None
|
|
16
|
+
region: str = "cn-north-1"
|
|
17
|
+
api_key: Optional[str] = None
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _credential_from_env(name: str) -> Optional[str]:
|
|
21
|
+
value = os.environ.get(name)
|
|
22
|
+
if value is None:
|
|
23
|
+
return None
|
|
24
|
+
return value.strip() or None
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def load_config() -> KnowledgeBaseConfig:
|
|
28
|
+
"""Load configuration from environment variables."""
|
|
29
|
+
ak = _credential_from_env("VOLCENGINE_ACCESS_KEY")
|
|
30
|
+
sk = _credential_from_env("VOLCENGINE_SECRET_KEY")
|
|
31
|
+
api_key = _credential_from_env("VIKING_API_KEY")
|
|
32
|
+
|
|
33
|
+
if not api_key and bool(ak) != bool(sk):
|
|
34
|
+
error_msg = (
|
|
35
|
+
"VOLCENGINE_ACCESS_KEY and VOLCENGINE_SECRET_KEY must be configured together"
|
|
36
|
+
)
|
|
37
|
+
logger.error(error_msg)
|
|
38
|
+
raise ValueError(error_msg)
|
|
39
|
+
|
|
40
|
+
if not api_key and not (ak and sk):
|
|
41
|
+
error_msg = (
|
|
42
|
+
"Configure an authentication method: VIKING_API_KEY or "
|
|
43
|
+
"VOLCENGINE_ACCESS_KEY with VOLCENGINE_SECRET_KEY"
|
|
44
|
+
)
|
|
45
|
+
logger.error(error_msg)
|
|
46
|
+
raise ValueError(error_msg)
|
|
47
|
+
|
|
48
|
+
return KnowledgeBaseConfig(
|
|
49
|
+
ak=ak,
|
|
50
|
+
sk=sk,
|
|
51
|
+
api_key=api_key,
|
|
52
|
+
project=os.environ.get("KNOWLEDGE_BASE_PROJECT", "default"),
|
|
53
|
+
region=os.environ.get("KNOWLEDGE_BASE_REGION", "cn-north-1"),
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
config = load_config()
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
from typing import Any, Literal, Optional
|
|
2
|
+
|
|
3
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class DocFilter(BaseModel):
|
|
7
|
+
"""Filter applied to Viking Knowledge Base search results."""
|
|
8
|
+
|
|
9
|
+
op: Literal["must", "must_not"]
|
|
10
|
+
field: str = Field(min_length=1)
|
|
11
|
+
conds: list[Any] = Field(min_length=1)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class AddDocumentResult(BaseModel):
|
|
15
|
+
collection_name: str
|
|
16
|
+
doc_id: str
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class DocumentStatus(BaseModel):
|
|
20
|
+
model_config = ConfigDict(extra="allow")
|
|
21
|
+
|
|
22
|
+
process_status: Optional[int] = None
|
|
23
|
+
failed_code: Optional[str] = None
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class DocumentInfo(BaseModel):
|
|
27
|
+
"""Known document fields; extra upstream fields are preserved."""
|
|
28
|
+
|
|
29
|
+
model_config = ConfigDict(extra="allow")
|
|
30
|
+
|
|
31
|
+
collection_name: Optional[str] = None
|
|
32
|
+
doc_id: Optional[str] = None
|
|
33
|
+
doc_name: Optional[str] = None
|
|
34
|
+
doc_type: Optional[str] = None
|
|
35
|
+
url: Optional[str] = None
|
|
36
|
+
add_type: Optional[str] = None
|
|
37
|
+
create_time: Optional[int] = None
|
|
38
|
+
update_time: Optional[int] = None
|
|
39
|
+
point_num: Optional[int] = None
|
|
40
|
+
status: Optional[DocumentStatus] = None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class CollectionInfoResult(BaseModel):
|
|
44
|
+
collection_name: str
|
|
45
|
+
description: str
|
|
46
|
+
status: int
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class CollectionSummary(BaseModel):
|
|
50
|
+
collection_name: str
|
|
51
|
+
description: str
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class ListCollectionsResult(BaseModel):
|
|
55
|
+
collection_list: list[CollectionSummary]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class SearchChunk(BaseModel):
|
|
59
|
+
id: str
|
|
60
|
+
content: str
|
|
61
|
+
doc_id: Optional[str] = None
|
|
62
|
+
doc_name: Optional[str] = None
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class SearchKnowledgeResult(BaseModel):
|
|
66
|
+
result_list: list[SearchChunk]
|
|
@@ -0,0 +1,421 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import logging
|
|
3
|
+
import os
|
|
4
|
+
from typing import Annotated, Any, Dict, Literal, Optional
|
|
5
|
+
|
|
6
|
+
import aiohttp
|
|
7
|
+
from mcp.server import MCPServer
|
|
8
|
+
from mcp.server.caching import CacheHint
|
|
9
|
+
from mcp.server.mcpserver.exceptions import ToolError
|
|
10
|
+
from mcp.types import ToolAnnotations
|
|
11
|
+
from pydantic import Field
|
|
12
|
+
|
|
13
|
+
from mcp_server_knowledgebase.common.auth import prepare_request
|
|
14
|
+
from mcp_server_knowledgebase.config import config
|
|
15
|
+
from mcp_server_knowledgebase.models import (
|
|
16
|
+
AddDocumentResult,
|
|
17
|
+
CollectionInfoResult,
|
|
18
|
+
CollectionSummary,
|
|
19
|
+
DocFilter,
|
|
20
|
+
DocumentInfo,
|
|
21
|
+
ListCollectionsResult,
|
|
22
|
+
SearchChunk,
|
|
23
|
+
SearchKnowledgeResult,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
logger = logging.getLogger(__name__)
|
|
27
|
+
logging.basicConfig(
|
|
28
|
+
level=logging.INFO, format="%(asctime)s - %(name)s - %(levelname)s - %(message)s"
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
# knowledge base domain
|
|
32
|
+
g_knowledge_base_domain = "api-knowledgebase.mlp.cn-beijing.volces.com"
|
|
33
|
+
|
|
34
|
+
# paths
|
|
35
|
+
search_knowledge_path = "/api/knowledge/collection/search_knowledge"
|
|
36
|
+
list_collections_path = "/api/knowledge/collection/list"
|
|
37
|
+
get_collections_path = "/api/knowledge/collection/info"
|
|
38
|
+
doc_add_path = "/api/knowledge/doc/add"
|
|
39
|
+
doc_info_path = "/api/knowledge/doc/info"
|
|
40
|
+
|
|
41
|
+
# Create MCP server
|
|
42
|
+
mcp = MCPServer(
|
|
43
|
+
"Knowledgebase MCP Server",
|
|
44
|
+
version="0.2.0",
|
|
45
|
+
cache_hints={"tools/list": CacheHint(ttl_ms=300_000, scope="public")},
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _transport_options(transport: str) -> Dict[str, Any]:
|
|
50
|
+
"""Build transport-specific options accepted by MCP SDK v2."""
|
|
51
|
+
if transport == "stdio":
|
|
52
|
+
return {}
|
|
53
|
+
if transport != "streamable-http":
|
|
54
|
+
raise ValueError(f"Unsupported transport: {transport}")
|
|
55
|
+
return {
|
|
56
|
+
"host": os.getenv("MCP_SERVER_HOST", "127.0.0.1"),
|
|
57
|
+
"port": int(os.getenv("MCP_SERVER_PORT") or os.getenv("PORT", "8000")),
|
|
58
|
+
"streamable_http_path": os.getenv("STREAMABLE_HTTP_PATH", "/mcp"),
|
|
59
|
+
"stateless_http": True,
|
|
60
|
+
"json_response": True,
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
async def _request_knowledgebase(path: str, data: Dict[str, Any]) -> Dict[str, Any]:
|
|
65
|
+
"""Send one signed request without blocking the MCP event loop."""
|
|
66
|
+
request = prepare_request(
|
|
67
|
+
method="POST",
|
|
68
|
+
path=path,
|
|
69
|
+
ak=config.ak,
|
|
70
|
+
sk=config.sk,
|
|
71
|
+
api_key=config.api_key,
|
|
72
|
+
data=data,
|
|
73
|
+
)
|
|
74
|
+
timeout = aiohttp.ClientTimeout(total=float(os.getenv("KNOWLEDGE_BASE_TIMEOUT", "30")))
|
|
75
|
+
async with aiohttp.ClientSession(timeout=timeout) as session:
|
|
76
|
+
async with session.request(
|
|
77
|
+
method=request.method,
|
|
78
|
+
url=f"https://{g_knowledge_base_domain}{request.path}",
|
|
79
|
+
headers=request.headers,
|
|
80
|
+
data=request.body,
|
|
81
|
+
) as response:
|
|
82
|
+
response.raise_for_status()
|
|
83
|
+
return await response.json()
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@mcp.tool(
|
|
87
|
+
annotations=ToolAnnotations(
|
|
88
|
+
readOnlyHint=False,
|
|
89
|
+
destructiveHint=False,
|
|
90
|
+
idempotentHint=False,
|
|
91
|
+
openWorldHint=True,
|
|
92
|
+
)
|
|
93
|
+
)
|
|
94
|
+
async def add_doc(
|
|
95
|
+
collection_name: Annotated[str, Field(min_length=1)],
|
|
96
|
+
add_type: Literal["url"],
|
|
97
|
+
doc_id: Annotated[
|
|
98
|
+
str,
|
|
99
|
+
Field(min_length=1, max_length=128, pattern=r"^[A-Za-z][A-Za-z0-9_]*$"),
|
|
100
|
+
],
|
|
101
|
+
doc_name: Annotated[str, Field(min_length=1, max_length=256)],
|
|
102
|
+
doc_type: Literal[
|
|
103
|
+
"xlsx",
|
|
104
|
+
"csv",
|
|
105
|
+
"jsonl",
|
|
106
|
+
"txt",
|
|
107
|
+
"doc",
|
|
108
|
+
"docx",
|
|
109
|
+
"pdf",
|
|
110
|
+
"markdown",
|
|
111
|
+
"faq.xlsx",
|
|
112
|
+
"pptx",
|
|
113
|
+
],
|
|
114
|
+
url: Annotated[str, Field(min_length=1)],
|
|
115
|
+
) -> AddDocumentResult:
|
|
116
|
+
"""
|
|
117
|
+
Add a document to a collection in your project.
|
|
118
|
+
This tool allows you to add a document to a collection in your project by collection_name.
|
|
119
|
+
Args:
|
|
120
|
+
collection_name: the name of the knowledge base collection to add document to.
|
|
121
|
+
add_type: the type of the document to add. so far only support "url" now. so you must assign this parameter to "url".
|
|
122
|
+
doc_id: you should generate a unique doc_id based on user's given url and timestamp, the doc_id can only use English letters, numbers, and underscores , and must start with an English letter. It cannot be empty.
|
|
123
|
+
Length requirement: [1, 128], you can use a format like "mcp_server_auto_gen_doc_id_xxxxxxx.
|
|
124
|
+
doc_name: the name of the document to add. you can1 generate a unique doc_name based on user given url and timestamp. the length of doc_name must between 1 and 256. you can use a
|
|
125
|
+
format like "mcp_server_auto_gen_doc_name_xxxxxxx.
|
|
126
|
+
doc_type: the type of the document to add. for structured document, we support xlsx, csv,jsonl, for unstructured document, wu support txt, doc, docx, pdf, markdown, faq.xlsx, pptx".
|
|
127
|
+
you should judge the doc_type based on user's given url and judge if we support this doc type. if supported, assign this parameter.
|
|
128
|
+
url: the url of the document to add. user should give a valid url, we will add the doc to the collection.
|
|
129
|
+
|
|
130
|
+
Returns:
|
|
131
|
+
collection_name: the name of the knowledge base collection.
|
|
132
|
+
doc_id: the doc_id of document user added to collection.
|
|
133
|
+
|
|
134
|
+
"""
|
|
135
|
+
try:
|
|
136
|
+
if not collection_name:
|
|
137
|
+
raise ValueError("Collection name cannot be empty.")
|
|
138
|
+
|
|
139
|
+
request_params = {
|
|
140
|
+
"collection_name": collection_name,
|
|
141
|
+
"project": config.project,
|
|
142
|
+
"add_type": add_type,
|
|
143
|
+
"doc_id": doc_id,
|
|
144
|
+
"doc_name": doc_name,
|
|
145
|
+
"doc_type": doc_type,
|
|
146
|
+
"url": url,
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
result = await _request_knowledgebase(doc_add_path, request_params)
|
|
150
|
+
if result['code'] != 0:
|
|
151
|
+
logger.error(f"Error in add_doc: {result['message']}")
|
|
152
|
+
raise ToolError(result['message'])
|
|
153
|
+
|
|
154
|
+
doc_add_data = result['data']
|
|
155
|
+
if not doc_add_data:
|
|
156
|
+
raise ValueError(f"doc {doc_id} has no data.")
|
|
157
|
+
|
|
158
|
+
return AddDocumentResult(collection_name=collection_name, doc_id=doc_id)
|
|
159
|
+
|
|
160
|
+
except ToolError:
|
|
161
|
+
raise
|
|
162
|
+
except Exception as e:
|
|
163
|
+
logger.error(f"Error in add_doc: {str(e)}")
|
|
164
|
+
raise ToolError(str(e)) from e
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
@mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True))
|
|
168
|
+
async def get_doc(
|
|
169
|
+
collection_name: str,
|
|
170
|
+
doc_id: str,
|
|
171
|
+
) -> DocumentInfo:
|
|
172
|
+
"""
|
|
173
|
+
Get information about a document from your collection.
|
|
174
|
+
This tool allows you to get information about a document from your project by collection_name and doc_id.
|
|
175
|
+
Args:
|
|
176
|
+
collection_name: the name of the knowledge base collection to get document from.
|
|
177
|
+
doc_id: the doc_id of document user want to get information.
|
|
178
|
+
Returns:
|
|
179
|
+
collection_name: the name of the knowledge base collection.
|
|
180
|
+
doc_id: the doc_id of document user added to collection.
|
|
181
|
+
doc_name: the name of the document.
|
|
182
|
+
doc_type: the type of the document.
|
|
183
|
+
url: the url of the document.
|
|
184
|
+
add_type: the type how to add document.
|
|
185
|
+
create_time: the time when document added to collection.
|
|
186
|
+
update_time: the time when document updated.
|
|
187
|
+
point_num: The number of points extracted from the document.
|
|
188
|
+
status: the status of the document. the status struct has two fields:
|
|
189
|
+
- process_status: The processing status of the document.
|
|
190
|
+
0 means the processing is completed,
|
|
191
|
+
1 means the processing failed,
|
|
192
|
+
2 or 3 means it is in queue,
|
|
193
|
+
5 means it is being deleted,
|
|
194
|
+
and 6 means it is processing.
|
|
195
|
+
- failed_code: the status message of the document.
|
|
196
|
+
"""
|
|
197
|
+
|
|
198
|
+
try:
|
|
199
|
+
if not collection_name:
|
|
200
|
+
raise ValueError("Collection name cannot be empty.")
|
|
201
|
+
|
|
202
|
+
request_params = {
|
|
203
|
+
"collection_name": collection_name,
|
|
204
|
+
"project": config.project,
|
|
205
|
+
"doc_id": doc_id,
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
result = await _request_knowledgebase(doc_info_path, request_params)
|
|
209
|
+
if result['code'] != 0:
|
|
210
|
+
logger.error(f"Error in get_doc: {result['message']}")
|
|
211
|
+
raise ToolError(result['message'])
|
|
212
|
+
|
|
213
|
+
doc_info_data = result['data']
|
|
214
|
+
if not doc_info_data:
|
|
215
|
+
raise ValueError(f"doc {doc_id} not found.")
|
|
216
|
+
|
|
217
|
+
return DocumentInfo.model_validate(doc_info_data)
|
|
218
|
+
|
|
219
|
+
except ToolError:
|
|
220
|
+
raise
|
|
221
|
+
except Exception as e:
|
|
222
|
+
logger.error(f"Error in get_doc: {str(e)}")
|
|
223
|
+
raise ToolError(str(e)) from e
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
@mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True))
|
|
227
|
+
async def get_collection(
|
|
228
|
+
collection_name: str,
|
|
229
|
+
) -> CollectionInfoResult:
|
|
230
|
+
"""
|
|
231
|
+
Get information about a collection from your project.
|
|
232
|
+
This tool allows you to get information about a collection from your project by collection_name.
|
|
233
|
+
Args:
|
|
234
|
+
collection_name: the name of the knowledge base collection to get info for.
|
|
235
|
+
|
|
236
|
+
Returns:
|
|
237
|
+
collection_name: the name of the knowledge base collection.
|
|
238
|
+
description: the description of the knowledge base collection.
|
|
239
|
+
status: the status of the knowledge base collection.
|
|
240
|
+
status:
|
|
241
|
+
-1: To be built
|
|
242
|
+
0: Building
|
|
243
|
+
1: Build completed
|
|
244
|
+
2: Build failed
|
|
245
|
+
3: Changing
|
|
246
|
+
|
|
247
|
+
"""
|
|
248
|
+
|
|
249
|
+
try:
|
|
250
|
+
if not collection_name:
|
|
251
|
+
raise ValueError("Collection name cannot be empty.")
|
|
252
|
+
|
|
253
|
+
request_params = {
|
|
254
|
+
"name": collection_name,
|
|
255
|
+
"project": config.project,
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
result = await _request_knowledgebase(get_collections_path, request_params)
|
|
259
|
+
if result['code'] != 0:
|
|
260
|
+
logger.error(f"Error in search_knowledge: {result['message']}")
|
|
261
|
+
raise ToolError(result['message'])
|
|
262
|
+
|
|
263
|
+
collection_info = result['data']
|
|
264
|
+
if not collection_info:
|
|
265
|
+
raise ValueError(f"Collection {collection_name} not found.")
|
|
266
|
+
|
|
267
|
+
return CollectionInfoResult(
|
|
268
|
+
collection_name=collection_info["collection_name"],
|
|
269
|
+
description=collection_info["description"],
|
|
270
|
+
status=collection_info["pipeline_list"][0]["index_list"][0]["status"],
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
except ToolError:
|
|
274
|
+
raise
|
|
275
|
+
except Exception as e:
|
|
276
|
+
logger.error(f"Error in get_collection: {str(e)}")
|
|
277
|
+
raise ToolError(str(e)) from e
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
@mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True))
|
|
281
|
+
async def list_collections() -> ListCollectionsResult:
|
|
282
|
+
"""
|
|
283
|
+
List all collections of the globally configured project from the Viking Knowledgebase service.
|
|
284
|
+
This tool allows you to list all collections in the Viking Knowledgebase service.
|
|
285
|
+
|
|
286
|
+
Returns:
|
|
287
|
+
A list of collections in the project.
|
|
288
|
+
collection_name: the name of the knowledge base collection.
|
|
289
|
+
description: the description of the knowledge base collection.
|
|
290
|
+
|
|
291
|
+
"""
|
|
292
|
+
|
|
293
|
+
try:
|
|
294
|
+
request_params = {
|
|
295
|
+
"project": config.project,
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
result = await _request_knowledgebase(list_collections_path, request_params)
|
|
299
|
+
if result['code'] != 0:
|
|
300
|
+
logger.error(f"Error in list_collections: {result['message']}")
|
|
301
|
+
raise ToolError(result['message'])
|
|
302
|
+
|
|
303
|
+
collections = result['data']['collection_list']
|
|
304
|
+
|
|
305
|
+
collection_list = []
|
|
306
|
+
|
|
307
|
+
for collection in collections:
|
|
308
|
+
collection_list.append(
|
|
309
|
+
CollectionSummary(
|
|
310
|
+
collection_name=collection["collection_name"],
|
|
311
|
+
description=collection["description"],
|
|
312
|
+
)
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
return ListCollectionsResult(collection_list=collection_list)
|
|
316
|
+
|
|
317
|
+
except ToolError:
|
|
318
|
+
raise
|
|
319
|
+
except Exception as e:
|
|
320
|
+
logger.error(f"Error in list_collections: {str(e)}")
|
|
321
|
+
raise ToolError(str(e)) from e
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
@mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True))
|
|
325
|
+
async def search_knowledge(
|
|
326
|
+
query: str,
|
|
327
|
+
collection_name: str,
|
|
328
|
+
limit: Annotated[int, Field(ge=1, le=100)] = 3,
|
|
329
|
+
doc_filter: Optional[DocFilter] = None,
|
|
330
|
+
) -> SearchKnowledgeResult:
|
|
331
|
+
"""Search knowledge from the Viking Knowledgebase service And return Top limit related chunks of your query.
|
|
332
|
+
This tool allows you to search knowledge in provided collection based on the given query.
|
|
333
|
+
|
|
334
|
+
Args:
|
|
335
|
+
query: the search query string.
|
|
336
|
+
limit: the maximum number of results to return (default: 3).
|
|
337
|
+
collection_name: the name of the knowledge base collection to search for.
|
|
338
|
+
doc_filter: the filter is used to filter search results(default: None), which is structured as a JSON object with
|
|
339
|
+
the following key components:
|
|
340
|
+
- 'op': (string, required) specifies the query operator that defines the filtering logic. Valid values are
|
|
341
|
+
'must' and 'must_not', 'must' means results must satisfy the condition (inclusion filter),'must_not' means
|
|
342
|
+
results must not satisfy the condition (exclusion filter).
|
|
343
|
+
- 'field': (string, required) indicates the specific document field to apply the filter on (e.g., "doc_id").
|
|
344
|
+
- 'conds': (array, required) contains the concrete values used for filtering. The data type
|
|
345
|
+
of elements in the array depends on the field.
|
|
346
|
+
|
|
347
|
+
Returns:
|
|
348
|
+
A list of search results.
|
|
349
|
+
id: the id of the knowledge base chunk.
|
|
350
|
+
content: the content of the knowledge base chunk.
|
|
351
|
+
doc_id: the id of the document containing the chunk, when available.
|
|
352
|
+
doc_name: the name of the document containing the chunk, when available.
|
|
353
|
+
"""
|
|
354
|
+
|
|
355
|
+
try:
|
|
356
|
+
if not collection_name:
|
|
357
|
+
raise ValueError("Collection name cannot be empty.")
|
|
358
|
+
|
|
359
|
+
request_params = {
|
|
360
|
+
"query": query,
|
|
361
|
+
"limit": limit,
|
|
362
|
+
"name": collection_name,
|
|
363
|
+
"project": config.project,
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
if doc_filter:
|
|
367
|
+
request_params['query_param'] = {
|
|
368
|
+
"doc_filter": doc_filter.model_dump(mode="json"),
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
result = await _request_knowledgebase(search_knowledge_path, request_params)
|
|
372
|
+
if result['code'] != 0:
|
|
373
|
+
logger.error(f"Error in search_knowledge: {result['message']}")
|
|
374
|
+
raise ToolError(result['message'])
|
|
375
|
+
|
|
376
|
+
chunks = result['data'].get('result_list', [])
|
|
377
|
+
|
|
378
|
+
search_result = []
|
|
379
|
+
|
|
380
|
+
for chunk in chunks:
|
|
381
|
+
raw_doc_info = chunk.get("doc_info")
|
|
382
|
+
doc_info = raw_doc_info if isinstance(raw_doc_info, dict) else {}
|
|
383
|
+
search_result.append(
|
|
384
|
+
SearchChunk(
|
|
385
|
+
id=chunk["id"],
|
|
386
|
+
content=chunk["content"],
|
|
387
|
+
doc_id=doc_info.get("doc_id"),
|
|
388
|
+
doc_name=doc_info.get("doc_name"),
|
|
389
|
+
)
|
|
390
|
+
)
|
|
391
|
+
|
|
392
|
+
return SearchKnowledgeResult(result_list=search_result)
|
|
393
|
+
except ToolError:
|
|
394
|
+
raise
|
|
395
|
+
except Exception as e:
|
|
396
|
+
logger.error(f"Error in search_knowledge: {str(e)}")
|
|
397
|
+
raise ToolError(str(e)) from e
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def main():
|
|
401
|
+
"""Main entry point for the Knowledgebase MCP server."""
|
|
402
|
+
parser = argparse.ArgumentParser(description='Run the Viking Knowledgebase MCP Server')
|
|
403
|
+
parser.add_argument(
|
|
404
|
+
"--transport",
|
|
405
|
+
"-t",
|
|
406
|
+
choices=["stdio", "streamable-http"],
|
|
407
|
+
default="stdio",
|
|
408
|
+
help="Transport protocol to use (stdio or streamable-http)",
|
|
409
|
+
)
|
|
410
|
+
args = parser.parse_args()
|
|
411
|
+
logger.info(f"Starting Knowledgebase MCP Server with {args.transport} transport")
|
|
412
|
+
|
|
413
|
+
try:
|
|
414
|
+
mcp.run(transport=args.transport, **_transport_options(args.transport))
|
|
415
|
+
except Exception as e:
|
|
416
|
+
logger.error(f"Error starting Knowledgebase MCP Server: {str(e)}")
|
|
417
|
+
raise
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
if __name__ == "__main__":
|
|
421
|
+
main()
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: mcp-server-knowledgebase
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: MCP server for Viking Knowledge Base Service
|
|
5
|
+
License: MIT
|
|
6
|
+
Requires-Python: >=3.10
|
|
7
|
+
Requires-Dist: aiohttp>=3.11.14
|
|
8
|
+
Requires-Dist: mcp[cli]<3,>=2.1.1
|
|
9
|
+
Requires-Dist: volcengine>=1.0.171
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
|
|
12
|
+
# Viking Knowledge Base MCP Server
|
|
13
|
+
|
|
14
|
+
This MCP server provides a tool to interact with the VolcEngine Viking Knowledge Base Service, allowing you to search and retrieve knowledge from your collections, meanwhile,
|
|
15
|
+
allowing you to add doc to your collections and get doc processing info by doc_id.
|
|
16
|
+
|
|
17
|
+
## Features
|
|
18
|
+
|
|
19
|
+
- Search knowledge based on queries with customizable parameters
|
|
20
|
+
|
|
21
|
+
## Setup
|
|
22
|
+
|
|
23
|
+
### Prerequisites
|
|
24
|
+
|
|
25
|
+
- Python 3.10 or higher
|
|
26
|
+
- A Viking Knowledge Base API key or VolcEngine AK/SK credentials
|
|
27
|
+
|
|
28
|
+
### Installation
|
|
29
|
+
|
|
30
|
+
1. Install the package:
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
pip install -e .
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Or with uv (recommended):
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
uv pip install -e .
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
### Configuration
|
|
43
|
+
|
|
44
|
+
The server requires at least one authentication method:
|
|
45
|
+
|
|
46
|
+
- API key: set `VIKING_API_KEY`. Requests use
|
|
47
|
+
`Authorization: Bearer <VIKING_API_KEY>`.
|
|
48
|
+
- AK/SK: set both `VOLCENGINE_ACCESS_KEY` and `VOLCENGINE_SECRET_KEY`.
|
|
49
|
+
Requests use VolcEngine SignerV4 authentication.
|
|
50
|
+
|
|
51
|
+
When both methods are configured, `VIKING_API_KEY` takes precedence and AK/SK
|
|
52
|
+
is ignored. When no API key is configured, AK and SK must be provided together.
|
|
53
|
+
The server rejects configurations with no usable authentication method.
|
|
54
|
+
|
|
55
|
+
Optional environment variables:
|
|
56
|
+
- `KNOWLEDGE_BASE_PROJECT`: Viking Knowledge Base project name (default: `default`)
|
|
57
|
+
- `KNOWLEDGE_BASE_REGION`: Viking Knowledge Base region (default: `cn-north-1`)
|
|
58
|
+
- `MCP_SERVER_HOST`: Streamable HTTP bind host (default: `127.0.0.1`)
|
|
59
|
+
- `MCP_SERVER_PORT`: Streamable HTTP port; falls back to `PORT` (default: `8000`)
|
|
60
|
+
- `STREAMABLE_HTTP_PATH`: Streamable HTTP endpoint path (default: `/mcp`)
|
|
61
|
+
- `KNOWLEDGE_BASE_TIMEOUT`: Upstream request timeout in seconds (default: `30`)
|
|
62
|
+
|
|
63
|
+
## Usage
|
|
64
|
+
|
|
65
|
+
### Running the Server
|
|
66
|
+
|
|
67
|
+
The server supports stdio for local integrations and stateless Streamable HTTP
|
|
68
|
+
for remote deployments:
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
python -m mcp_server_knowledgebase.server --transport stdio
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Or:
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
python -m mcp_server_knowledgebase.server --transport streamable-http
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
The Streamable HTTP endpoint is `http://127.0.0.1:8000/mcp` by default.
|
|
81
|
+
Set `MCP_SERVER_HOST=0.0.0.0` when running behind a trusted gateway.
|
|
82
|
+
|
|
83
|
+
### MCP protocol compatibility
|
|
84
|
+
|
|
85
|
+
This server uses MCP Python SDK 2.x and speaks protocol revision `2026-07-28`.
|
|
86
|
+
Modern clients use the stateless per-request protocol and `server/discover`;
|
|
87
|
+
the same process also supports older handshake-based clients automatically.
|
|
88
|
+
Legacy HTTP+SSE is intentionally not exposed because it is deprecated by the
|
|
89
|
+
`2026-07-28` specification.
|
|
90
|
+
|
|
91
|
+
The HTTP endpoint does not turn the configured API key or VolcEngine AK/SK into
|
|
92
|
+
client authentication. Protect remote deployments with an authentication
|
|
93
|
+
gateway or MCP-compatible OAuth, and never expose the service credentials to
|
|
94
|
+
callers.
|
|
95
|
+
|
|
96
|
+
### Available Tools
|
|
97
|
+
|
|
98
|
+
#### add_doc
|
|
99
|
+
|
|
100
|
+
Add a document to a collection in your project.
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
add_doc(
|
|
104
|
+
collection_name="collection_name",
|
|
105
|
+
add_type="url",
|
|
106
|
+
doc_id="mcp_server_auto_gen_doc_id_xxxxxxx",
|
|
107
|
+
doc_name="doc_xxxx",
|
|
108
|
+
doc_type="pdf",
|
|
109
|
+
url="http://xxxxx.pdf"
|
|
110
|
+
)
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
Parameters:
|
|
114
|
+
- `collection_name` (required): the name of the collection you want to add document .
|
|
115
|
+
- `add_type` (required): the type of the document to add. so far only support "url" now.
|
|
116
|
+
- `doc_id` (required): you should generate a unique doc_id based on user's given url and timestamp, the doc_id can only use English letters, numbers, and underscores , and must start with an English letter. It cannot be empty. Length requirement: [1, 128], you can use a format like "mcp_server_auto_gen_doc_id_xxxxxxx".
|
|
117
|
+
- `doc_name` (required): the name of the document to add. You can generate a unique doc_name based on the user-provided URL and timestamp. The length of doc_name must be between 1 and 256; for example, "mcp_server_auto_gen_doc_name_xxxxxxx".
|
|
118
|
+
- `doc_type` (required): the type of the document to add. for structured document, we support xlsx, csv,jsonl, for unstructured document, wu support txt, doc, docx, pdf, markdown, faq.xlsx, pptx". you should judge the doc_type based on user's given url and judge if we support this doc type. if supported, assign this parameter.
|
|
119
|
+
- `url` (required): the url of the document to add. user should give a valid url, we will add the doc to the collection.
|
|
120
|
+
|
|
121
|
+
#### get_doc
|
|
122
|
+
|
|
123
|
+
Get information about document by collection_name and doc_id .
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
get_doc(
|
|
127
|
+
collection_name="collection_name",
|
|
128
|
+
doc_id="mcp_server_auto_gen_doc_id_xxxxxxx",
|
|
129
|
+
)
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
Parameters:
|
|
133
|
+
- `collection_name` (required): the name of the collection you want to get information .
|
|
134
|
+
- `doc_id` (required): the doc_id of document user want to get information .
|
|
135
|
+
|
|
136
|
+
#### get_collection
|
|
137
|
+
|
|
138
|
+
Get information about a viking knowledge base collection from your project .
|
|
139
|
+
|
|
140
|
+
```python
|
|
141
|
+
get_collection(
|
|
142
|
+
collection_name="collection_name",
|
|
143
|
+
)
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
Parameters:
|
|
147
|
+
- `collection_name` (required): the name of the collection you want to get information .
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
#### list_collections
|
|
151
|
+
|
|
152
|
+
List all knowledge base collections of the globally configured project .
|
|
153
|
+
|
|
154
|
+
```python
|
|
155
|
+
list_collections(
|
|
156
|
+
)
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
#### search_knowledge
|
|
161
|
+
|
|
162
|
+
Search for knowledge in the configured collection based on a query.
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
search_knowledge(
|
|
166
|
+
query="How to reset my password?",
|
|
167
|
+
limit=3,
|
|
168
|
+
collection_name="collection_name",
|
|
169
|
+
doc_filter=None,
|
|
170
|
+
)
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
Parameters:
|
|
174
|
+
- `query` (required): The search query string
|
|
175
|
+
- `limit` (optional): Maximum number of results to return, from 1 to 100 (default: 3)
|
|
176
|
+
- `collection_name` (required): Knowledge Base collection name to search
|
|
177
|
+
- `doc_filter` (optional): the filter is used to filter search results(default: None), which is structured as a JSON object with the following key components:
|
|
178
|
+
- `op` (string, required): specifies the query operator that defines the filtering logic. Valid values are 'must' and 'must_not', 'must' means results must satisfy the condition (inclusion filter),'must_not' means results must not satisfy the condition (exclusion filter).
|
|
179
|
+
- `field` (string, required): indicates the specific document field to apply the filter on (e.g., "doc_id").
|
|
180
|
+
- `conds` (array, required): contains the concrete values used for filtering. The data type of elements in the array depends on the field.
|
|
181
|
+
|
|
182
|
+
Each result contains the chunk `id` and `content`, plus the source document's
|
|
183
|
+
`doc_id` and `doc_name`. The metadata fields are `null` when Viking does not
|
|
184
|
+
provide them. A non-null `doc_id` can be passed directly to `get_doc`.
|
|
185
|
+
|
|
186
|
+
## MCP Integration
|
|
187
|
+
|
|
188
|
+
To add this server to your MCP configuration, add the following to your MCP settings file:
|
|
189
|
+
|
|
190
|
+
```json
|
|
191
|
+
{
|
|
192
|
+
"mcpServers": {
|
|
193
|
+
"knowledgebase": {
|
|
194
|
+
"command": "uvx",
|
|
195
|
+
"args": [
|
|
196
|
+
"--from",
|
|
197
|
+
"mcp-server-knowledgebase>=0.2.0",
|
|
198
|
+
"mcp-server-knowledgebase"
|
|
199
|
+
],
|
|
200
|
+
"env": {
|
|
201
|
+
"VIKING_API_KEY": "your-viking-api-key",
|
|
202
|
+
"KNOWLEDGE_BASE_PROJECT": "your-project-name",
|
|
203
|
+
"KNOWLEDGE_BASE_REGION": "your-region"
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
You may alternatively or additionally configure both
|
|
211
|
+
`VOLCENGINE_ACCESS_KEY` and `VOLCENGINE_SECRET_KEY`. If all three variables are
|
|
212
|
+
set, `VIKING_API_KEY` takes precedence.
|
|
213
|
+
|
|
214
|
+
## Troubleshooting
|
|
215
|
+
|
|
216
|
+
### Common Issues
|
|
217
|
+
|
|
218
|
+
1. **Authentication Errors**
|
|
219
|
+
- Verify your API key or AK/SK credentials are correct
|
|
220
|
+
- Ensure at least one authentication method is configured
|
|
221
|
+
- Check that you have the necessary permissions for the collection
|
|
222
|
+
|
|
223
|
+
2. **Connection Timeouts**
|
|
224
|
+
- Check your network connection to the VolcEngine API
|
|
225
|
+
- Verify the host configuration is correct
|
|
226
|
+
|
|
227
|
+
3. **Empty Results**
|
|
228
|
+
- Verify the collection name is correct
|
|
229
|
+
- Try broadening your search query
|
|
230
|
+
|
|
231
|
+
### Logging
|
|
232
|
+
|
|
233
|
+
The server uses Python's logging module with INFO level by default. You can see detailed logs in the console when running the server.
|
|
234
|
+
|
|
235
|
+
## Contributing
|
|
236
|
+
|
|
237
|
+
Contributions to improve the Viking Knowledge Base MCP Server are welcome. Please follow these steps:
|
|
238
|
+
|
|
239
|
+
1. Fork the repository
|
|
240
|
+
2. Create a feature branch
|
|
241
|
+
3. Make your changes
|
|
242
|
+
4. Submit a pull request
|
|
243
|
+
|
|
244
|
+
Please ensure your code follows the project's coding standards and includes appropriate tests.
|
|
245
|
+
|
|
246
|
+
## License
|
|
247
|
+
|
|
248
|
+
volcengine/mcp-server is licensed under the [MIT License](https://github.com/volcengine/mcp-server/blob/main/LICENSE).
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
mcp_server_knowledgebase/__init__.py,sha256=aUck0pBUWsafYyf6Ch2F7iYQaxLQrH198HQbXPDp8ms,25
|
|
2
|
+
mcp_server_knowledgebase/config.py,sha256=jQdEtU3ho37T88w-IuFCaYK4OP9g_rLN7hiwpo-0S7M,1580
|
|
3
|
+
mcp_server_knowledgebase/models.py,sha256=Gly__XmaYyU1bjzPqm9hU9wb4XXQ97iywSWxzAvT6Ns,1541
|
|
4
|
+
mcp_server_knowledgebase/server.py,sha256=73D_84J8LebZK8BbnhYBatqYQSOfR2VXKyjjsRSJRME,15546
|
|
5
|
+
mcp_server_knowledgebase/common/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
6
|
+
mcp_server_knowledgebase/common/auth.py,sha256=SV2fVjtto321yKPYTCK56D_Q6MSqyuN8ey-Mee1PzNg,1710
|
|
7
|
+
mcp_server_knowledgebase-0.2.0.dist-info/METADATA,sha256=sYuZbYuSIsuYji8QvhTtu3fhs_YXhr13UvB0xCVgRLU,8408
|
|
8
|
+
mcp_server_knowledgebase-0.2.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
9
|
+
mcp_server_knowledgebase-0.2.0.dist-info/entry_points.txt,sha256=Ou4PBntYVXnKxPxzpNWCyVEPaJySuEIY24xogVOeGcw,82
|
|
10
|
+
mcp_server_knowledgebase-0.2.0.dist-info/RECORD,,
|