mcp-server-knowledgebase 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1 @@
1
+ # Package initialization
File without changes
@@ -0,0 +1,53 @@
1
+ import json
2
+
3
+ from volcengine.auth.SignerV4 import SignerV4
4
+ from volcengine.base.Request import Request
5
+ from volcengine.Credentials import Credentials
6
+ from mcp_server_knowledgebase.config import config
7
+
8
+
9
+ def prepare_request(
10
+ method, path, ak=None, sk=None, params=None, data=None, doseq=0, *, api_key=None
11
+ ):
12
+ ak = ak.strip() if isinstance(ak, str) else ak
13
+ sk = sk.strip() if isinstance(sk, str) else sk
14
+ api_key = api_key.strip() if isinstance(api_key, str) else api_key
15
+
16
+ if not api_key:
17
+ if bool(ak) != bool(sk):
18
+ raise ValueError("AK and SK must be configured together")
19
+ if not ak or not sk:
20
+ raise ValueError("Configure an authentication method: VIKING_API_KEY or AK/SK")
21
+
22
+ if params:
23
+ for key in params:
24
+ if (
25
+ type(params[key]) == int
26
+ or type(params[key]) == float
27
+ or type(params[key]) == bool
28
+ ):
29
+ params[key] = str(params[key])
30
+ elif type(params[key]) == list:
31
+ if not doseq:
32
+ params[key] = ",".join(params[key])
33
+ r = Request()
34
+ r.set_shema("https")
35
+ r.set_method(method)
36
+ r.set_connection_timeout(10)
37
+ r.set_socket_timeout(10)
38
+ mheaders = {
39
+ "Accept": "application/json",
40
+ "Content-Type": "application/json",
41
+ }
42
+ if api_key:
43
+ mheaders["Authorization"] = f"Bearer {api_key}"
44
+ r.set_headers(mheaders)
45
+ if params:
46
+ r.set_query(params)
47
+ r.set_path(path)
48
+ if data is not None:
49
+ r.set_body(json.dumps(data))
50
+ if not api_key:
51
+ credentials = Credentials(ak, sk, "air", config.region)
52
+ SignerV4.sign(r, credentials)
53
+ return r
@@ -0,0 +1,57 @@
1
+ import logging
2
+ import os
3
+ from dataclasses import dataclass
4
+ from typing import Optional
5
+
6
+ logger = logging.getLogger(__name__)
7
+
8
+
9
+ @dataclass
10
+ class KnowledgeBaseConfig:
11
+ """Configuration for Viking Knowledge Base MCP Server."""
12
+
13
+ ak: Optional[str] = None
14
+ sk: Optional[str] = None
15
+ project: Optional[str] = None
16
+ region: str = "cn-north-1"
17
+ api_key: Optional[str] = None
18
+
19
+
20
+ def _credential_from_env(name: str) -> Optional[str]:
21
+ value = os.environ.get(name)
22
+ if value is None:
23
+ return None
24
+ return value.strip() or None
25
+
26
+
27
+ def load_config() -> KnowledgeBaseConfig:
28
+ """Load configuration from environment variables."""
29
+ ak = _credential_from_env("VOLCENGINE_ACCESS_KEY")
30
+ sk = _credential_from_env("VOLCENGINE_SECRET_KEY")
31
+ api_key = _credential_from_env("VIKING_API_KEY")
32
+
33
+ if not api_key and bool(ak) != bool(sk):
34
+ error_msg = (
35
+ "VOLCENGINE_ACCESS_KEY and VOLCENGINE_SECRET_KEY must be configured together"
36
+ )
37
+ logger.error(error_msg)
38
+ raise ValueError(error_msg)
39
+
40
+ if not api_key and not (ak and sk):
41
+ error_msg = (
42
+ "Configure an authentication method: VIKING_API_KEY or "
43
+ "VOLCENGINE_ACCESS_KEY with VOLCENGINE_SECRET_KEY"
44
+ )
45
+ logger.error(error_msg)
46
+ raise ValueError(error_msg)
47
+
48
+ return KnowledgeBaseConfig(
49
+ ak=ak,
50
+ sk=sk,
51
+ api_key=api_key,
52
+ project=os.environ.get("KNOWLEDGE_BASE_PROJECT", "default"),
53
+ region=os.environ.get("KNOWLEDGE_BASE_REGION", "cn-north-1"),
54
+ )
55
+
56
+
57
+ config = load_config()
@@ -0,0 +1,66 @@
1
+ from typing import Any, Literal, Optional
2
+
3
+ from pydantic import BaseModel, ConfigDict, Field
4
+
5
+
6
+ class DocFilter(BaseModel):
7
+ """Filter applied to Viking Knowledge Base search results."""
8
+
9
+ op: Literal["must", "must_not"]
10
+ field: str = Field(min_length=1)
11
+ conds: list[Any] = Field(min_length=1)
12
+
13
+
14
+ class AddDocumentResult(BaseModel):
15
+ collection_name: str
16
+ doc_id: str
17
+
18
+
19
+ class DocumentStatus(BaseModel):
20
+ model_config = ConfigDict(extra="allow")
21
+
22
+ process_status: Optional[int] = None
23
+ failed_code: Optional[str] = None
24
+
25
+
26
+ class DocumentInfo(BaseModel):
27
+ """Known document fields; extra upstream fields are preserved."""
28
+
29
+ model_config = ConfigDict(extra="allow")
30
+
31
+ collection_name: Optional[str] = None
32
+ doc_id: Optional[str] = None
33
+ doc_name: Optional[str] = None
34
+ doc_type: Optional[str] = None
35
+ url: Optional[str] = None
36
+ add_type: Optional[str] = None
37
+ create_time: Optional[int] = None
38
+ update_time: Optional[int] = None
39
+ point_num: Optional[int] = None
40
+ status: Optional[DocumentStatus] = None
41
+
42
+
43
+ class CollectionInfoResult(BaseModel):
44
+ collection_name: str
45
+ description: str
46
+ status: int
47
+
48
+
49
+ class CollectionSummary(BaseModel):
50
+ collection_name: str
51
+ description: str
52
+
53
+
54
+ class ListCollectionsResult(BaseModel):
55
+ collection_list: list[CollectionSummary]
56
+
57
+
58
+ class SearchChunk(BaseModel):
59
+ id: str
60
+ content: str
61
+ doc_id: Optional[str] = None
62
+ doc_name: Optional[str] = None
63
+
64
+
65
+ class SearchKnowledgeResult(BaseModel):
66
+ result_list: list[SearchChunk]
@@ -0,0 +1,421 @@
1
+ import argparse
2
+ import logging
3
+ import os
4
+ from typing import Annotated, Any, Dict, Literal, Optional
5
+
6
+ import aiohttp
7
+ from mcp.server import MCPServer
8
+ from mcp.server.caching import CacheHint
9
+ from mcp.server.mcpserver.exceptions import ToolError
10
+ from mcp.types import ToolAnnotations
11
+ from pydantic import Field
12
+
13
+ from mcp_server_knowledgebase.common.auth import prepare_request
14
+ from mcp_server_knowledgebase.config import config
15
+ from mcp_server_knowledgebase.models import (
16
+ AddDocumentResult,
17
+ CollectionInfoResult,
18
+ CollectionSummary,
19
+ DocFilter,
20
+ DocumentInfo,
21
+ ListCollectionsResult,
22
+ SearchChunk,
23
+ SearchKnowledgeResult,
24
+ )
25
+
26
+ logger = logging.getLogger(__name__)
27
+ logging.basicConfig(
28
+ level=logging.INFO, format="%(asctime)s - %(name)s - %(levelname)s - %(message)s"
29
+ )
30
+
31
+ # knowledge base domain
32
+ g_knowledge_base_domain = "api-knowledgebase.mlp.cn-beijing.volces.com"
33
+
34
+ # paths
35
+ search_knowledge_path = "/api/knowledge/collection/search_knowledge"
36
+ list_collections_path = "/api/knowledge/collection/list"
37
+ get_collections_path = "/api/knowledge/collection/info"
38
+ doc_add_path = "/api/knowledge/doc/add"
39
+ doc_info_path = "/api/knowledge/doc/info"
40
+
41
+ # Create MCP server
42
+ mcp = MCPServer(
43
+ "Knowledgebase MCP Server",
44
+ version="0.2.0",
45
+ cache_hints={"tools/list": CacheHint(ttl_ms=300_000, scope="public")},
46
+ )
47
+
48
+
49
+ def _transport_options(transport: str) -> Dict[str, Any]:
50
+ """Build transport-specific options accepted by MCP SDK v2."""
51
+ if transport == "stdio":
52
+ return {}
53
+ if transport != "streamable-http":
54
+ raise ValueError(f"Unsupported transport: {transport}")
55
+ return {
56
+ "host": os.getenv("MCP_SERVER_HOST", "127.0.0.1"),
57
+ "port": int(os.getenv("MCP_SERVER_PORT") or os.getenv("PORT", "8000")),
58
+ "streamable_http_path": os.getenv("STREAMABLE_HTTP_PATH", "/mcp"),
59
+ "stateless_http": True,
60
+ "json_response": True,
61
+ }
62
+
63
+
64
+ async def _request_knowledgebase(path: str, data: Dict[str, Any]) -> Dict[str, Any]:
65
+ """Send one signed request without blocking the MCP event loop."""
66
+ request = prepare_request(
67
+ method="POST",
68
+ path=path,
69
+ ak=config.ak,
70
+ sk=config.sk,
71
+ api_key=config.api_key,
72
+ data=data,
73
+ )
74
+ timeout = aiohttp.ClientTimeout(total=float(os.getenv("KNOWLEDGE_BASE_TIMEOUT", "30")))
75
+ async with aiohttp.ClientSession(timeout=timeout) as session:
76
+ async with session.request(
77
+ method=request.method,
78
+ url=f"https://{g_knowledge_base_domain}{request.path}",
79
+ headers=request.headers,
80
+ data=request.body,
81
+ ) as response:
82
+ response.raise_for_status()
83
+ return await response.json()
84
+
85
+
86
+ @mcp.tool(
87
+ annotations=ToolAnnotations(
88
+ readOnlyHint=False,
89
+ destructiveHint=False,
90
+ idempotentHint=False,
91
+ openWorldHint=True,
92
+ )
93
+ )
94
+ async def add_doc(
95
+ collection_name: Annotated[str, Field(min_length=1)],
96
+ add_type: Literal["url"],
97
+ doc_id: Annotated[
98
+ str,
99
+ Field(min_length=1, max_length=128, pattern=r"^[A-Za-z][A-Za-z0-9_]*$"),
100
+ ],
101
+ doc_name: Annotated[str, Field(min_length=1, max_length=256)],
102
+ doc_type: Literal[
103
+ "xlsx",
104
+ "csv",
105
+ "jsonl",
106
+ "txt",
107
+ "doc",
108
+ "docx",
109
+ "pdf",
110
+ "markdown",
111
+ "faq.xlsx",
112
+ "pptx",
113
+ ],
114
+ url: Annotated[str, Field(min_length=1)],
115
+ ) -> AddDocumentResult:
116
+ """
117
+ Add a document to a collection in your project.
118
+ This tool allows you to add a document to a collection in your project by collection_name.
119
+ Args:
120
+ collection_name: the name of the knowledge base collection to add document to.
121
+ add_type: the type of the document to add. so far only support "url" now. so you must assign this parameter to "url".
122
+ doc_id: you should generate a unique doc_id based on user's given url and timestamp, the doc_id can only use English letters, numbers, and underscores , and must start with an English letter. It cannot be empty.
123
+ Length requirement: [1, 128], you can use a format like "mcp_server_auto_gen_doc_id_xxxxxxx.
124
+ doc_name: the name of the document to add. you can1 generate a unique doc_name based on user given url and timestamp. the length of doc_name must between 1 and 256. you can use a
125
+ format like "mcp_server_auto_gen_doc_name_xxxxxxx.
126
+ doc_type: the type of the document to add. for structured document, we support xlsx, csv,jsonl, for unstructured document, wu support txt, doc, docx, pdf, markdown, faq.xlsx, pptx".
127
+ you should judge the doc_type based on user's given url and judge if we support this doc type. if supported, assign this parameter.
128
+ url: the url of the document to add. user should give a valid url, we will add the doc to the collection.
129
+
130
+ Returns:
131
+ collection_name: the name of the knowledge base collection.
132
+ doc_id: the doc_id of document user added to collection.
133
+
134
+ """
135
+ try:
136
+ if not collection_name:
137
+ raise ValueError("Collection name cannot be empty.")
138
+
139
+ request_params = {
140
+ "collection_name": collection_name,
141
+ "project": config.project,
142
+ "add_type": add_type,
143
+ "doc_id": doc_id,
144
+ "doc_name": doc_name,
145
+ "doc_type": doc_type,
146
+ "url": url,
147
+ }
148
+
149
+ result = await _request_knowledgebase(doc_add_path, request_params)
150
+ if result['code'] != 0:
151
+ logger.error(f"Error in add_doc: {result['message']}")
152
+ raise ToolError(result['message'])
153
+
154
+ doc_add_data = result['data']
155
+ if not doc_add_data:
156
+ raise ValueError(f"doc {doc_id} has no data.")
157
+
158
+ return AddDocumentResult(collection_name=collection_name, doc_id=doc_id)
159
+
160
+ except ToolError:
161
+ raise
162
+ except Exception as e:
163
+ logger.error(f"Error in add_doc: {str(e)}")
164
+ raise ToolError(str(e)) from e
165
+
166
+
167
+ @mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True))
168
+ async def get_doc(
169
+ collection_name: str,
170
+ doc_id: str,
171
+ ) -> DocumentInfo:
172
+ """
173
+ Get information about a document from your collection.
174
+ This tool allows you to get information about a document from your project by collection_name and doc_id.
175
+ Args:
176
+ collection_name: the name of the knowledge base collection to get document from.
177
+ doc_id: the doc_id of document user want to get information.
178
+ Returns:
179
+ collection_name: the name of the knowledge base collection.
180
+ doc_id: the doc_id of document user added to collection.
181
+ doc_name: the name of the document.
182
+ doc_type: the type of the document.
183
+ url: the url of the document.
184
+ add_type: the type how to add document.
185
+ create_time: the time when document added to collection.
186
+ update_time: the time when document updated.
187
+ point_num: The number of points extracted from the document.
188
+ status: the status of the document. the status struct has two fields:
189
+ - process_status: The processing status of the document.
190
+ 0 means the processing is completed,
191
+ 1 means the processing failed,
192
+ 2 or 3 means it is in queue,
193
+ 5 means it is being deleted,
194
+ and 6 means it is processing.
195
+ - failed_code: the status message of the document.
196
+ """
197
+
198
+ try:
199
+ if not collection_name:
200
+ raise ValueError("Collection name cannot be empty.")
201
+
202
+ request_params = {
203
+ "collection_name": collection_name,
204
+ "project": config.project,
205
+ "doc_id": doc_id,
206
+ }
207
+
208
+ result = await _request_knowledgebase(doc_info_path, request_params)
209
+ if result['code'] != 0:
210
+ logger.error(f"Error in get_doc: {result['message']}")
211
+ raise ToolError(result['message'])
212
+
213
+ doc_info_data = result['data']
214
+ if not doc_info_data:
215
+ raise ValueError(f"doc {doc_id} not found.")
216
+
217
+ return DocumentInfo.model_validate(doc_info_data)
218
+
219
+ except ToolError:
220
+ raise
221
+ except Exception as e:
222
+ logger.error(f"Error in get_doc: {str(e)}")
223
+ raise ToolError(str(e)) from e
224
+
225
+
226
+ @mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True))
227
+ async def get_collection(
228
+ collection_name: str,
229
+ ) -> CollectionInfoResult:
230
+ """
231
+ Get information about a collection from your project.
232
+ This tool allows you to get information about a collection from your project by collection_name.
233
+ Args:
234
+ collection_name: the name of the knowledge base collection to get info for.
235
+
236
+ Returns:
237
+ collection_name: the name of the knowledge base collection.
238
+ description: the description of the knowledge base collection.
239
+ status: the status of the knowledge base collection.
240
+ status:
241
+ -1: To be built
242
+ 0: Building
243
+ 1: Build completed
244
+ 2: Build failed
245
+ 3: Changing
246
+
247
+ """
248
+
249
+ try:
250
+ if not collection_name:
251
+ raise ValueError("Collection name cannot be empty.")
252
+
253
+ request_params = {
254
+ "name": collection_name,
255
+ "project": config.project,
256
+ }
257
+
258
+ result = await _request_knowledgebase(get_collections_path, request_params)
259
+ if result['code'] != 0:
260
+ logger.error(f"Error in search_knowledge: {result['message']}")
261
+ raise ToolError(result['message'])
262
+
263
+ collection_info = result['data']
264
+ if not collection_info:
265
+ raise ValueError(f"Collection {collection_name} not found.")
266
+
267
+ return CollectionInfoResult(
268
+ collection_name=collection_info["collection_name"],
269
+ description=collection_info["description"],
270
+ status=collection_info["pipeline_list"][0]["index_list"][0]["status"],
271
+ )
272
+
273
+ except ToolError:
274
+ raise
275
+ except Exception as e:
276
+ logger.error(f"Error in get_collection: {str(e)}")
277
+ raise ToolError(str(e)) from e
278
+
279
+
280
+ @mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True))
281
+ async def list_collections() -> ListCollectionsResult:
282
+ """
283
+ List all collections of the globally configured project from the Viking Knowledgebase service.
284
+ This tool allows you to list all collections in the Viking Knowledgebase service.
285
+
286
+ Returns:
287
+ A list of collections in the project.
288
+ collection_name: the name of the knowledge base collection.
289
+ description: the description of the knowledge base collection.
290
+
291
+ """
292
+
293
+ try:
294
+ request_params = {
295
+ "project": config.project,
296
+ }
297
+
298
+ result = await _request_knowledgebase(list_collections_path, request_params)
299
+ if result['code'] != 0:
300
+ logger.error(f"Error in list_collections: {result['message']}")
301
+ raise ToolError(result['message'])
302
+
303
+ collections = result['data']['collection_list']
304
+
305
+ collection_list = []
306
+
307
+ for collection in collections:
308
+ collection_list.append(
309
+ CollectionSummary(
310
+ collection_name=collection["collection_name"],
311
+ description=collection["description"],
312
+ )
313
+ )
314
+
315
+ return ListCollectionsResult(collection_list=collection_list)
316
+
317
+ except ToolError:
318
+ raise
319
+ except Exception as e:
320
+ logger.error(f"Error in list_collections: {str(e)}")
321
+ raise ToolError(str(e)) from e
322
+
323
+
324
+ @mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True))
325
+ async def search_knowledge(
326
+ query: str,
327
+ collection_name: str,
328
+ limit: Annotated[int, Field(ge=1, le=100)] = 3,
329
+ doc_filter: Optional[DocFilter] = None,
330
+ ) -> SearchKnowledgeResult:
331
+ """Search knowledge from the Viking Knowledgebase service And return Top limit related chunks of your query.
332
+ This tool allows you to search knowledge in provided collection based on the given query.
333
+
334
+ Args:
335
+ query: the search query string.
336
+ limit: the maximum number of results to return (default: 3).
337
+ collection_name: the name of the knowledge base collection to search for.
338
+ doc_filter: the filter is used to filter search results(default: None), which is structured as a JSON object with
339
+ the following key components:
340
+ - 'op': (string, required) specifies the query operator that defines the filtering logic. Valid values are
341
+ 'must' and 'must_not', 'must' means results must satisfy the condition (inclusion filter),'must_not' means
342
+ results must not satisfy the condition (exclusion filter).
343
+ - 'field': (string, required) indicates the specific document field to apply the filter on (e.g., "doc_id").
344
+ - 'conds': (array, required) contains the concrete values used for filtering. The data type
345
+ of elements in the array depends on the field.
346
+
347
+ Returns:
348
+ A list of search results.
349
+ id: the id of the knowledge base chunk.
350
+ content: the content of the knowledge base chunk.
351
+ doc_id: the id of the document containing the chunk, when available.
352
+ doc_name: the name of the document containing the chunk, when available.
353
+ """
354
+
355
+ try:
356
+ if not collection_name:
357
+ raise ValueError("Collection name cannot be empty.")
358
+
359
+ request_params = {
360
+ "query": query,
361
+ "limit": limit,
362
+ "name": collection_name,
363
+ "project": config.project,
364
+ }
365
+
366
+ if doc_filter:
367
+ request_params['query_param'] = {
368
+ "doc_filter": doc_filter.model_dump(mode="json"),
369
+ }
370
+
371
+ result = await _request_knowledgebase(search_knowledge_path, request_params)
372
+ if result['code'] != 0:
373
+ logger.error(f"Error in search_knowledge: {result['message']}")
374
+ raise ToolError(result['message'])
375
+
376
+ chunks = result['data'].get('result_list', [])
377
+
378
+ search_result = []
379
+
380
+ for chunk in chunks:
381
+ raw_doc_info = chunk.get("doc_info")
382
+ doc_info = raw_doc_info if isinstance(raw_doc_info, dict) else {}
383
+ search_result.append(
384
+ SearchChunk(
385
+ id=chunk["id"],
386
+ content=chunk["content"],
387
+ doc_id=doc_info.get("doc_id"),
388
+ doc_name=doc_info.get("doc_name"),
389
+ )
390
+ )
391
+
392
+ return SearchKnowledgeResult(result_list=search_result)
393
+ except ToolError:
394
+ raise
395
+ except Exception as e:
396
+ logger.error(f"Error in search_knowledge: {str(e)}")
397
+ raise ToolError(str(e)) from e
398
+
399
+
400
+ def main():
401
+ """Main entry point for the Knowledgebase MCP server."""
402
+ parser = argparse.ArgumentParser(description='Run the Viking Knowledgebase MCP Server')
403
+ parser.add_argument(
404
+ "--transport",
405
+ "-t",
406
+ choices=["stdio", "streamable-http"],
407
+ default="stdio",
408
+ help="Transport protocol to use (stdio or streamable-http)",
409
+ )
410
+ args = parser.parse_args()
411
+ logger.info(f"Starting Knowledgebase MCP Server with {args.transport} transport")
412
+
413
+ try:
414
+ mcp.run(transport=args.transport, **_transport_options(args.transport))
415
+ except Exception as e:
416
+ logger.error(f"Error starting Knowledgebase MCP Server: {str(e)}")
417
+ raise
418
+
419
+
420
+ if __name__ == "__main__":
421
+ main()
@@ -0,0 +1,248 @@
1
+ Metadata-Version: 2.5
2
+ Name: mcp-server-knowledgebase
3
+ Version: 0.2.0
4
+ Summary: MCP server for Viking Knowledge Base Service
5
+ License: MIT
6
+ Requires-Python: >=3.10
7
+ Requires-Dist: aiohttp>=3.11.14
8
+ Requires-Dist: mcp[cli]<3,>=2.1.1
9
+ Requires-Dist: volcengine>=1.0.171
10
+ Description-Content-Type: text/markdown
11
+
12
+ # Viking Knowledge Base MCP Server
13
+
14
+ This MCP server provides a tool to interact with the VolcEngine Viking Knowledge Base Service, allowing you to search and retrieve knowledge from your collections, meanwhile,
15
+ allowing you to add doc to your collections and get doc processing info by doc_id.
16
+
17
+ ## Features
18
+
19
+ - Search knowledge based on queries with customizable parameters
20
+
21
+ ## Setup
22
+
23
+ ### Prerequisites
24
+
25
+ - Python 3.10 or higher
26
+ - A Viking Knowledge Base API key or VolcEngine AK/SK credentials
27
+
28
+ ### Installation
29
+
30
+ 1. Install the package:
31
+
32
+ ```bash
33
+ pip install -e .
34
+ ```
35
+
36
+ Or with uv (recommended):
37
+
38
+ ```bash
39
+ uv pip install -e .
40
+ ```
41
+
42
+ ### Configuration
43
+
44
+ The server requires at least one authentication method:
45
+
46
+ - API key: set `VIKING_API_KEY`. Requests use
47
+ `Authorization: Bearer <VIKING_API_KEY>`.
48
+ - AK/SK: set both `VOLCENGINE_ACCESS_KEY` and `VOLCENGINE_SECRET_KEY`.
49
+ Requests use VolcEngine SignerV4 authentication.
50
+
51
+ When both methods are configured, `VIKING_API_KEY` takes precedence and AK/SK
52
+ is ignored. When no API key is configured, AK and SK must be provided together.
53
+ The server rejects configurations with no usable authentication method.
54
+
55
+ Optional environment variables:
56
+ - `KNOWLEDGE_BASE_PROJECT`: Viking Knowledge Base project name (default: `default`)
57
+ - `KNOWLEDGE_BASE_REGION`: Viking Knowledge Base region (default: `cn-north-1`)
58
+ - `MCP_SERVER_HOST`: Streamable HTTP bind host (default: `127.0.0.1`)
59
+ - `MCP_SERVER_PORT`: Streamable HTTP port; falls back to `PORT` (default: `8000`)
60
+ - `STREAMABLE_HTTP_PATH`: Streamable HTTP endpoint path (default: `/mcp`)
61
+ - `KNOWLEDGE_BASE_TIMEOUT`: Upstream request timeout in seconds (default: `30`)
62
+
63
+ ## Usage
64
+
65
+ ### Running the Server
66
+
67
+ The server supports stdio for local integrations and stateless Streamable HTTP
68
+ for remote deployments:
69
+
70
+ ```bash
71
+ python -m mcp_server_knowledgebase.server --transport stdio
72
+ ```
73
+
74
+ Or:
75
+
76
+ ```bash
77
+ python -m mcp_server_knowledgebase.server --transport streamable-http
78
+ ```
79
+
80
+ The Streamable HTTP endpoint is `http://127.0.0.1:8000/mcp` by default.
81
+ Set `MCP_SERVER_HOST=0.0.0.0` when running behind a trusted gateway.
82
+
83
+ ### MCP protocol compatibility
84
+
85
+ This server uses MCP Python SDK 2.x and speaks protocol revision `2026-07-28`.
86
+ Modern clients use the stateless per-request protocol and `server/discover`;
87
+ the same process also supports older handshake-based clients automatically.
88
+ Legacy HTTP+SSE is intentionally not exposed because it is deprecated by the
89
+ `2026-07-28` specification.
90
+
91
+ The HTTP endpoint does not turn the configured API key or VolcEngine AK/SK into
92
+ client authentication. Protect remote deployments with an authentication
93
+ gateway or MCP-compatible OAuth, and never expose the service credentials to
94
+ callers.
95
+
96
+ ### Available Tools
97
+
98
+ #### add_doc
99
+
100
+ Add a document to a collection in your project.
101
+
102
+ ```python
103
+ add_doc(
104
+ collection_name="collection_name",
105
+ add_type="url",
106
+ doc_id="mcp_server_auto_gen_doc_id_xxxxxxx",
107
+ doc_name="doc_xxxx",
108
+ doc_type="pdf",
109
+ url="http://xxxxx.pdf"
110
+ )
111
+ ```
112
+
113
+ Parameters:
114
+ - `collection_name` (required): the name of the collection you want to add document .
115
+ - `add_type` (required): the type of the document to add. so far only support "url" now.
116
+ - `doc_id` (required): you should generate a unique doc_id based on user's given url and timestamp, the doc_id can only use English letters, numbers, and underscores , and must start with an English letter. It cannot be empty. Length requirement: [1, 128], you can use a format like "mcp_server_auto_gen_doc_id_xxxxxxx".
117
+ - `doc_name` (required): the name of the document to add. You can generate a unique doc_name based on the user-provided URL and timestamp. The length of doc_name must be between 1 and 256; for example, "mcp_server_auto_gen_doc_name_xxxxxxx".
118
+ - `doc_type` (required): the type of the document to add. for structured document, we support xlsx, csv,jsonl, for unstructured document, wu support txt, doc, docx, pdf, markdown, faq.xlsx, pptx". you should judge the doc_type based on user's given url and judge if we support this doc type. if supported, assign this parameter.
119
+ - `url` (required): the url of the document to add. user should give a valid url, we will add the doc to the collection.
120
+
121
+ #### get_doc
122
+
123
+ Get information about document by collection_name and doc_id .
124
+
125
+ ```python
126
+ get_doc(
127
+ collection_name="collection_name",
128
+ doc_id="mcp_server_auto_gen_doc_id_xxxxxxx",
129
+ )
130
+ ```
131
+
132
+ Parameters:
133
+ - `collection_name` (required): the name of the collection you want to get information .
134
+ - `doc_id` (required): the doc_id of document user want to get information .
135
+
136
+ #### get_collection
137
+
138
+ Get information about a viking knowledge base collection from your project .
139
+
140
+ ```python
141
+ get_collection(
142
+ collection_name="collection_name",
143
+ )
144
+ ```
145
+
146
+ Parameters:
147
+ - `collection_name` (required): the name of the collection you want to get information .
148
+
149
+
150
+ #### list_collections
151
+
152
+ List all knowledge base collections of the globally configured project .
153
+
154
+ ```python
155
+ list_collections(
156
+ )
157
+ ```
158
+
159
+
160
+ #### search_knowledge
161
+
162
+ Search for knowledge in the configured collection based on a query.
163
+
164
+ ```python
165
+ search_knowledge(
166
+ query="How to reset my password?",
167
+ limit=3,
168
+ collection_name="collection_name",
169
+ doc_filter=None,
170
+ )
171
+ ```
172
+
173
+ Parameters:
174
+ - `query` (required): The search query string
175
+ - `limit` (optional): Maximum number of results to return, from 1 to 100 (default: 3)
176
+ - `collection_name` (required): Knowledge Base collection name to search
177
+ - `doc_filter` (optional): the filter is used to filter search results(default: None), which is structured as a JSON object with the following key components:
178
+ - `op` (string, required): specifies the query operator that defines the filtering logic. Valid values are 'must' and 'must_not', 'must' means results must satisfy the condition (inclusion filter),'must_not' means results must not satisfy the condition (exclusion filter).
179
+ - `field` (string, required): indicates the specific document field to apply the filter on (e.g., "doc_id").
180
+ - `conds` (array, required): contains the concrete values used for filtering. The data type of elements in the array depends on the field.
181
+
182
+ Each result contains the chunk `id` and `content`, plus the source document's
183
+ `doc_id` and `doc_name`. The metadata fields are `null` when Viking does not
184
+ provide them. A non-null `doc_id` can be passed directly to `get_doc`.
185
+
186
+ ## MCP Integration
187
+
188
+ To add this server to your MCP configuration, add the following to your MCP settings file:
189
+
190
+ ```json
191
+ {
192
+ "mcpServers": {
193
+ "knowledgebase": {
194
+ "command": "uvx",
195
+ "args": [
196
+ "--from",
197
+ "mcp-server-knowledgebase>=0.2.0",
198
+ "mcp-server-knowledgebase"
199
+ ],
200
+ "env": {
201
+ "VIKING_API_KEY": "your-viking-api-key",
202
+ "KNOWLEDGE_BASE_PROJECT": "your-project-name",
203
+ "KNOWLEDGE_BASE_REGION": "your-region"
204
+ }
205
+ }
206
+ }
207
+ }
208
+ ```
209
+
210
+ You may alternatively or additionally configure both
211
+ `VOLCENGINE_ACCESS_KEY` and `VOLCENGINE_SECRET_KEY`. If all three variables are
212
+ set, `VIKING_API_KEY` takes precedence.
213
+
214
+ ## Troubleshooting
215
+
216
+ ### Common Issues
217
+
218
+ 1. **Authentication Errors**
219
+ - Verify your API key or AK/SK credentials are correct
220
+ - Ensure at least one authentication method is configured
221
+ - Check that you have the necessary permissions for the collection
222
+
223
+ 2. **Connection Timeouts**
224
+ - Check your network connection to the VolcEngine API
225
+ - Verify the host configuration is correct
226
+
227
+ 3. **Empty Results**
228
+ - Verify the collection name is correct
229
+ - Try broadening your search query
230
+
231
+ ### Logging
232
+
233
+ The server uses Python's logging module with INFO level by default. You can see detailed logs in the console when running the server.
234
+
235
+ ## Contributing
236
+
237
+ Contributions to improve the Viking Knowledge Base MCP Server are welcome. Please follow these steps:
238
+
239
+ 1. Fork the repository
240
+ 2. Create a feature branch
241
+ 3. Make your changes
242
+ 4. Submit a pull request
243
+
244
+ Please ensure your code follows the project's coding standards and includes appropriate tests.
245
+
246
+ ## License
247
+
248
+ volcengine/mcp-server is licensed under the [MIT License](https://github.com/volcengine/mcp-server/blob/main/LICENSE).
@@ -0,0 +1,10 @@
1
+ mcp_server_knowledgebase/__init__.py,sha256=aUck0pBUWsafYyf6Ch2F7iYQaxLQrH198HQbXPDp8ms,25
2
+ mcp_server_knowledgebase/config.py,sha256=jQdEtU3ho37T88w-IuFCaYK4OP9g_rLN7hiwpo-0S7M,1580
3
+ mcp_server_knowledgebase/models.py,sha256=Gly__XmaYyU1bjzPqm9hU9wb4XXQ97iywSWxzAvT6Ns,1541
4
+ mcp_server_knowledgebase/server.py,sha256=73D_84J8LebZK8BbnhYBatqYQSOfR2VXKyjjsRSJRME,15546
5
+ mcp_server_knowledgebase/common/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
6
+ mcp_server_knowledgebase/common/auth.py,sha256=SV2fVjtto321yKPYTCK56D_Q6MSqyuN8ey-Mee1PzNg,1710
7
+ mcp_server_knowledgebase-0.2.0.dist-info/METADATA,sha256=sYuZbYuSIsuYji8QvhTtu3fhs_YXhr13UvB0xCVgRLU,8408
8
+ mcp_server_knowledgebase-0.2.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
9
+ mcp_server_knowledgebase-0.2.0.dist-info/entry_points.txt,sha256=Ou4PBntYVXnKxPxzpNWCyVEPaJySuEIY24xogVOeGcw,82
10
+ mcp_server_knowledgebase-0.2.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ mcp-server-knowledgebase = mcp_server_knowledgebase.server:main