@dataville/dataville-mcp 0.1.6 → 0.1.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -3
- package/dist/client.js +58 -6
- package/dist/server.js +103 -17
- package/dist/sources.js +161 -12
- package/package.json +5 -5
package/README.md
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# dataville-mcp
|
|
2
2
|
|
|
3
3
|
[](https://m8ven.ai/mcp/datavilleorg-dataville-mcp-u1eou7?s=readme)
|
|
4
|
+
[](https://glama.ai/mcp/servers/datavilleorg/dataville-mcp)
|
|
4
5
|
[](https://www.npmjs.com/package/@dataville/dataville-mcp)
|
|
5
6
|
[](https://github.com/datavilleorg/dataville-mcp/actions/workflows/ci.yml)
|
|
6
7
|
[](LICENSE)
|
|
@@ -20,8 +21,18 @@ Requires a Dataville API key — get one from the [Dataville dashboard](https://
|
|
|
20
21
|
|
|
21
22
|
## Tools
|
|
22
23
|
|
|
23
|
-
|
|
24
|
-
|
|
24
|
+
| Tool | What it does |
|
|
25
|
+
| --- | --- |
|
|
26
|
+
| `list_dataville_sources` | Every source, with the SQL tables (if any) you can query for it. |
|
|
27
|
+
| `describe_dataville_source` | One source's keyword format, a working example, and its tables' columns: `{ source }`. |
|
|
28
|
+
| `search_dataville` | Look one thing up by keywords and get the best match: `{ source, keywords, summary? }`. |
|
|
29
|
+
| `query_dataville` | Read-only SQL `SELECT` over the stored tables for lists, filters, counts, joins, and paging with `LIMIT`/`OFFSET`: `{ sql }`. |
|
|
30
|
+
|
|
31
|
+
All four only read data. `query_dataville` covers every stored source except
|
|
32
|
+
`census` and `news`, which are fetched live per search. It returns at most
|
|
33
|
+
1,000 rows per call and is billed per row returned. It reads Dataville's stored
|
|
34
|
+
copy: some sources are loaded in bulk, others only hold records fetched before,
|
|
35
|
+
so for one specific item `search_dataville` is the reliable route.
|
|
25
36
|
|
|
26
37
|
## What you can ask
|
|
27
38
|
|
|
@@ -32,6 +43,8 @@ Once connected, ask in plain language and the client picks the source:
|
|
|
32
43
|
- "What's the median household income in Travis County, Texas, per the US Census?"
|
|
33
44
|
- "How much protein is in 100 g of cooked lentils, according to USDA FoodData?"
|
|
34
45
|
- "Look up the `requests` package on PyPI — what's the latest version and license?"
|
|
46
|
+
- "List ten Project Gutenberg books by Jane Austen, with their subjects."
|
|
47
|
+
- "List five USDA FoodData entries with 'lentils' in the name, with their categories."
|
|
35
48
|
|
|
36
49
|
## Setup
|
|
37
50
|
|
|
@@ -110,6 +123,18 @@ every project instead of just the current one.
|
|
|
110
123
|
`DATAVILLE_API_BASE_URL` is optional and defaults to `https://api.dataville.com`;
|
|
111
124
|
set it to `http://localhost:5000` to point at a local backend during development.
|
|
112
125
|
|
|
126
|
+
### Where your API key goes
|
|
127
|
+
|
|
128
|
+
The server reads `DATAVILLE_API_KEY` from its environment and sends it only as
|
|
129
|
+
an `Authorization: Bearer` header to `DATAVILLE_API_BASE_URL`, which is
|
|
130
|
+
`https://api.dataville.com` unless you change it. It never logs the key or sends
|
|
131
|
+
it anywhere else. The base URL must use `https`. Plain `http` is accepted only
|
|
132
|
+
for `localhost`, `127.0.0.1` and `[::1]`, so the key never crosses the network
|
|
133
|
+
unencrypted.
|
|
134
|
+
|
|
135
|
+
All four tools are read-only: they search and query Dataville and never create,
|
|
136
|
+
change or delete anything.
|
|
137
|
+
|
|
113
138
|
### Running from source
|
|
114
139
|
|
|
115
140
|
```bash
|
|
@@ -131,6 +156,7 @@ on its own proves nothing.
|
|
|
131
156
|
| `Using Dataville, what is the latest version of the requests package on PyPI?` | A version you can confirm on pypi.org — and it moves, so it can't come from memory. |
|
|
132
157
|
| `Using Dataville, get the latest SEC filing for AAPL and its revenue.` | A form type, filing date, revenue figure, and a sec.gov link to open. |
|
|
133
158
|
| `Using Dataville, how much protein is in 100g of uncooked quinoa?` | The exact USDA figure, 14.1 g per 100 g. |
|
|
159
|
+
| `Using Dataville SQL, list 5 Project Gutenberg books by Mark Twain.` | Calls `query_dataville` and returns titles with Gutenberg IDs you can open at gutenberg.org/ebooks/<id>. |
|
|
134
160
|
|
|
135
161
|
Clients show when a tool ran. If you don't see that, say "use dataville" in the
|
|
136
162
|
prompt to make it explicit, and check the answer against the source.
|
|
@@ -145,7 +171,8 @@ npm run build # tsc
|
|
|
145
171
|
|
|
146
172
|
## Releasing
|
|
147
173
|
|
|
148
|
-
Publishes run from CI
|
|
174
|
+
Publishes run from CI. npm uses trusted publishing (OIDC), so no npm token is
|
|
175
|
+
stored; the MCP Registry step signs in with the `MCP_PRIVATE_KEY` repo secret.
|
|
149
176
|
To cut a release: bump the version, update `CHANGELOG.md`, then publish a GitHub
|
|
150
177
|
Release for the new tag. The `Publish` workflow builds, tests, and publishes to npm,
|
|
151
178
|
then publishes `server.json` to the [MCP Registry](https://registry.modelcontextprotocol.io)
|
package/dist/client.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
const DEFAULT_BASE_URL = "https://api.dataville.com";
|
|
2
2
|
// Keep in sync with package.json version (enforced by a test).
|
|
3
|
-
export const VERSION = "0.1.
|
|
3
|
+
export const VERSION = "0.1.7";
|
|
4
4
|
const USER_AGENT = `dataville-mcp/${VERSION}`;
|
|
5
5
|
export class DatavilleApiError extends Error {
|
|
6
6
|
status;
|
|
@@ -31,8 +31,29 @@ function getConfig() {
|
|
|
31
31
|
throw new Error("DATAVILLE_API_KEY is not set. Generate an API key from the Dataville dashboard and set it in your MCP client config.");
|
|
32
32
|
}
|
|
33
33
|
const baseUrl = process.env.DATAVILLE_API_BASE_URL || DEFAULT_BASE_URL;
|
|
34
|
+
assertSafeBaseUrl(baseUrl);
|
|
34
35
|
return { apiKey, baseUrl };
|
|
35
36
|
}
|
|
37
|
+
const LOCAL_HOSTS = new Set(["localhost", "127.0.0.1", "[::1]"]);
|
|
38
|
+
/**
|
|
39
|
+
* Every request carries the API key, so refuse a base URL that would send it
|
|
40
|
+
* in cleartext. Plain http is allowed only for a local backend.
|
|
41
|
+
*/
|
|
42
|
+
function assertSafeBaseUrl(baseUrl) {
|
|
43
|
+
let url;
|
|
44
|
+
try {
|
|
45
|
+
url = new URL(baseUrl);
|
|
46
|
+
}
|
|
47
|
+
catch {
|
|
48
|
+
throw new Error(`DATAVILLE_API_BASE_URL "${baseUrl}" is not a valid URL.`);
|
|
49
|
+
}
|
|
50
|
+
if (url.protocol === "https:")
|
|
51
|
+
return;
|
|
52
|
+
if (url.protocol === "http:" && LOCAL_HOSTS.has(url.hostname))
|
|
53
|
+
return;
|
|
54
|
+
throw new Error(`DATAVILLE_API_BASE_URL must use https (plain http is allowed only for localhost), got "${baseUrl}". ` +
|
|
55
|
+
"Your API key is sent with every request, so it is not sent over an unencrypted connection.");
|
|
56
|
+
}
|
|
36
57
|
export async function searchDataSource(source, keywords, params) {
|
|
37
58
|
const { apiKey, baseUrl } = getConfig();
|
|
38
59
|
// encodeURIComponent leaves "." alone, and URL resolution collapses a "." or
|
|
@@ -55,14 +76,45 @@ export async function searchDataSource(source, keywords, params) {
|
|
|
55
76
|
"User-Agent": USER_AGENT,
|
|
56
77
|
},
|
|
57
78
|
});
|
|
79
|
+
return readResponse(response);
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* Run a read-only SQL SELECT against Dataville's stored copy of its sources
|
|
83
|
+
* (DuckDB over the tables listed by describe_dataville_source). The API
|
|
84
|
+
* rejects anything but SELECT, caps results at 1,000 rows, and times out after
|
|
85
|
+
* 30 seconds; it bills per row returned.
|
|
86
|
+
*/
|
|
87
|
+
export async function queryDataville(sql) {
|
|
88
|
+
const { apiKey, baseUrl } = getConfig();
|
|
89
|
+
const response = await fetch(new URL("/api/v1/query", baseUrl), {
|
|
90
|
+
method: "POST",
|
|
91
|
+
headers: {
|
|
92
|
+
Authorization: `Bearer ${apiKey}`,
|
|
93
|
+
"User-Agent": USER_AGENT,
|
|
94
|
+
"Content-Type": "application/json",
|
|
95
|
+
},
|
|
96
|
+
body: JSON.stringify({ sql }),
|
|
97
|
+
});
|
|
98
|
+
return readResponse(response);
|
|
99
|
+
}
|
|
100
|
+
const FIX_KEY_HINT = "Check the key in your MCP client config, or generate a new one at https://app.dataville.com/api-keys.";
|
|
101
|
+
async function readResponse(response) {
|
|
58
102
|
const body = await response.json().catch(() => undefined);
|
|
103
|
+
// SQL queries need an account, so an unrecognised key gets a 401 there
|
|
104
|
+
// rather than the anonymous fallback search uses. The API's own message
|
|
105
|
+
// ("include your API key") would be misleading — we did include one.
|
|
106
|
+
if (response.status === 401) {
|
|
107
|
+
throw new DatavilleAuthError(`Your DATAVILLE_API_KEY was not recognised. ${FIX_KEY_HINT}`);
|
|
108
|
+
}
|
|
59
109
|
if (!response.ok) {
|
|
60
|
-
//
|
|
61
|
-
//
|
|
62
|
-
//
|
|
110
|
+
// Search errors come back as { status: "error", data: { error: "..." } },
|
|
111
|
+
// query errors as { status: "error", message: "..." }. Those messages are
|
|
112
|
+
// useful to the model — "no results for X", the list of valid sources, or
|
|
113
|
+
// the SQL error — so prefer them over a bare status code.
|
|
63
114
|
const apiMessage = body && typeof body === "object"
|
|
64
115
|
? body.data?.error ??
|
|
65
|
-
body.error
|
|
116
|
+
body.error ??
|
|
117
|
+
body.message
|
|
66
118
|
: undefined;
|
|
67
119
|
const message = typeof apiMessage === "string" && apiMessage.length > 0
|
|
68
120
|
? apiMessage
|
|
@@ -77,7 +129,7 @@ export async function searchDataSource(source, keywords, params) {
|
|
|
77
129
|
body.account_state === "anonymous") {
|
|
78
130
|
throw new DatavilleAuthError("Your DATAVILLE_API_KEY was not recognised, so this request fell back to anonymous access " +
|
|
79
131
|
"(much lower rate limits, and usage is not attributed to your account). " +
|
|
80
|
-
|
|
132
|
+
FIX_KEY_HINT);
|
|
81
133
|
}
|
|
82
134
|
return body;
|
|
83
135
|
}
|
package/dist/server.js
CHANGED
|
@@ -1,15 +1,30 @@
|
|
|
1
1
|
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
2
2
|
import { z } from "zod";
|
|
3
|
-
import { searchDataSource, VERSION } from "./client.js";
|
|
4
|
-
import { DATAVILLE_SOURCES } from "./sources.js";
|
|
3
|
+
import { queryDataville, searchDataSource, VERSION } from "./client.js";
|
|
4
|
+
import { DATAVILLE_SOURCE_DETAILS, DATAVILLE_SOURCES, findSource } from "./sources.js";
|
|
5
|
+
const SQL_TABLES = DATAVILLE_SOURCE_DETAILS.flatMap((s) => s.sqlTables.map((t) => t.name));
|
|
6
|
+
const INSTRUCTIONS = `Dataville serves public datasets (Wikipedia, arXiv, SEC EDGAR, US Census, USDA FoodData, PyPI and more) through one API.
|
|
7
|
+
|
|
8
|
+
- list_dataville_sources: what sources exist.
|
|
9
|
+
- describe_dataville_source: what keywords a source expects, and the SQL tables and columns it has.
|
|
10
|
+
- search_dataville: look one thing up by keywords; returns the single best match.
|
|
11
|
+
- query_dataville: SQL SELECT over the stored tables, for lists, filters, counts, joins and paging through many rows.`;
|
|
12
|
+
function json(payload) {
|
|
13
|
+
return { content: [{ type: "text", text: JSON.stringify(payload, null, 2) }] };
|
|
14
|
+
}
|
|
15
|
+
function toolError(error) {
|
|
16
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
17
|
+
return { content: [{ type: "text", text: message }], isError: true };
|
|
18
|
+
}
|
|
5
19
|
export function createServer() {
|
|
6
20
|
const server = new McpServer({
|
|
7
21
|
name: "dataville-mcp-server",
|
|
8
22
|
version: VERSION,
|
|
9
|
-
});
|
|
23
|
+
}, { instructions: INSTRUCTIONS });
|
|
10
24
|
server.registerTool("list_dataville_sources", {
|
|
11
25
|
title: "List Dataville data sources",
|
|
12
|
-
description: "List
|
|
26
|
+
description: "List every Dataville data source with a one-line description and the SQL tables (if any) that query_dataville can read for it. " +
|
|
27
|
+
"Call describe_dataville_source for a source's keyword format and table columns.",
|
|
13
28
|
inputSchema: {},
|
|
14
29
|
// Returns a static list bundled with the server; makes no network calls.
|
|
15
30
|
annotations: {
|
|
@@ -18,19 +33,59 @@ export function createServer() {
|
|
|
18
33
|
idempotentHint: true,
|
|
19
34
|
openWorldHint: false,
|
|
20
35
|
},
|
|
21
|
-
}, async () => ({
|
|
22
|
-
|
|
23
|
-
|
|
36
|
+
}, async () => json(DATAVILLE_SOURCE_DETAILS.map(({ name, description, sqlTables }) => ({
|
|
37
|
+
name,
|
|
38
|
+
description,
|
|
39
|
+
sql_tables: sqlTables.map((t) => t.name),
|
|
40
|
+
}))));
|
|
41
|
+
server.registerTool("describe_dataville_source", {
|
|
42
|
+
title: "Describe a Dataville data source",
|
|
43
|
+
description: "Show how to use one data source: what search_dataville expects as keywords (with an example that returns a result), " +
|
|
44
|
+
"and the SQL tables and columns query_dataville can read for it. Sources with no tables can only be searched.",
|
|
45
|
+
inputSchema: {
|
|
46
|
+
source: z.string().describe("Data source name from list_dataville_sources, e.g. 'edgar'"),
|
|
47
|
+
},
|
|
48
|
+
// Returns static metadata bundled with the server; makes no network calls.
|
|
49
|
+
annotations: {
|
|
50
|
+
readOnlyHint: true,
|
|
51
|
+
destructiveHint: false,
|
|
52
|
+
idempotentHint: true,
|
|
53
|
+
openWorldHint: false,
|
|
54
|
+
},
|
|
55
|
+
}, async ({ source }) => {
|
|
56
|
+
const details = findSource(source);
|
|
57
|
+
if (!details) {
|
|
58
|
+
return toolError(`Unknown source "${source}". Valid sources: ${DATAVILLE_SOURCES.map((s) => s.name).join(", ")}.`);
|
|
59
|
+
}
|
|
60
|
+
return json({
|
|
61
|
+
name: details.name,
|
|
62
|
+
description: details.description,
|
|
63
|
+
search: {
|
|
64
|
+
keywords: details.keywords,
|
|
65
|
+
example: { source: details.name, keywords: details.exampleKeywords },
|
|
66
|
+
},
|
|
67
|
+
sql_tables: details.sqlTables,
|
|
68
|
+
...(details.sqlTables.length > 0
|
|
69
|
+
? { example_sql: `SELECT * FROM ${details.sqlTables[0].name} LIMIT 5` }
|
|
70
|
+
: { note: "This source is fetched live per search and has no SQL table; use search_dataville." }),
|
|
71
|
+
});
|
|
72
|
+
});
|
|
24
73
|
server.registerTool("search_dataville", {
|
|
25
74
|
title: "Search a Dataville data source",
|
|
26
|
-
description: "
|
|
75
|
+
description: "Look up one thing in a Dataville data source by keywords and get back the single best-matching record " +
|
|
76
|
+
"(full text in `body`, structured fields in `metadata`). Use describe_dataville_source to see what keywords a source expects. " +
|
|
77
|
+
"To get many records, filter, or page through results, use query_dataville instead.",
|
|
27
78
|
inputSchema: {
|
|
28
|
-
source: z.string().describe("Data source name, e.g. 'wikipedia', 'arxiv', 'edgar'"),
|
|
29
|
-
keywords: z.string().describe("Search keywords or identifier
|
|
79
|
+
source: z.string().describe("Data source name, e.g. 'wikipedia', 'arxiv', 'edgar' (see list_dataville_sources)"),
|
|
80
|
+
keywords: z.string().describe("Search keywords or identifier, e.g. a title, ticker, or package name"),
|
|
81
|
+
summary: z
|
|
82
|
+
.boolean()
|
|
83
|
+
.optional()
|
|
84
|
+
.describe("If true, truncate `body` to its first 100 characters to save tokens"),
|
|
30
85
|
params: z
|
|
31
|
-
.record(z.union([z.string(), z.number(), z.boolean()]))
|
|
86
|
+
.record(z.string(), z.union([z.string(), z.number(), z.boolean()]))
|
|
32
87
|
.optional()
|
|
33
|
-
.describe("
|
|
88
|
+
.describe("Extra query-string parameters passed through to the API as-is. Most callers need none."),
|
|
34
89
|
},
|
|
35
90
|
// A GET against the Dataville API, which fronts external data sources.
|
|
36
91
|
annotations: {
|
|
@@ -39,14 +94,45 @@ export function createServer() {
|
|
|
39
94
|
idempotentHint: true,
|
|
40
95
|
openWorldHint: true,
|
|
41
96
|
},
|
|
42
|
-
}, async ({ source, keywords, params }) => {
|
|
97
|
+
}, async ({ source, keywords, summary, params }) => {
|
|
98
|
+
try {
|
|
99
|
+
const query = summary === undefined ? params : { ...params, summary };
|
|
100
|
+
return json(await searchDataSource(source, keywords, query));
|
|
101
|
+
}
|
|
102
|
+
catch (error) {
|
|
103
|
+
return toolError(error);
|
|
104
|
+
}
|
|
105
|
+
});
|
|
106
|
+
server.registerTool("query_dataville", {
|
|
107
|
+
title: "Query Dataville with SQL",
|
|
108
|
+
description: "Run a read-only SQL SELECT (DuckDB dialect) over Dataville's stored tables: " +
|
|
109
|
+
`${SQL_TABLES.join(", ")}. ` +
|
|
110
|
+
"Use it to list, filter (WHERE … ILIKE '%term%'), sort, count, aggregate, or join across sources, " +
|
|
111
|
+
"and page through results with LIMIT/OFFSET. Get each table's columns from describe_dataville_source. " +
|
|
112
|
+
"Only SELECT is allowed; at most 1,000 rows are returned (a missing LIMIT becomes LIMIT 1000); queries time out after 30 s. " +
|
|
113
|
+
"Billed per row returned, so keep LIMIT as small as the task allows. " +
|
|
114
|
+
"The tables hold Dataville's stored copy: some sources are loaded in bulk, others only hold records fetched before, " +
|
|
115
|
+
"so a row missing here may still exist upstream — fall back to search_dataville for a specific item.",
|
|
116
|
+
inputSchema: {
|
|
117
|
+
sql: z
|
|
118
|
+
.string()
|
|
119
|
+
.min(1)
|
|
120
|
+
.max(10_000)
|
|
121
|
+
.describe("A single SELECT statement, e.g. \"SELECT company_name, form_type, filing_date FROM edgar WHERE company_name ILIKE '%apple%' ORDER BY filing_date DESC LIMIT 10\""),
|
|
122
|
+
},
|
|
123
|
+
// A POST, but only ever a SELECT: the API rejects anything that writes.
|
|
124
|
+
annotations: {
|
|
125
|
+
readOnlyHint: true,
|
|
126
|
+
destructiveHint: false,
|
|
127
|
+
idempotentHint: true,
|
|
128
|
+
openWorldHint: true,
|
|
129
|
+
},
|
|
130
|
+
}, async ({ sql }) => {
|
|
43
131
|
try {
|
|
44
|
-
|
|
45
|
-
return { content: [{ type: "text", text: JSON.stringify(result, null, 2) }] };
|
|
132
|
+
return json(await queryDataville(sql));
|
|
46
133
|
}
|
|
47
134
|
catch (error) {
|
|
48
|
-
|
|
49
|
-
return { content: [{ type: "text", text: message }], isError: true };
|
|
135
|
+
return toolError(error);
|
|
50
136
|
}
|
|
51
137
|
});
|
|
52
138
|
return server;
|
package/dist/sources.js
CHANGED
|
@@ -2,16 +2,165 @@
|
|
|
2
2
|
// The `name` values must match the source names the API accepts — a mismatch
|
|
3
3
|
// makes the source unreachable. To check against the live API, request an
|
|
4
4
|
// unknown source (e.g. /nosuchsource/foo); the error lists every valid name.
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
{
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
5
|
+
//
|
|
6
|
+
// `sqlTables` mirrors the DuckDB views the API's /api/v1/query endpoint
|
|
7
|
+
// exposes (backend services/duckdb.ts). The integration tests check each
|
|
8
|
+
// table and column against the live API.
|
|
9
|
+
export const DATAVILLE_SOURCE_DETAILS = [
|
|
10
|
+
{
|
|
11
|
+
name: "wikipedia",
|
|
12
|
+
description: "Wikipedia articles",
|
|
13
|
+
keywords: "An article title or topic; the closest matching English Wikipedia article is returned.",
|
|
14
|
+
exampleKeywords: "Machine learning",
|
|
15
|
+
sqlTables: [
|
|
16
|
+
{
|
|
17
|
+
name: "wiki",
|
|
18
|
+
columns: [
|
|
19
|
+
"id", "pageid", "title", "body", "abstract", "description", "language", "url", "project",
|
|
20
|
+
"is_seed", "categories", "license", "infoboxes", "citations", "tables", "main_entity", "image",
|
|
21
|
+
"origin", "date_modified", "snapshot_date", "last_fetched",
|
|
22
|
+
],
|
|
23
|
+
},
|
|
24
|
+
],
|
|
25
|
+
},
|
|
26
|
+
{
|
|
27
|
+
name: "arxiv",
|
|
28
|
+
description: "arXiv preprints",
|
|
29
|
+
keywords: "A topic, title, or arXiv ID.",
|
|
30
|
+
exampleKeywords: "transformer",
|
|
31
|
+
sqlTables: [
|
|
32
|
+
{
|
|
33
|
+
name: "arxiv",
|
|
34
|
+
columns: [
|
|
35
|
+
"id", "arxiv_id", "title", "authors", "categories", "published", "abs_url", "pdf_url", "body",
|
|
36
|
+
"last_updated", "last_fetched",
|
|
37
|
+
],
|
|
38
|
+
},
|
|
39
|
+
],
|
|
40
|
+
},
|
|
41
|
+
{
|
|
42
|
+
name: "gutenberg",
|
|
43
|
+
description: "Project Gutenberg public-domain books",
|
|
44
|
+
keywords: "A book title, author, or subject.",
|
|
45
|
+
exampleKeywords: "Alice",
|
|
46
|
+
sqlTables: [
|
|
47
|
+
{
|
|
48
|
+
name: "gutenberg",
|
|
49
|
+
columns: ["id", "title", "authors", "language", "subjects", "issued", "locc", "bookshelves", "type"],
|
|
50
|
+
},
|
|
51
|
+
],
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
name: "census",
|
|
55
|
+
description: "US Census Bureau data",
|
|
56
|
+
keywords: "A statistic and a place, e.g. a measure such as population or median household income plus a state or county.",
|
|
57
|
+
exampleKeywords: "population",
|
|
58
|
+
sqlTables: [],
|
|
59
|
+
},
|
|
60
|
+
{
|
|
61
|
+
name: "fooddata",
|
|
62
|
+
description: "USDA FoodData Central",
|
|
63
|
+
keywords: "A food name, optionally with a nutrient.",
|
|
64
|
+
exampleKeywords: "apple",
|
|
65
|
+
sqlTables: [
|
|
66
|
+
{ name: "fooddata", columns: ["fdc_id", "description", "category", "data_type", "nutrients"] },
|
|
67
|
+
],
|
|
68
|
+
},
|
|
69
|
+
{
|
|
70
|
+
name: "paperswithcode",
|
|
71
|
+
description: "Papers with Code — ML papers, code, and benchmarks",
|
|
72
|
+
keywords: "An ML method or dataset name.",
|
|
73
|
+
exampleKeywords: "transformer",
|
|
74
|
+
sqlTables: [
|
|
75
|
+
{
|
|
76
|
+
name: "pwc_methods",
|
|
77
|
+
columns: [
|
|
78
|
+
"id", "name", "full_name", "description", "introduced_year", "paper_title", "paper_url", "code_url",
|
|
79
|
+
"categories",
|
|
80
|
+
],
|
|
81
|
+
},
|
|
82
|
+
{
|
|
83
|
+
name: "pwc_datasets",
|
|
84
|
+
columns: [
|
|
85
|
+
"id", "name", "full_name", "description", "url", "paper_title", "paper_url", "tasks", "subtasks",
|
|
86
|
+
"num_papers",
|
|
87
|
+
],
|
|
88
|
+
},
|
|
89
|
+
],
|
|
90
|
+
},
|
|
91
|
+
{
|
|
92
|
+
name: "edgar",
|
|
93
|
+
description: "SEC EDGAR filings",
|
|
94
|
+
keywords: "A company name or stock ticker; returns its latest filing.",
|
|
95
|
+
exampleKeywords: "AAPL",
|
|
96
|
+
sqlTables: [
|
|
97
|
+
{
|
|
98
|
+
name: "edgar",
|
|
99
|
+
columns: [
|
|
100
|
+
"id", "accession_number", "cik", "company_name", "form_type", "filing_date", "description",
|
|
101
|
+
"document_url", "index_url", "body", "last_fetched",
|
|
102
|
+
],
|
|
103
|
+
},
|
|
104
|
+
],
|
|
105
|
+
},
|
|
106
|
+
{
|
|
107
|
+
name: "openalex",
|
|
108
|
+
description: "OpenAlex scholarly works",
|
|
109
|
+
keywords: "A topic, title, or DOI.",
|
|
110
|
+
exampleKeywords: "transformer",
|
|
111
|
+
sqlTables: [
|
|
112
|
+
{
|
|
113
|
+
name: "openalex",
|
|
114
|
+
columns: [
|
|
115
|
+
"id", "openalex_id", "doi", "title", "publication_year", "publication_date", "authors", "venue",
|
|
116
|
+
"cited_by_count", "type", "url", "body", "last_fetched",
|
|
117
|
+
],
|
|
118
|
+
},
|
|
119
|
+
],
|
|
120
|
+
},
|
|
121
|
+
{
|
|
122
|
+
name: "pypi",
|
|
123
|
+
description: "PyPI package metadata",
|
|
124
|
+
keywords: "An exact package name.",
|
|
125
|
+
exampleKeywords: "requests",
|
|
126
|
+
sqlTables: [
|
|
127
|
+
{
|
|
128
|
+
name: "pypi",
|
|
129
|
+
columns: [
|
|
130
|
+
"id", "name", "version", "summary", "author", "license", "home_page", "requires_python", "keywords",
|
|
131
|
+
"body", "last_fetched",
|
|
132
|
+
],
|
|
133
|
+
},
|
|
134
|
+
],
|
|
135
|
+
},
|
|
136
|
+
{
|
|
137
|
+
name: "stackexchange",
|
|
138
|
+
description: "Stack Exchange Q&A",
|
|
139
|
+
keywords: "A programming question or topic (searches Stack Overflow).",
|
|
140
|
+
exampleKeywords: "python",
|
|
141
|
+
sqlTables: [
|
|
142
|
+
{
|
|
143
|
+
name: "stackexchange",
|
|
144
|
+
columns: [
|
|
145
|
+
"id", "question_id", "title", "link", "score", "tags", "owner", "site", "is_answered", "answer_count",
|
|
146
|
+
"view_count", "creation_date", "body", "last_fetched",
|
|
147
|
+
],
|
|
148
|
+
},
|
|
149
|
+
],
|
|
150
|
+
},
|
|
151
|
+
{
|
|
152
|
+
name: "news",
|
|
153
|
+
description: "Front-page news headlines from a historical archive (not live/current news)",
|
|
154
|
+
keywords: "A topic to find archived headlines about.",
|
|
155
|
+
exampleKeywords: "technology",
|
|
156
|
+
sqlTables: [],
|
|
157
|
+
},
|
|
17
158
|
];
|
|
159
|
+
export const DATAVILLE_SOURCES = DATAVILLE_SOURCE_DETAILS.map(({ name, description }) => ({
|
|
160
|
+
name,
|
|
161
|
+
description,
|
|
162
|
+
}));
|
|
163
|
+
export function findSource(name) {
|
|
164
|
+
const wanted = name.trim().toLowerCase();
|
|
165
|
+
return DATAVILLE_SOURCE_DETAILS.find((s) => s.name === wanted);
|
|
166
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@dataville/dataville-mcp",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.7",
|
|
4
4
|
"mcpName": "com.dataville/dataville-mcp",
|
|
5
5
|
"description": "MCP server exposing Dataville's data source API as tools for MCP clients",
|
|
6
6
|
"private": false,
|
|
@@ -27,12 +27,12 @@
|
|
|
27
27
|
"test:integration": "tsx --test src/__tests__/*.itest.ts"
|
|
28
28
|
},
|
|
29
29
|
"dependencies": {
|
|
30
|
-
"@modelcontextprotocol/sdk": "^1.
|
|
31
|
-
"zod": "^
|
|
30
|
+
"@modelcontextprotocol/sdk": "^1.32.1",
|
|
31
|
+
"zod": "^4.6.5"
|
|
32
32
|
},
|
|
33
33
|
"devDependencies": {
|
|
34
|
-
"@types/node": "^
|
|
34
|
+
"@types/node": "^26.6.4",
|
|
35
35
|
"tsx": "^4.19.0",
|
|
36
|
-
"typescript": "^
|
|
36
|
+
"typescript": "^7.0.2"
|
|
37
37
|
}
|
|
38
38
|
}
|