subgraph-registry-mcp 0.9.6 → 0.9.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/data/openapi.json +1 -1
- package/openapi.yaml +1 -1
- package/package.json +1 -1
- package/src/index.js +106 -4
package/data/openapi.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"info": {
|
|
4
4
|
"title": "Subgraph Registry",
|
|
5
5
|
"description": "Agent-friendly subgraph discovery on The Graph Network. 15,330 classified subgraphs with semantic search, reliability scoring, 30-day query volume and schema-evolution tracking. Discovery only: returns subgraph ids and starter queries, which you run with a Graph Studio API key (Authorization: Bearer) or over x402 ($0.01 USDC on Base, no key).",
|
|
6
|
-
"version": "0.9.
|
|
6
|
+
"version": "0.9.7",
|
|
7
7
|
"license": {
|
|
8
8
|
"name": "MIT"
|
|
9
9
|
},
|
package/openapi.yaml
CHANGED
|
@@ -5,7 +5,7 @@ openapi: "3.1.0"
|
|
|
5
5
|
info:
|
|
6
6
|
title: "Subgraph Registry"
|
|
7
7
|
description: "Agent-friendly subgraph discovery on The Graph Network. 15,330 classified subgraphs with semantic search, reliability scoring, 30-day query volume and schema-evolution tracking. Discovery only: returns subgraph ids and starter queries, which you run with a Graph Studio API key (Authorization: Bearer) or over x402 ($0.01 USDC on Base, no key)."
|
|
8
|
-
version: "0.9.
|
|
8
|
+
version: "0.9.7"
|
|
9
9
|
license:
|
|
10
10
|
name: "MIT"
|
|
11
11
|
contact:
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "subgraph-registry-mcp",
|
|
3
|
-
"version": "0.9.
|
|
3
|
+
"version": "0.9.7",
|
|
4
4
|
"mcpName": "io.github.PaulieB14/subgraph-registry-mcp",
|
|
5
5
|
"description": "MCP server for agent-friendly subgraph discovery on The Graph Network. 15,330 classified subgraphs with reliability scoring, 30-day query volume, semantic search and protocol classification. Discovery only — returns subgraph ids and ready-to-run queries for you to run with a Graph Studio key or over x402.",
|
|
6
6
|
"type": "module",
|
package/src/index.js
CHANGED
|
@@ -256,6 +256,27 @@ function normalizeNetwork(name) {
|
|
|
256
256
|
return NETWORK_ALIASES[k] || k;
|
|
257
257
|
}
|
|
258
258
|
|
|
259
|
+
// Networks the corpus actually contains, so a chain word in a goal can be told
|
|
260
|
+
// apart from a protocol that happens to share a chain's name. Built once from
|
|
261
|
+
// the DB rather than hardcoded, so a new chain in a re-crawl works immediately.
|
|
262
|
+
let _knownNetworks = null;
|
|
263
|
+
const KNOWN_NETWORKS = {
|
|
264
|
+
has(name) {
|
|
265
|
+
if (!name) return false;
|
|
266
|
+
if (_knownNetworks === null) {
|
|
267
|
+
try {
|
|
268
|
+
_knownNetworks = new Set(
|
|
269
|
+
getDb().prepare("SELECT DISTINCT network FROM subgraphs WHERE network IS NOT NULL")
|
|
270
|
+
.all().map((r) => r.network),
|
|
271
|
+
);
|
|
272
|
+
} catch {
|
|
273
|
+
_knownNetworks = new Set();
|
|
274
|
+
}
|
|
275
|
+
}
|
|
276
|
+
return _knownNetworks.has(name);
|
|
277
|
+
},
|
|
278
|
+
};
|
|
279
|
+
|
|
259
280
|
// ── Testnets ───────────────────────────────────────────────
|
|
260
281
|
// 723 of the 5,425 served, non-denied subgraphs (13.3%) are on testnets, and
|
|
261
282
|
// they compete directly with production because a testnet deployment's text is
|
|
@@ -305,12 +326,26 @@ function shouldExcludeTestnets({ include_testnets, network }) {
|
|
|
305
326
|
// tokens through.
|
|
306
327
|
const VERSION_TOKEN_RE = /^v\d+$/;
|
|
307
328
|
|
|
329
|
+
// Function words carry no signal about WHICH subgraph is wanted, and because
|
|
330
|
+
// matching is substring-based they actively mislead: "reputation scores for
|
|
331
|
+
// onchain agents" scored forsage-x2-prod above agent0, because "for" is inside
|
|
332
|
+
// "forsage" and a display-name hit is worth 4 while agent0's two real matches
|
|
333
|
+
// ("reputation", "agents", in the description) were worth 1 each. Dropping them
|
|
334
|
+
// costs nothing — no one distinguishes two subgraphs by the word "the".
|
|
335
|
+
const STOPWORDS = new Set([
|
|
336
|
+
"the", "and", "for", "with", "from", "that", "this", "are", "was", "were",
|
|
337
|
+
"get", "all", "any", "how", "what", "which", "who", "into", "onto", "over",
|
|
338
|
+
"per", "via", "out", "its", "their", "your", "our", "his", "her",
|
|
339
|
+
"show", "give", "find", "list", "want", "need", "using", "use", "used",
|
|
340
|
+
"data", "info", "about", "some", "more", "most", "have", "has", "had",
|
|
341
|
+
]);
|
|
342
|
+
|
|
308
343
|
function queryTerms(query) {
|
|
309
344
|
return query
|
|
310
345
|
.trim()
|
|
311
346
|
.toLowerCase()
|
|
312
347
|
.split(/\s+/)
|
|
313
|
-
.filter((w) => w.length > 2 || VERSION_TOKEN_RE.test(w))
|
|
348
|
+
.filter((w) => (w.length > 2 || VERSION_TOKEN_RE.test(w)) && !STOPWORDS.has(w))
|
|
314
349
|
.slice(0, 5);
|
|
315
350
|
}
|
|
316
351
|
|
|
@@ -662,8 +697,40 @@ function recommendSubgraph({ goal, chain = "" }) {
|
|
|
662
697
|
// actually asked.
|
|
663
698
|
const scoreParts = [];
|
|
664
699
|
const scoreParams = [];
|
|
700
|
+
let inferredChain = null;
|
|
701
|
+
|
|
702
|
+
// A chain named in the goal is a chain, not part of a protocol's name.
|
|
703
|
+
//
|
|
704
|
+
// "lido staking on ethereum" scored Lido Ethereum (2,644 queries/30d) above
|
|
705
|
+
// Lido (4,692,414) because "ethereum" matched Lido Ethereum's DISPLAY NAME
|
|
706
|
+
// for +4, and the ordering is lexicographic — goal_score first, reliability
|
|
707
|
+
// only as a tie-break — so a 0.88-vs-0.63 reliability gap and a 1,775x volume
|
|
708
|
+
// gap never got a vote. The subgraph won for having the chain in its title.
|
|
709
|
+
//
|
|
710
|
+
// The chain already has its own column and its own normalizer. Pull chain
|
|
711
|
+
// words out of the scoring terms and use them the way the `chain` parameter
|
|
712
|
+
// is used, so "on ethereum" narrows the network instead of flattering any
|
|
713
|
+
// subgraph that happens to be called something-Ethereum.
|
|
714
|
+
const allWords = queryTerms(goalLower);
|
|
715
|
+
const chainWords = [];
|
|
716
|
+
const words = [];
|
|
717
|
+
for (const w of allWords) {
|
|
718
|
+
const canonical = normalizeNetwork(w);
|
|
719
|
+
// Only treat it as a chain if it resolves to a network the corpus has —
|
|
720
|
+
// otherwise a protocol genuinely called "Base" or "Mode" would vanish.
|
|
721
|
+
if (canonical !== w || KNOWN_NETWORKS.has(canonical)) chainWords.push(canonical);
|
|
722
|
+
else words.push(w);
|
|
723
|
+
}
|
|
724
|
+
// An explicit `chain` argument always wins over one inferred from prose.
|
|
725
|
+
if (!chain && chainWords.length === 1) {
|
|
726
|
+
conditions.push("network = ?");
|
|
727
|
+
params.push(chainWords[0]);
|
|
728
|
+
inferredChain = chainWords[0];
|
|
729
|
+
} else if (chainWords.length) {
|
|
730
|
+
// Ambiguous or already-specified: keep them as weak text signal only.
|
|
731
|
+
words.push(...chainWords);
|
|
732
|
+
}
|
|
665
733
|
|
|
666
|
-
const words = queryTerms(goalLower);
|
|
667
734
|
if (words.length) {
|
|
668
735
|
const textConds = words.map(() => "(display_name LIKE ? OR description LIKE ? OR auto_description LIKE ?)");
|
|
669
736
|
scoreParts.push(
|
|
@@ -694,11 +761,43 @@ function recommendSubgraph({ goal, chain = "" }) {
|
|
|
694
761
|
FROM subgraphs
|
|
695
762
|
${where}
|
|
696
763
|
ORDER BY goal_score DESC, reliability_score DESC
|
|
697
|
-
LIMIT
|
|
764
|
+
LIMIT 60
|
|
698
765
|
`;
|
|
699
766
|
|
|
700
767
|
// SELECT-clause params bind before WHERE-clause params.
|
|
701
|
-
|
|
768
|
+
let rows = getDb().prepare(sql).all(...scoreParams, ...params);
|
|
769
|
+
|
|
770
|
+
// Re-rank on WORD boundaries, which SQL LIKE cannot express.
|
|
771
|
+
//
|
|
772
|
+
// The SQL score treats any substring of the display name as a name hit, so
|
|
773
|
+
// "scores" scored scoresquare-base at 4 and "for" scored forsage-x2-prod at
|
|
774
|
+
// 4, both beating agent0's two genuine description matches at 1 apiece. A
|
|
775
|
+
// term that appears as a whole word in the name is a real signal; a term that
|
|
776
|
+
// merely happens to be a prefix of a longer word is not, and was outranking
|
|
777
|
+
// it 4 to 1.
|
|
778
|
+
//
|
|
779
|
+
// Done here rather than in SQL because expressing "\bterm\b" in LIKE needs
|
|
780
|
+
// half a dozen OR-ed patterns per word for space, hyphen and underscore
|
|
781
|
+
// delimiters. The SQL score stays as the recall net (LIMIT 60); this decides
|
|
782
|
+
// the order of what it caught.
|
|
783
|
+
if (words.length) {
|
|
784
|
+
const bounded = words.map((w) => new RegExp("(^|[^a-z0-9])" + w.replace(/[.*+?^${}()|[\]\\]/g, "\\$&") + "([^a-z0-9]|$)", "i"));
|
|
785
|
+
const rescore = (r) => {
|
|
786
|
+
const name = r.display_name || "";
|
|
787
|
+
const text = `${r.description || ""} ${r.auto_description || ""}`;
|
|
788
|
+
let score = 0;
|
|
789
|
+
bounded.forEach((re, i) => {
|
|
790
|
+
if (re.test(name)) score += 6; // whole word in the name
|
|
791
|
+
else if (name.toLowerCase().includes(words[i])) score += 1; // incidental substring
|
|
792
|
+
if (re.test(text)) score += 2; // whole word in the description
|
|
793
|
+
});
|
|
794
|
+
return score;
|
|
795
|
+
};
|
|
796
|
+
rows = rows
|
|
797
|
+
.map((r) => ({ r, s: rescore(r) }))
|
|
798
|
+
.sort((a, b) => b.s - a.s || (b.r.reliability_score || 0) - (a.r.reliability_score || 0))
|
|
799
|
+
.map((x) => x.r);
|
|
800
|
+
}
|
|
702
801
|
// De-dup first so we batch the stability lookup over the trimmed set.
|
|
703
802
|
const seenIpfs = new Set();
|
|
704
803
|
const keep = [];
|
|
@@ -745,6 +844,9 @@ function recommendSubgraph({ goal, chain = "" }) {
|
|
|
745
844
|
goal,
|
|
746
845
|
chain_filter: chain || null,
|
|
747
846
|
inferred_domain: domains.length ? domains : null,
|
|
847
|
+
// Surfaced so a caller can see that "on ethereum" in their goal became a
|
|
848
|
+
// network filter, and correct it if that was not what they meant.
|
|
849
|
+
inferred_chain: inferredChain,
|
|
748
850
|
inferred_protocol_type: ptypes.length ? ptypes : null,
|
|
749
851
|
total_matches: recommendations.length,
|
|
750
852
|
recommendations,
|