subgraph-registry-mcp 0.9.6 → 0.9.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/data/openapi.json CHANGED
@@ -3,7 +3,7 @@
3
3
  "info": {
4
4
  "title": "Subgraph Registry",
5
5
  "description": "Agent-friendly subgraph discovery on The Graph Network. 15,330 classified subgraphs with semantic search, reliability scoring, 30-day query volume and schema-evolution tracking. Discovery only: returns subgraph ids and starter queries, which you run with a Graph Studio API key (Authorization: Bearer) or over x402 ($0.01 USDC on Base, no key).",
6
- "version": "0.9.6",
6
+ "version": "0.9.7",
7
7
  "license": {
8
8
  "name": "MIT"
9
9
  },
package/openapi.yaml CHANGED
@@ -5,7 +5,7 @@ openapi: "3.1.0"
5
5
  info:
6
6
  title: "Subgraph Registry"
7
7
  description: "Agent-friendly subgraph discovery on The Graph Network. 15,330 classified subgraphs with semantic search, reliability scoring, 30-day query volume and schema-evolution tracking. Discovery only: returns subgraph ids and starter queries, which you run with a Graph Studio API key (Authorization: Bearer) or over x402 ($0.01 USDC on Base, no key)."
8
- version: "0.9.6"
8
+ version: "0.9.7"
9
9
  license:
10
10
  name: "MIT"
11
11
  contact:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "subgraph-registry-mcp",
3
- "version": "0.9.6",
3
+ "version": "0.9.7",
4
4
  "mcpName": "io.github.PaulieB14/subgraph-registry-mcp",
5
5
  "description": "MCP server for agent-friendly subgraph discovery on The Graph Network. 15,330 classified subgraphs with reliability scoring, 30-day query volume, semantic search and protocol classification. Discovery only — returns subgraph ids and ready-to-run queries for you to run with a Graph Studio key or over x402.",
6
6
  "type": "module",
package/src/index.js CHANGED
@@ -256,6 +256,27 @@ function normalizeNetwork(name) {
256
256
  return NETWORK_ALIASES[k] || k;
257
257
  }
258
258
 
259
+ // Networks the corpus actually contains, so a chain word in a goal can be told
260
+ // apart from a protocol that happens to share a chain's name. Built once from
261
+ // the DB rather than hardcoded, so a new chain in a re-crawl works immediately.
262
+ let _knownNetworks = null;
263
+ const KNOWN_NETWORKS = {
264
+ has(name) {
265
+ if (!name) return false;
266
+ if (_knownNetworks === null) {
267
+ try {
268
+ _knownNetworks = new Set(
269
+ getDb().prepare("SELECT DISTINCT network FROM subgraphs WHERE network IS NOT NULL")
270
+ .all().map((r) => r.network),
271
+ );
272
+ } catch {
273
+ _knownNetworks = new Set();
274
+ }
275
+ }
276
+ return _knownNetworks.has(name);
277
+ },
278
+ };
279
+
259
280
  // ── Testnets ───────────────────────────────────────────────
260
281
  // 723 of the 5,425 served, non-denied subgraphs (13.3%) are on testnets, and
261
282
  // they compete directly with production because a testnet deployment's text is
@@ -305,12 +326,26 @@ function shouldExcludeTestnets({ include_testnets, network }) {
305
326
  // tokens through.
306
327
  const VERSION_TOKEN_RE = /^v\d+$/;
307
328
 
329
+ // Function words carry no signal about WHICH subgraph is wanted, and because
330
+ // matching is substring-based they actively mislead: "reputation scores for
331
+ // onchain agents" scored forsage-x2-prod above agent0, because "for" is inside
332
+ // "forsage" and a display-name hit is worth 4 while agent0's two real matches
333
+ // ("reputation", "agents", in the description) were worth 1 each. Dropping them
334
+ // costs nothing — no one distinguishes two subgraphs by the word "the".
335
+ const STOPWORDS = new Set([
336
+ "the", "and", "for", "with", "from", "that", "this", "are", "was", "were",
337
+ "get", "all", "any", "how", "what", "which", "who", "into", "onto", "over",
338
+ "per", "via", "out", "its", "their", "your", "our", "his", "her",
339
+ "show", "give", "find", "list", "want", "need", "using", "use", "used",
340
+ "data", "info", "about", "some", "more", "most", "have", "has", "had",
341
+ ]);
342
+
308
343
  function queryTerms(query) {
309
344
  return query
310
345
  .trim()
311
346
  .toLowerCase()
312
347
  .split(/\s+/)
313
- .filter((w) => w.length > 2 || VERSION_TOKEN_RE.test(w))
348
+ .filter((w) => (w.length > 2 || VERSION_TOKEN_RE.test(w)) && !STOPWORDS.has(w))
314
349
  .slice(0, 5);
315
350
  }
316
351
 
@@ -662,8 +697,40 @@ function recommendSubgraph({ goal, chain = "" }) {
662
697
  // actually asked.
663
698
  const scoreParts = [];
664
699
  const scoreParams = [];
700
+ let inferredChain = null;
701
+
702
+ // A chain named in the goal is a chain, not part of a protocol's name.
703
+ //
704
+ // "lido staking on ethereum" scored Lido Ethereum (2,644 queries/30d) above
705
+ // Lido (4,692,414) because "ethereum" matched Lido Ethereum's DISPLAY NAME
706
+ // for +4, and the ordering is lexicographic — goal_score first, reliability
707
+ // only as a tie-break — so a 0.88-vs-0.63 reliability gap and a 1,775x volume
708
+ // gap never got a vote. The subgraph won for having the chain in its title.
709
+ //
710
+ // The chain already has its own column and its own normalizer. Pull chain
711
+ // words out of the scoring terms and use them the way the `chain` parameter
712
+ // is used, so "on ethereum" narrows the network instead of flattering any
713
+ // subgraph that happens to be called something-Ethereum.
714
+ const allWords = queryTerms(goalLower);
715
+ const chainWords = [];
716
+ const words = [];
717
+ for (const w of allWords) {
718
+ const canonical = normalizeNetwork(w);
719
+ // Only treat it as a chain if it resolves to a network the corpus has —
720
+ // otherwise a protocol genuinely called "Base" or "Mode" would vanish.
721
+ if (canonical !== w || KNOWN_NETWORKS.has(canonical)) chainWords.push(canonical);
722
+ else words.push(w);
723
+ }
724
+ // An explicit `chain` argument always wins over one inferred from prose.
725
+ if (!chain && chainWords.length === 1) {
726
+ conditions.push("network = ?");
727
+ params.push(chainWords[0]);
728
+ inferredChain = chainWords[0];
729
+ } else if (chainWords.length) {
730
+ // Ambiguous or already-specified: keep them as weak text signal only.
731
+ words.push(...chainWords);
732
+ }
665
733
 
666
- const words = queryTerms(goalLower);
667
734
  if (words.length) {
668
735
  const textConds = words.map(() => "(display_name LIKE ? OR description LIKE ? OR auto_description LIKE ?)");
669
736
  scoreParts.push(
@@ -694,11 +761,43 @@ function recommendSubgraph({ goal, chain = "" }) {
694
761
  FROM subgraphs
695
762
  ${where}
696
763
  ORDER BY goal_score DESC, reliability_score DESC
697
- LIMIT 15
764
+ LIMIT 60
698
765
  `;
699
766
 
700
767
  // SELECT-clause params bind before WHERE-clause params.
701
- const rows = getDb().prepare(sql).all(...scoreParams, ...params);
768
+ let rows = getDb().prepare(sql).all(...scoreParams, ...params);
769
+
770
+ // Re-rank on WORD boundaries, which SQL LIKE cannot express.
771
+ //
772
+ // The SQL score treats any substring of the display name as a name hit, so
773
+ // "scores" scored scoresquare-base at 4 and "for" scored forsage-x2-prod at
774
+ // 4, both beating agent0's two genuine description matches at 1 apiece. A
775
+ // term that appears as a whole word in the name is a real signal; a term that
776
+ // merely happens to be a prefix of a longer word is not, and was outranking
777
+ // it 4 to 1.
778
+ //
779
+ // Done here rather than in SQL because expressing "\bterm\b" in LIKE needs
780
+ // half a dozen OR-ed patterns per word for space, hyphen and underscore
781
+ // delimiters. The SQL score stays as the recall net (LIMIT 60); this decides
782
+ // the order of what it caught.
783
+ if (words.length) {
784
+ const bounded = words.map((w) => new RegExp("(^|[^a-z0-9])" + w.replace(/[.*+?^${}()|[\]\\]/g, "\\$&") + "([^a-z0-9]|$)", "i"));
785
+ const rescore = (r) => {
786
+ const name = r.display_name || "";
787
+ const text = `${r.description || ""} ${r.auto_description || ""}`;
788
+ let score = 0;
789
+ bounded.forEach((re, i) => {
790
+ if (re.test(name)) score += 6; // whole word in the name
791
+ else if (name.toLowerCase().includes(words[i])) score += 1; // incidental substring
792
+ if (re.test(text)) score += 2; // whole word in the description
793
+ });
794
+ return score;
795
+ };
796
+ rows = rows
797
+ .map((r) => ({ r, s: rescore(r) }))
798
+ .sort((a, b) => b.s - a.s || (b.r.reliability_score || 0) - (a.r.reliability_score || 0))
799
+ .map((x) => x.r);
800
+ }
702
801
  // De-dup first so we batch the stability lookup over the trimmed set.
703
802
  const seenIpfs = new Set();
704
803
  const keep = [];
@@ -745,6 +844,9 @@ function recommendSubgraph({ goal, chain = "" }) {
745
844
  goal,
746
845
  chain_filter: chain || null,
747
846
  inferred_domain: domains.length ? domains : null,
847
+ // Surfaced so a caller can see that "on ethereum" in their goal became a
848
+ // network filter, and correct it if that was not what they meant.
849
+ inferred_chain: inferredChain,
748
850
  inferred_protocol_type: ptypes.length ? ptypes : null,
749
851
  total_matches: recommendations.length,
750
852
  recommendations,