subgraph-registry-mcp 0.9.0 → 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -0
- package/data/registry.db +0 -0
- package/package.json +1 -1
- package/src/index.js +165 -30
package/README.md
CHANGED
|
@@ -188,6 +188,25 @@ in the main list and the 40-day-old Monad perps subgraph under `emerging`.
|
|
|
188
188
|
so it is already age-neutral — it carries the `maturity` labels but no
|
|
189
189
|
`emerging` list, because a three-week-old subgraph can top it on merit.
|
|
190
190
|
|
|
191
|
+
### Ranking
|
|
192
|
+
|
|
193
|
+
Three tools rank, and each ranks differently on purpose:
|
|
194
|
+
|
|
195
|
+
- **`search_subgraphs`** — orders by how many of your query terms matched, then
|
|
196
|
+
by reliability. OR-ing the terms and ordering on reliability alone meant a
|
|
197
|
+
popular subgraph matching one incidental word beat a precise match on all
|
|
198
|
+
three, so being *more* specific returned worse answers. Version tokens
|
|
199
|
+
(`v2`, `v3`, `v4`) are kept rather than dropped as too short.
|
|
200
|
+
- **`semantic_search_subgraphs`** — orders by `semantic_score × (0.5 + 0.5 ×
|
|
201
|
+
reliability)`. Pure cosine put testnets first, since their text is nearly
|
|
202
|
+
identical to mainnet's. The 0.5 floor keeps new subgraphs competitive.
|
|
203
|
+
- **`recommend_subgraph`** — infers domain and protocol type from the goal, but
|
|
204
|
+
as a *ranking bonus*, never a filter. As a filter, one bad keyword collapsed
|
|
205
|
+
the candidate pool to nothing.
|
|
206
|
+
|
|
207
|
+
Chain names are aliased, so `ethereum`, `arbitrum`, `polygon` and `bnb` resolve
|
|
208
|
+
to the corpus values `mainnet`, `arbitrum-one`, `matic` and `bsc`.
|
|
209
|
+
|
|
191
210
|
### Denied deployments
|
|
192
211
|
|
|
193
212
|
Curation-denied deployments (`deniedAt > 0` — denied indexing rewards, usually
|
package/data/registry.db
CHANGED
|
Binary file
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "subgraph-registry-mcp",
|
|
3
|
-
"version": "0.9.
|
|
3
|
+
"version": "0.9.1",
|
|
4
4
|
"mcpName": "io.github.PaulieB14/subgraph-registry-mcp",
|
|
5
5
|
"description": "MCP server for agent-friendly subgraph discovery on The Graph Network. 15,330 classified subgraphs with x402 query URLs ($0.01 USDC on Base, no API key required), reliability scoring, and protocol classification.",
|
|
6
6
|
"type": "module",
|
package/src/index.js
CHANGED
|
@@ -65,7 +65,7 @@ const GITHUB_DB_URL =
|
|
|
65
65
|
// 3. Paste the new hash here and bump package.json version
|
|
66
66
|
// 4. Update SKILL.md "Verifying the registry" section
|
|
67
67
|
const EXPECTED_DB_SHA256 =
|
|
68
|
-
"
|
|
68
|
+
"425b7a5bde8f61d8ae2f26ea6e201ffd3308c3328a0547fb0d29530222eba0d2";
|
|
69
69
|
// Skip-verification escape hatch (set to "1" only if you're rebuilding the DB
|
|
70
70
|
// locally and know what you're doing — never set in agent-runtime defaults).
|
|
71
71
|
const SKIP_VERIFY = process.env.SUBGRAPH_REGISTRY_SKIP_VERIFY === "1";
|
|
@@ -177,6 +177,70 @@ function getDb() {
|
|
|
177
177
|
return db;
|
|
178
178
|
}
|
|
179
179
|
|
|
180
|
+
// ── Network aliases ────────────────────────────────────────
|
|
181
|
+
// The corpus stores graph-node's chain IDs (`mainnet`, `arbitrum-one`,
|
|
182
|
+
// `matic`), but every human and every model says "ethereum", "arbitrum",
|
|
183
|
+
// "polygon" — and so does our own auto_description, which prints the pretty
|
|
184
|
+
// name from classifier.py NETWORK_NAMES. So an agent reads "Ethereum" in a
|
|
185
|
+
// description, passes network:"ethereum" back, and gets zero results with no
|
|
186
|
+
// error. SKILL.md made it worse by documenting `ethereum, arbitrum, base` as
|
|
187
|
+
// the example values; two of those three matched nothing.
|
|
188
|
+
// mainnet/bsc/arbitrum-one/matic alone are ~45% of the corpus.
|
|
189
|
+
const NETWORK_ALIASES = {
|
|
190
|
+
ethereum: "mainnet",
|
|
191
|
+
eth: "mainnet",
|
|
192
|
+
"ethereum-mainnet": "mainnet",
|
|
193
|
+
arbitrum: "arbitrum-one",
|
|
194
|
+
"arbitrum one": "arbitrum-one",
|
|
195
|
+
arb: "arbitrum-one",
|
|
196
|
+
polygon: "matic",
|
|
197
|
+
"polygon-pos": "matic",
|
|
198
|
+
bnb: "bsc",
|
|
199
|
+
"bnb-chain": "bsc",
|
|
200
|
+
"binance-smart-chain": "bsc",
|
|
201
|
+
binance: "bsc",
|
|
202
|
+
op: "optimism",
|
|
203
|
+
"optimism-mainnet": "optimism",
|
|
204
|
+
avax: "avalanche",
|
|
205
|
+
"avalanche-c-chain": "avalanche",
|
|
206
|
+
xdai: "gnosis",
|
|
207
|
+
zksync: "zksync-era",
|
|
208
|
+
"zksync era": "zksync-era",
|
|
209
|
+
blast: "blast-mainnet",
|
|
210
|
+
"polygon-zk": "polygon-zkevm",
|
|
211
|
+
near: "near-mainnet",
|
|
212
|
+
mode: "mode-mainnet",
|
|
213
|
+
sei: "sei-mainnet",
|
|
214
|
+
ftm: "fantom",
|
|
215
|
+
};
|
|
216
|
+
|
|
217
|
+
// Normalize a caller-supplied chain name to the value stored in the corpus.
|
|
218
|
+
// Unknown values pass through untouched so a legitimate new chain still works.
|
|
219
|
+
function normalizeNetwork(name) {
|
|
220
|
+
if (!name || typeof name !== "string") return name;
|
|
221
|
+
const k = name.trim().toLowerCase();
|
|
222
|
+
return NETWORK_ALIASES[k] || k;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
// Tokenize a free-text query into searchable terms.
|
|
226
|
+
//
|
|
227
|
+
// The old inline version was `.filter((w) => w.length > 2)`, which silently
|
|
228
|
+
// dropped every protocol version token — "v2", "v3", "v4" are all two chars.
|
|
229
|
+
// That made "uniswap v3" byte-identical to "uniswap", so the single most
|
|
230
|
+
// natural way to disambiguate the largest protocol family in the corpus did
|
|
231
|
+
// nothing at all. Keep the length floor for noise words, but let version
|
|
232
|
+
// tokens through.
|
|
233
|
+
const VERSION_TOKEN_RE = /^v\d+$/;
|
|
234
|
+
|
|
235
|
+
function queryTerms(query) {
|
|
236
|
+
return query
|
|
237
|
+
.trim()
|
|
238
|
+
.toLowerCase()
|
|
239
|
+
.split(/\s+/)
|
|
240
|
+
.filter((w) => w.length > 2 || VERSION_TOKEN_RE.test(w))
|
|
241
|
+
.slice(0, 5);
|
|
242
|
+
}
|
|
243
|
+
|
|
180
244
|
// ── Maturity / cold-start handling ─────────────────────────
|
|
181
245
|
// reliability_score is built from four CUMULATIVE inputs (curation signal,
|
|
182
246
|
// indexer stake, lifetime query fees, 30d volume — see _reliability_score in
|
|
@@ -297,7 +361,7 @@ function searchSubgraphs({
|
|
|
297
361
|
}
|
|
298
362
|
if (network) {
|
|
299
363
|
conditions.push("network = ?");
|
|
300
|
-
params.push(network);
|
|
364
|
+
params.push(normalizeNetwork(network));
|
|
301
365
|
}
|
|
302
366
|
if (protocol_type) {
|
|
303
367
|
conditions.push("protocol_type = ?");
|
|
@@ -311,9 +375,27 @@ function searchSubgraphs({
|
|
|
311
375
|
conditions.push("reliability_score >= ?");
|
|
312
376
|
params.push(min_reliability);
|
|
313
377
|
}
|
|
378
|
+
// Terms are OR'd for recall, then RANKED by how many of them matched.
|
|
379
|
+
//
|
|
380
|
+
// Before this, a multi-word query OR'd its terms and ordered the result by
|
|
381
|
+
// reliability alone, so a high-reliability subgraph matching ONE incidental
|
|
382
|
+
// word beat a lower-reliability one matching all three. The effect was that
|
|
383
|
+
// being more specific made the answer worse: "aave lending arbitrum"
|
|
384
|
+
// returned uniswap-v3-arbitrum, Arbitrum Minimal, camelot-amm-v3 and Graph
|
|
385
|
+
// TAP — not one Aave subgraph — while the bare query "aave" was correct.
|
|
386
|
+
//
|
|
387
|
+
// Keep the OR (dropping to AND would kill recall on descriptions that
|
|
388
|
+
// phrase things differently) and let matched_terms break the tie first.
|
|
389
|
+
const matchParams = [];
|
|
390
|
+
let matchExpr = "0";
|
|
314
391
|
if (query) {
|
|
315
|
-
const words = query
|
|
392
|
+
const words = queryTerms(query);
|
|
316
393
|
if (words.length) {
|
|
394
|
+
matchExpr = words
|
|
395
|
+
.map(() => "(CASE WHEN (display_name LIKE ? OR description LIKE ? OR auto_description LIKE ?) THEN 1 ELSE 0 END)")
|
|
396
|
+
.join(" + ");
|
|
397
|
+
words.forEach((w) => matchParams.push(`%${w}%`, `%${w}%`, `%${w}%`));
|
|
398
|
+
|
|
317
399
|
const wordConds = words.map(() => "(display_name LIKE ? OR description LIKE ? OR auto_description LIKE ?)");
|
|
318
400
|
words.forEach((w) => params.push(`%${w}%`, `%${w}%`, `%${w}%`));
|
|
319
401
|
conditions.push(`(${wordConds.join(" OR ")})`);
|
|
@@ -327,18 +409,20 @@ function searchSubgraphs({
|
|
|
327
409
|
// Over-fetch to allow dedup by IPFS hash (same deployment, different subgraph IDs)
|
|
328
410
|
const fetchLimit = limit * 3;
|
|
329
411
|
const sql = `
|
|
330
|
-
SELECT ${SEARCH_COLS}
|
|
412
|
+
SELECT ${SEARCH_COLS}, (${matchExpr}) AS matched_terms
|
|
331
413
|
FROM subgraphs
|
|
332
414
|
${where}
|
|
333
|
-
ORDER BY reliability_score DESC
|
|
415
|
+
ORDER BY matched_terms DESC, reliability_score DESC
|
|
334
416
|
LIMIT ?
|
|
335
417
|
`;
|
|
336
418
|
// Snapshot the filter params BEFORE the LIMIT is appended — the emerging
|
|
337
419
|
// companion query reuses the same WHERE and must not inherit this LIMIT.
|
|
338
420
|
const filterParams = [...params];
|
|
339
|
-
params.push(fetchLimit);
|
|
340
421
|
|
|
341
|
-
|
|
422
|
+
// Positional binding order: the SELECT-clause scoring expression is bound
|
|
423
|
+
// before the WHERE clause, so matchParams must lead. filterParams stays
|
|
424
|
+
// WHERE-only, which is what the emerging companion query needs.
|
|
425
|
+
const rows = getDb().prepare(sql).all(...matchParams, ...filterParams, fetchLimit);
|
|
342
426
|
// Dedup by IPFS hash — keep highest reliability per deployment
|
|
343
427
|
const seenIpfs = new Set();
|
|
344
428
|
const results = [];
|
|
@@ -427,18 +511,31 @@ function recommendSubgraph({ goal, chain = "" }) {
|
|
|
427
511
|
lending: ["lend", "borrow", "loan", "collateral", "aave", "compound"],
|
|
428
512
|
bridge: ["bridge", "cross-chain"],
|
|
429
513
|
staking: ["stake", "validator", "delegation"],
|
|
430
|
-
|
|
514
|
+
// "call", "put" and "strike" are gone. They are ordinary English words —
|
|
515
|
+
// "reputation" contains "put", so the goal "reputation scores for onchain
|
|
516
|
+
// agents" inferred protocol_type ["options"] and returned the Polygon
|
|
517
|
+
// Optimistic Oracle. "option" alone is specific enough to keep.
|
|
518
|
+
options: ["option", "derivatives contract"],
|
|
431
519
|
perpetuals: ["perp", "perpetual", "leverage", "margin"],
|
|
432
520
|
governance: ["governance", "vote", "proposal"],
|
|
433
521
|
"name-service": ["ens", "name service", "domain name"],
|
|
434
522
|
"nft-marketplace": ["nft market", "opensea", "blur"],
|
|
435
523
|
};
|
|
436
524
|
|
|
525
|
+
// Match on word boundaries, not bare substrings. Boundaries alone would not
|
|
526
|
+
// have saved "reputation"/"put" if "put" had stayed in the list — hence the
|
|
527
|
+
// removals above and the demotion from filter to bonus below — but they do
|
|
528
|
+
// stop "smart contract" inferring nfts via "art", and "start"/"chart"/
|
|
529
|
+
// "party" doing the same.
|
|
530
|
+
const hitsGoal = (kws) =>
|
|
531
|
+
kws.some((k) =>
|
|
532
|
+
new RegExp(`\\b${k.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}\\b`).test(goalLower),
|
|
533
|
+
);
|
|
437
534
|
const domains = Object.entries(domainMap)
|
|
438
|
-
.filter(([, kws]) => kws
|
|
535
|
+
.filter(([, kws]) => hitsGoal(kws))
|
|
439
536
|
.map(([d]) => d);
|
|
440
537
|
const ptypes = Object.entries(typeMap)
|
|
441
|
-
.filter(([, kws]) => kws
|
|
538
|
+
.filter(([, kws]) => hitsGoal(kws))
|
|
442
539
|
.map(([t]) => t);
|
|
443
540
|
|
|
444
541
|
// recommend_subgraph exposes no include_* escape hatches — it answers "which
|
|
@@ -449,37 +546,56 @@ function recommendSubgraph({ goal, chain = "" }) {
|
|
|
449
546
|
|
|
450
547
|
if (chain) {
|
|
451
548
|
conditions.push("network = ?");
|
|
452
|
-
params.push(chain);
|
|
549
|
+
params.push(normalizeNetwork(chain));
|
|
550
|
+
}
|
|
551
|
+
// The inferred domain/protocol_type used to go into the WHERE clause, so a
|
|
552
|
+
// single bad substring collapsed the candidate pool instead of merely
|
|
553
|
+
// mis-ordering it: "tokens" contains "ens" and cut 5,479 candidates to 78
|
|
554
|
+
// with ENS on top, and chain:"arbitrum" returned total_matches 0 with no
|
|
555
|
+
// error at all. Inference is a guess about intent; a guess belongs in the
|
|
556
|
+
// ORDER BY, where being wrong costs a few positions, not every result.
|
|
557
|
+
//
|
|
558
|
+
// The text terms now ALWAYS constrain (they used to be skipped entirely
|
|
559
|
+
// whenever anything was inferred), so the pool stays tied to what was
|
|
560
|
+
// actually asked.
|
|
561
|
+
const scoreParts = [];
|
|
562
|
+
const scoreParams = [];
|
|
563
|
+
|
|
564
|
+
const words = queryTerms(goalLower);
|
|
565
|
+
if (words.length) {
|
|
566
|
+
const textConds = words.map(() => "(display_name LIKE ? OR description LIKE ? OR auto_description LIKE ?)");
|
|
567
|
+
scoreParts.push(
|
|
568
|
+
words
|
|
569
|
+
.map(() => "(CASE WHEN (display_name LIKE ? OR description LIKE ? OR auto_description LIKE ?) THEN 2 ELSE 0 END)")
|
|
570
|
+
.join(" + "),
|
|
571
|
+
);
|
|
572
|
+
words.forEach((w) => scoreParams.push(`%${w}%`, `%${w}%`, `%${w}%`));
|
|
573
|
+
words.forEach((w) => params.push(`%${w}%`, `%${w}%`, `%${w}%`));
|
|
574
|
+
conditions.push(`(${textConds.join(" OR ")})`);
|
|
453
575
|
}
|
|
454
576
|
if (domains.length) {
|
|
455
|
-
|
|
456
|
-
|
|
577
|
+
scoreParts.push(`(CASE WHEN domain IN (${domains.map(() => "?").join(",")}) THEN 1 ELSE 0 END)`);
|
|
578
|
+
scoreParams.push(...domains);
|
|
457
579
|
}
|
|
458
580
|
if (ptypes.length) {
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
}
|
|
462
|
-
|
|
463
|
-
if (!domains.length && !ptypes.length) {
|
|
464
|
-
const words = goalLower.split(/\s+/).filter((w) => w.length > 2).slice(0, 5);
|
|
465
|
-
if (words.length) {
|
|
466
|
-
const textConds = words.map(() => "(display_name LIKE ? OR description LIKE ?)");
|
|
467
|
-
words.forEach((w) => params.push(`%${w}%`, `%${w}%`));
|
|
468
|
-
conditions.push(`(${textConds.join(" OR ")})`);
|
|
469
|
-
}
|
|
581
|
+
scoreParts.push(`(CASE WHEN protocol_type IN (${ptypes.map(() => "?").join(",")}) THEN 1 ELSE 0 END)`);
|
|
582
|
+
scoreParams.push(...ptypes);
|
|
470
583
|
}
|
|
471
584
|
|
|
585
|
+
const goalScore = scoreParts.length ? scoreParts.join(" + ") : "0";
|
|
472
586
|
const where = `WHERE ${conditions.join(" AND ")}`;
|
|
473
587
|
const sql = `
|
|
474
588
|
SELECT id, display_name, description, auto_description, domain, protocol_type, network,
|
|
475
|
-
reliability_score, ipfs_hash, canonical_entities, active_allocation_count, example_query
|
|
589
|
+
reliability_score, ipfs_hash, canonical_entities, active_allocation_count, example_query,
|
|
590
|
+
(${goalScore}) AS goal_score
|
|
476
591
|
FROM subgraphs
|
|
477
592
|
${where}
|
|
478
|
-
ORDER BY reliability_score DESC
|
|
593
|
+
ORDER BY goal_score DESC, reliability_score DESC
|
|
479
594
|
LIMIT 15
|
|
480
595
|
`;
|
|
481
596
|
|
|
482
|
-
|
|
597
|
+
// SELECT-clause params bind before WHERE-clause params.
|
|
598
|
+
const rows = getDb().prepare(sql).all(...scoreParams, ...params);
|
|
483
599
|
// De-dup first so we batch the stability lookup over the trimmed set.
|
|
484
600
|
const seenIpfs = new Set();
|
|
485
601
|
const keep = [];
|
|
@@ -774,7 +890,7 @@ async function semanticSearchSubgraphs({
|
|
|
774
890
|
}
|
|
775
891
|
if (network) {
|
|
776
892
|
conditions.push("network = ?");
|
|
777
|
-
params.push(network);
|
|
893
|
+
params.push(normalizeNetwork(network));
|
|
778
894
|
}
|
|
779
895
|
if (protocol_type) {
|
|
780
896
|
conditions.push("protocol_type = ?");
|
|
@@ -806,7 +922,23 @@ async function semanticSearchSubgraphs({
|
|
|
806
922
|
if (score < min_score) continue;
|
|
807
923
|
scored.push({ row: r, score });
|
|
808
924
|
}
|
|
809
|
-
|
|
925
|
+
// Rank on similarity WEIGHTED by reliability, not similarity alone.
|
|
926
|
+
//
|
|
927
|
+
// Pure cosine made this the only tool in the package that ignored the
|
|
928
|
+
// registry's own quality signal, and testnets win on pure cosine because
|
|
929
|
+
// their text is near-identical to mainnet's. Observed: "ENS domain name
|
|
930
|
+
// registrations" put ENS Sepolia (58 queries/30d, reliability 0.2463) above
|
|
931
|
+
// ENS mainnet (34.8M queries/30d, reliability 0.9775) on a cosine margin of
|
|
932
|
+
// 0.0105 — a rounding error deciding between a toy and the real thing.
|
|
933
|
+
//
|
|
934
|
+
// The 0.5 floor is deliberate: reliability is itself age-biased (see the
|
|
935
|
+
// maturity block above), so a multiplier that ran to 0 would re-bury every
|
|
936
|
+
// new subgraph and undo the emerging work. At 0.5 + 0.5*r a brand-new
|
|
937
|
+
// subgraph keeps half its similarity and can still outrank an established
|
|
938
|
+
// one it genuinely beats on meaning, while a 0.01 cosine tie resolves
|
|
939
|
+
// toward the subgraph that is actually serving traffic.
|
|
940
|
+
const effective = (s) => s.score * (0.5 + 0.5 * (s.row.reliability_score || 0));
|
|
941
|
+
scored.sort((a, b) => effective(b) - effective(a));
|
|
810
942
|
|
|
811
943
|
const results = [];
|
|
812
944
|
for (const { row: r, score } of scored) {
|
|
@@ -887,6 +1019,7 @@ function getSchemaChanges({ subgraph_id, since_timestamp = 0 }) {
|
|
|
887
1019
|
`SELECT fingerprint, prev_fingerprint, detected_at, ipfs_hash
|
|
888
1020
|
FROM schema_history
|
|
889
1021
|
WHERE subgraph_id = ? AND detected_at >= ?
|
|
1022
|
+
AND prev_fingerprint IS NOT NULL
|
|
890
1023
|
ORDER BY detected_at DESC`,
|
|
891
1024
|
)
|
|
892
1025
|
.all(subgraph_id, since);
|
|
@@ -932,7 +1065,8 @@ function getSchemaStabilityFor(id) {
|
|
|
932
1065
|
const r = getDb()
|
|
933
1066
|
.prepare(
|
|
934
1067
|
"SELECT MAX(detected_at) AS schema_changed_at " +
|
|
935
|
-
"FROM schema_history WHERE subgraph_id = ?"
|
|
1068
|
+
"FROM schema_history WHERE subgraph_id = ? " +
|
|
1069
|
+
"AND prev_fingerprint IS NOT NULL",
|
|
936
1070
|
)
|
|
937
1071
|
.get(id);
|
|
938
1072
|
if (!r || r.schema_changed_at == null) {
|
|
@@ -958,6 +1092,7 @@ function getSchemaStabilityBatch(ids) {
|
|
|
958
1092
|
`SELECT subgraph_id, MAX(detected_at) AS schema_changed_at
|
|
959
1093
|
FROM schema_history
|
|
960
1094
|
WHERE subgraph_id IN (${placeholders})
|
|
1095
|
+
AND prev_fingerprint IS NOT NULL
|
|
961
1096
|
GROUP BY subgraph_id`,
|
|
962
1097
|
)
|
|
963
1098
|
.all(...ids);
|