@cliwant/mcp-sam-gov 1.11.0 → 1.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/README.md +4 -0
  2. package/dist/bls.d.ts.map +1 -1
  3. package/dist/bls.js +6 -1
  4. package/dist/bls.js.map +1 -1
  5. package/dist/bonfire.d.ts.map +1 -1
  6. package/dist/bonfire.js +12 -1
  7. package/dist/bonfire.js.map +1 -1
  8. package/dist/census-economic.d.ts.map +1 -1
  9. package/dist/census-economic.js +10 -1
  10. package/dist/census-economic.js.map +1 -1
  11. package/dist/ckan.d.ts.map +1 -1
  12. package/dist/ckan.js +17 -3
  13. package/dist/ckan.js.map +1 -1
  14. package/dist/clinicaltrials.d.ts.map +1 -1
  15. package/dist/clinicaltrials.js +23 -6
  16. package/dist/clinicaltrials.js.map +1 -1
  17. package/dist/fdic.js +7 -7
  18. package/dist/fdic.js.map +1 -1
  19. package/dist/fema.d.ts.map +1 -1
  20. package/dist/fema.js +13 -1
  21. package/dist/fema.js.map +1 -1
  22. package/dist/grants.d.ts.map +1 -1
  23. package/dist/grants.js +13 -3
  24. package/dist/grants.js.map +1 -1
  25. package/dist/gsa-csv.d.ts +9 -0
  26. package/dist/gsa-csv.d.ts.map +1 -1
  27. package/dist/gsa-csv.js +37 -8
  28. package/dist/gsa-csv.js.map +1 -1
  29. package/dist/nppes.d.ts.map +1 -1
  30. package/dist/nppes.js +17 -4
  31. package/dist/nppes.js.map +1 -1
  32. package/dist/nsf.d.ts.map +1 -1
  33. package/dist/nsf.js +9 -0
  34. package/dist/nsf.js.map +1 -1
  35. package/dist/openfda-device.d.ts.map +1 -1
  36. package/dist/openfda-device.js +8 -5
  37. package/dist/openfda-device.js.map +1 -1
  38. package/dist/openfda-drugsfda.d.ts.map +1 -1
  39. package/dist/openfda-drugsfda.js +8 -5
  40. package/dist/openfda-drugsfda.js.map +1 -1
  41. package/dist/openfda.d.ts +21 -0
  42. package/dist/openfda.d.ts.map +1 -1
  43. package/dist/openfda.js +38 -4
  44. package/dist/openfda.js.map +1 -1
  45. package/dist/opengov.d.ts.map +1 -1
  46. package/dist/opengov.js +8 -4
  47. package/dist/opengov.js.map +1 -1
  48. package/dist/server.d.ts +7 -0
  49. package/dist/server.d.ts.map +1 -1
  50. package/dist/server.js +46 -3
  51. package/dist/server.js.map +1 -1
  52. package/dist/socrata.d.ts.map +1 -1
  53. package/dist/socrata.js +14 -7
  54. package/dist/socrata.js.map +1 -1
  55. package/dist/usaspending.d.ts +2 -2
  56. package/dist/usaspending.d.ts.map +1 -1
  57. package/dist/usaspending.js +21 -10
  58. package/dist/usaspending.js.map +1 -1
  59. package/package.json +1 -1
  60. package/src/bls.ts +8 -1
  61. package/src/bonfire.ts +12 -1
  62. package/src/census-economic.ts +13 -2
  63. package/src/ckan.ts +19 -3
  64. package/src/clinicaltrials.ts +25 -8
  65. package/src/fdic.ts +7 -7
  66. package/src/fema.ts +15 -1
  67. package/src/grants.ts +15 -2
  68. package/src/gsa-csv.ts +38 -7
  69. package/src/nppes.ts +18 -4
  70. package/src/nsf.ts +10 -0
  71. package/src/openfda-device.ts +10 -4
  72. package/src/openfda-drugsfda.ts +10 -4
  73. package/src/openfda.ts +47 -5
  74. package/src/opengov.ts +8 -4
  75. package/src/server.ts +51 -3
  76. package/src/socrata.ts +14 -7
  77. package/src/usaspending.ts +21 -10
package/src/server.ts CHANGED
@@ -106,7 +106,7 @@ import { realpathSync } from "node:fs";
106
106
  const SERVER_NAME = "mcp-sam-gov";
107
107
  // Kept in lockstep with package.json / manifest.json / server.json.
108
108
  // Keep in sync with package.json "version" (asserted at release; see CHANGELOG).
109
- const SERVER_VERSION = "1.11.0";
109
+ const SERVER_VERSION = "1.12.0";
110
110
 
111
111
  // ─── Tool input schemas (Zod) ────────────────────────────────────
112
112
 
@@ -3759,7 +3759,7 @@ const CensusBusinessPatternsInput = z.object({
3759
3759
  .regex(/^\d{4}$/)
3760
3760
  .optional()
3761
3761
  .describe(
3762
- "The CBP data year (default '2022', the latest confirmed vintage). Validated ^\\d{4}$ (it rides in the request path).",
3762
+ "The CBP data year (default '2023', the latest published vintage — CBP is released with a ~2-year lag). Validated ^\\d{4}$ (it rides in the request path).",
3763
3763
  ),
3764
3764
  limit: z
3765
3765
  .number()
@@ -6284,7 +6284,7 @@ export const TOOLS: ToolDef[] = [
6284
6284
  defineTool({
6285
6285
  name: "census_business_patterns",
6286
6286
  description:
6287
- "Market sizing by NAICS × geography — establishments, employment, and annual payroll from the US Census County Business Patterns (CBP) API (api.census.gov/data/{year}/cbp). ★REQUIRES a free CENSUS_API_KEY: the Census Data API has NO keyless tier, so without the key this tool THROWS an honest config error (get one at https://api.census.gov/data/key_signup.html; call api_key_status to see every source's key requirement). Input: optional `naics` (2–6 digit NAICS-2017, e.g. '5415'; omit to aggregate all sectors), `geography` (us|state|county, default us; county REQUIRES `state`), `state` (2-digit FIPS, e.g. '06'), `year` (default '2022'), optional `limit` (client-side top-N; CBP has no server pagination). Returns { rows:[{ name, geoId, naicsCode, naicsLabel, establishments, employees, annualPayrollUsd, state }] } + honest _meta. HONESTY: establishments/employees are integer counts and annualPayrollUsd is annual US dollars (×1000 from the source's $1,000-unit PAYANN); large-negative suppression sentinels map to null — NEVER a negative number and NEVER 0 (a genuine 0 stays 0; note CBP primarily uses noise-infusion + suppression flags, surfaced as reported — see the tool's suppression note); geoId/naicsCode/state are STRINGS (leading zeros survive). CBP returns the COMPLETE geography set for the filter (no pagination) ⇒ totalAvailable = the row count, complete:true. A missing/invalid key ⇒ invalid_input (a 302 to the Missing-Key page); a header-only body ⇒ honest empty (returned:0); a 5xx ⇒ THROWS; a 200 non-JSON ⇒ schema_drift. The key rides ONLY in the &key= query param — never logged or echoed.",
6287
+ "Market sizing by NAICS × geography — establishments, employment, and annual payroll from the US Census County Business Patterns (CBP) API (api.census.gov/data/{year}/cbp). ★REQUIRES a free CENSUS_API_KEY: the Census Data API has NO keyless tier, so without the key this tool THROWS an honest config error (get one at https://api.census.gov/data/key_signup.html; call api_key_status to see every source's key requirement). Input: optional `naics` (2–6 digit NAICS-2017, e.g. '5415'; omit to aggregate all sectors), `geography` (us|state|county, default us; county REQUIRES `state`), `state` (2-digit FIPS, e.g. '06'), `year` (default '2023'), optional `limit` (client-side top-N; CBP has no server pagination). Returns { rows:[{ name, geoId, naicsCode, naicsLabel, establishments, employees, annualPayrollUsd, state }] } + honest _meta. HONESTY: establishments/employees are integer counts and annualPayrollUsd is annual US dollars (×1000 from the source's $1,000-unit PAYANN); large-negative suppression sentinels map to null — NEVER a negative number and NEVER 0 (a genuine 0 stays 0; note CBP primarily uses noise-infusion + suppression flags, surfaced as reported — see the tool's suppression note); geoId/naicsCode/state are STRINGS (leading zeros survive). CBP returns the COMPLETE geography set for the filter (no pagination) ⇒ totalAvailable = the row count, complete:true. A missing/invalid key ⇒ invalid_input (a 302 to the Missing-Key page); a header-only body ⇒ honest empty (returned:0); a 5xx ⇒ THROWS; a 200 non-JSON ⇒ schema_drift. The key rides ONLY in the &key= query param — never logged or echoed.",
6288
6288
  inputSchema: CensusBusinessPatternsInput,
6289
6289
  handler: (input) => censusEconomic.businessPatterns(input),
6290
6290
  }),
@@ -6633,6 +6633,7 @@ async function main() {
6633
6633
  name: t.name,
6634
6634
  description: t.description,
6635
6635
  inputSchema: zodToJsonSchema(t.inputSchema),
6636
+ annotations: toolAnnotations(t.name),
6636
6637
  })),
6637
6638
  };
6638
6639
  });
@@ -6808,6 +6809,53 @@ export async function runTool(
6808
6809
  throw new Error(`Unknown tool: ${name}`);
6809
6810
  }
6810
6811
 
6812
+ // ─── MCP tool annotations ────────────────────────────────────────────────────
6813
+ // Every tool advertises a human-friendly `title` plus behaviour hints, per the MCP
6814
+ // spec and the Anthropic Connectors Directory requirement (title + readOnlyHint).
6815
+ // EVERY tool in this server is strictly READ-ONLY (queries public data; none mutate
6816
+ // upstream state) and OPEN-WORLD (calls external government/public APIs), so those
6817
+ // two hints are uniform. `title` is derived from the tool name via a source-prefix
6818
+ // map so e.g. `sam_search_opportunities` → "SAM.gov: Search Opportunities".
6819
+ const TOOL_SOURCE_LABELS: Record<string, string> = {
6820
+ sam: "SAM.gov", usas: "USAspending", edgar: "SEC EDGAR", fdic: "FDIC", cms: "CMS",
6821
+ treasury: "Treasury", regulations: "Regulations.gov", openfda: "openFDA",
6822
+ govinfo: "GovInfo", fema: "FEMA", far: "FAR/DFARS", ecfr: "eCFR", census: "Census",
6823
+ clinicaltrials: "ClinicalTrials.gov", bls: "BLS", socrata: "Socrata", opengov: "OpenGov",
6824
+ nsf: "NSF", nonprofit: "Nonprofit", nhtsa: "NHTSA", gsa: "GSA", grants: "Grants.gov",
6825
+ fred: "FRED", fac: "FAC", echo: "EPA ECHO", dol: "DOL", congress: "Congress.gov",
6826
+ ckan: "data.gov", bonfire: "Bonfire", arcgis: "ArcGIS", sba: "SBA", ofac: "OFAC",
6827
+ nws: "NWS", nppes: "NPPES", nist: "NIST", nih: "NIH", lda: "Senate LDA", hts: "USITC HTS",
6828
+ bea: "BEA", cbp: "CBP", cpsc: "CPSC", nvd: "NVD", courtlistener: "CourtListener",
6829
+ fpds: "FPDS", gao: "GAO", epa: "EPA", nsn: "NSN",
6830
+ };
6831
+ const TOOL_ACRONYMS = new Set([
6832
+ "naics", "psc", "cfda", "npi", "cve", "csv", "id", "url", "rss", "api", "hts", "sic",
6833
+ "dmepos", "cbp", "sam", "fdic", "epa", "ofac", "far", "dfars", "irs", "usc", "cfr",
6834
+ ]);
6835
+ export function humanizeToolTitle(name: string): string {
6836
+ const tokens = name.split("_");
6837
+ // `fed_register_*` is a two-token source label.
6838
+ let source = tokens[0] ?? "";
6839
+ let rest = tokens.slice(1);
6840
+ if (source === "fed" && tokens[1] === "register") {
6841
+ source = "fed_register";
6842
+ rest = tokens.slice(2);
6843
+ }
6844
+ const label = source === "fed_register" ? "Federal Register" : TOOL_SOURCE_LABELS[source];
6845
+ const words = (label ? rest : tokens)
6846
+ .map((w) => (TOOL_ACRONYMS.has(w) ? w.toUpperCase() : w.charAt(0).toUpperCase() + w.slice(1)))
6847
+ .join(" ");
6848
+ return label ? (words ? `${label}: ${words}` : label) : words;
6849
+ }
6850
+ /** Uniform MCP annotations for a tool: derived title + read-only / open-world hints. */
6851
+ export function toolAnnotations(name: string): {
6852
+ title: string;
6853
+ readOnlyHint: true;
6854
+ openWorldHint: true;
6855
+ } {
6856
+ return { title: humanizeToolTitle(name), readOnlyHint: true, openWorldHint: true };
6857
+ }
6858
+
6811
6859
  /**
6812
6860
  * Hand-rolled Zod → JSON Schema converter (subset we use).
6813
6861
  */
package/src/socrata.ts CHANGED
@@ -238,14 +238,21 @@ const CATALOG_URL = `https://${CATALOG_HOST}/api/catalog/v1`;
238
238
  // regex `$` alone would admit ("abcd-1234\n" passes /…$/ in JS).
239
239
  const DATASET_ID_RE = /^[a-z0-9]{4}-[a-z0-9]{4}$/;
240
240
 
241
- // D2 — an AGGREGATE $select projection. Matches a SoQL aggregate function
242
- // (count/sum/avg/min/max) applied via `fn(` — the `\b…\s*\(` shape avoids false
243
- // hits on column names like `max_temperature` (no paren) or `xmax(` (no word
244
- // boundary) — OR an explicit `group by`/`$group`. Case-insensitive. When the
245
- // caller's own $select is aggregate, the count(*) companion is skipped (its
246
- // raw-row total would be false for aggregate result rows). See `query`.
241
+ // D2 — a cardinality-CHANGING $select projection: ANY function call (`word(` —
242
+ // count/sum/avg/min/max/median/stddev/count_distinct/percentile/…), the bare
243
+ // `distinct` keyword, or an explicit `group by`/`$group`. Case-insensitive. The
244
+ // `\b\w+\s*\(` shape matches a real function CALL (a word followed by a paren),
245
+ // never a plain column name like `max_temperature` (no paren). When the caller's
246
+ // own $select is one of these, the result rows are NOT raw records, so the
247
+ // count(*) companion (which counts SOURCE rows) would be a FALSE total — e.g.
248
+ // `$select=distinct state` returns 54 rows but count(*) is 137,700 → a wrong total
249
+ // AND a hasMore livelock (paging forever over empty pages). So we skip the
250
+ // companion (totalAvailable:null + page-fullness hasMore). A plain comma-separated
251
+ // column list is NOT matched and keeps its real count(*) total. (A per-row
252
+ // function like `upper(x)` also matches → a conservative null-total, an honest
253
+ // "unknown" rather than risking a wrong one.) See `query`.
247
254
  const AGGREGATE_SELECT_RE =
248
- /\b(?:count|sum|avg|min|max)\s*\(|\bgroup\s+by\b|\$group\b/i;
255
+ /\b\w+\s*\(|\bdistinct\b|\bgroup\s+by\b|\$group\b/i;
249
256
 
250
257
  // ─── HONESTY-CRITICAL coercions (null, never 0, for absent) ───────
251
258
  // `num`/`str` are the shared, audited null-never-0 coercions in ./coerce.js
@@ -1468,7 +1468,7 @@ export async function searchRecompetes(args: {
1468
1468
  awardId: string;
1469
1469
  generatedInternalId: string;
1470
1470
  incumbent: string;
1471
- amount: number;
1471
+ amount: number | null;
1472
1472
  currentEndDate: string;
1473
1473
  daysUntilCurrentEnd: number;
1474
1474
  potentialEndDate?: string | null;
@@ -1515,8 +1515,11 @@ export async function searchRecompetes(args: {
1515
1515
  pastWindow = true;
1516
1516
  break;
1517
1517
  }
1518
- const amount = row["Award Amount"] ?? 0;
1519
- if (amount < minAwardValue) continue;
1518
+ // Filter/sort on a numeric view (absent = 0 for COMPARISON only), but EMIT the
1519
+ // raw value below so an ABSENT amount stays null — P3 null-never-0, matching every
1520
+ // sibling search tool ("an ABSENT Award Amount → null, NEVER a fabricated $0").
1521
+ const amountNum = row["Award Amount"] ?? 0;
1522
+ if (amountNum < minAwardValue) continue;
1520
1523
  const potentialEnd = includePotentialEnd
1521
1524
  ? row["Period of Performance Potential End Date"] ?? null
1522
1525
  : undefined;
@@ -1529,7 +1532,7 @@ export async function searchRecompetes(args: {
1529
1532
  awardId: row["Award ID"] ?? "",
1530
1533
  generatedInternalId: row.generated_internal_id ?? "",
1531
1534
  incumbent: row["Recipient Name"] ?? "",
1532
- amount,
1535
+ amount: row["Award Amount"] ?? null,
1533
1536
  currentEndDate: end as string,
1534
1537
  daysUntilCurrentEnd: d,
1535
1538
  ...(includePotentialEnd
@@ -1557,7 +1560,10 @@ export async function searchRecompetes(args: {
1557
1560
  results.sort((a, b) => {
1558
1561
  if (a.daysUntilCurrentEnd !== b.daysUntilCurrentEnd)
1559
1562
  return a.daysUntilCurrentEnd - b.daysUntilCurrentEnd;
1560
- if (b.amount !== a.amount) return b.amount - a.amount;
1563
+ // A null amount (P3 — absent, not $0) sorts as 0 for this tiebreak ONLY; it is
1564
+ // never surfaced as 0 in the output.
1565
+ const av = a.amount ?? 0, bv = b.amount ?? 0;
1566
+ if (bv !== av) return bv - av;
1561
1567
  return a.awardId.localeCompare(b.awardId);
1562
1568
  });
1563
1569
 
@@ -1570,10 +1576,15 @@ export async function searchRecompetes(args: {
1570
1576
  // (early-stop fired). If the scan budget truncated, it is unknown → null,
1571
1577
  // and the returned set is a lower bound.
1572
1578
  const totalAvailable = scanTruncated ? null : totalInWindow;
1573
- const nextOffset = startIdx + pageSize;
1574
- const hasMore = scanTruncated
1575
- ? true // more may exist beyond the scanned pages
1576
- : nextOffset < totalInWindow;
1579
+ // `nextOffset` may ONLY advance into rows we ACTUALLY scanned. The un-scanned
1580
+ // remainder (scanTruncated) is NOT offset-reachable: paging past totalInWindow
1581
+ // re-runs the whole bounded scan and returns EMPTY pages forever — the
1582
+ // "re-fetch the SAME page" livelock every other tool in this module emits
1583
+ // nextOffset:null to avoid. So advance the cursor only while more SCANNED rows
1584
+ // remain; signal the un-scanned tail via hasMore:true + nextOffset:null (the
1585
+ // notes already tell the caller to raise scanBudgetPages, not to paginate).
1586
+ const moreInScanned = startIdx + pageSize < totalInWindow;
1587
+ const hasMore = moreInScanned || scanTruncated;
1577
1588
  const truncated = hasMore || scanTruncated;
1578
1589
 
1579
1590
  const notes: string[] = [
@@ -1612,7 +1623,7 @@ export async function searchRecompetes(args: {
1612
1623
  pagination: {
1613
1624
  offset: startIdx,
1614
1625
  limit: pageSize,
1615
- nextOffset: hasMore ? nextOffset : null,
1626
+ nextOffset: moreInScanned ? startIdx + pageSize : null,
1616
1627
  hasMore,
1617
1628
  },
1618
1629
  filtersApplied,