@cliwant/mcp-sam-gov 1.13.2 → 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,435 @@
1
+ /**
2
+ * Toolset profiles for MCP_SAM_GOV_TOOLSETS.
3
+ *
4
+ * Each of the 152 registered tools belongs to exactly one named toolset.
5
+ * A user may load a subset by setting the env var (comma- or space-separated,
6
+ * case-insensitive). Unset or "all" → all 152 tools loaded (default; byte-
7
+ * identical to today's behaviour). Unknown names → stderr warning AND a note
8
+ * in the server instructions; if no valid name remains → fall back to all,
9
+ * with a stderr warning and an instructions note.
10
+ *
11
+ * `feedback` and `api_key_status` are always loaded in every profile.
12
+ * Their toolset mapping is "core" (so a one-set-per-tool mapping is maintained),
13
+ * but `resolveToolsets` always injects them into any profile's loaded set.
14
+ * They are listed in the README table as "always loaded".
15
+ *
16
+ * Available toolsets:
17
+ * core — SAM.gov discovery/awards/wage determinations/exclusions/integrity,
18
+ * Grants.gov, all USAspending, FPDS, GAO, FAR/eCFR/Federal Register,
19
+ * SBA, OFAC, GSA labor-rate benchmarks; api_key_status + feedback
20
+ * (always loaded in every profile)
21
+ * sled — State/local (SLED): OpenGov, Bonfire, ArcGIS, Socrata,
22
+ * data.gov/CKAN, Tableau, Open Checkbook, search.gov domains
23
+ * vetting — Partner due-diligence: FAC, FDIC, EPA ECHO/TRI,
24
+ * CourtListener, nonprofit (IRS 990), Senate LDA lobbying
25
+ * disclosure — SEC EDGAR financial filings and XBRL frames
26
+ * regulatory — Regulations.gov, Congress.gov, GovInfo
27
+ * pricing — GSA per-diem, BLS, Treasury, BEA, Census business-patterns,
28
+ * FRED, DOL, USITC HTS
29
+ * health — CMS, NPPES, NIH, NSF, ClinicalTrials.gov, openFDA
30
+ * safety — NHTSA vehicle recalls, CPSC consumer-product recalls
31
+ * geo — Census geocode, FEMA disasters, NWS alerts, CBP border wait
32
+ * times, data.gov catalog
33
+ * cyber — NVD CVE, CISA KEV, NIST SP 800-53
34
+ */
35
+
36
+ /** Every valid toolset name. */
37
+ export const ALL_TOOLSET_NAMES = [
38
+ "core",
39
+ "sled",
40
+ "vetting",
41
+ "disclosure",
42
+ "regulatory",
43
+ "pricing",
44
+ "health",
45
+ "safety",
46
+ "geo",
47
+ "cyber",
48
+ ] as const;
49
+
50
+ export type ToolsetName = typeof ALL_TOOLSET_NAMES[number];
51
+
52
+ /**
53
+ * Short 2–4 word description for each toolset, shown in the server instructions
54
+ * when a profile is active so agents know what other sets exist.
55
+ */
56
+ export const TOOLSET_HINTS: Readonly<Record<ToolsetName, string>> = {
57
+ core: "SAM, USAspending, FAR",
58
+ sled: "state/local procurement",
59
+ vetting: "FAC, FDIC, EPA, OFAC",
60
+ disclosure: "SEC EDGAR filings",
61
+ regulatory: "Regs.gov, Congress, GovInfo",
62
+ pricing: "GSA, BLS, Treasury, DOL",
63
+ health: "CMS, NIH, openFDA",
64
+ safety: "NHTSA, CPSC recalls",
65
+ geo: "Census geocode, FEMA, NWS",
66
+ cyber: "CVE, CISA KEV, NIST",
67
+ };
68
+
69
+ /**
70
+ * Tools that are always loaded regardless of the active profile.
71
+ * Their toolset mapping is "core" (one-set-per-tool), but resolveToolsets
72
+ * injects them into every profile's loaded set.
73
+ */
74
+ export const ALWAYS_LOADED_TOOLS: ReadonlySet<string> = new Set([
75
+ "feedback",
76
+ "api_key_status",
77
+ ]);
78
+
79
+ /**
80
+ * Canonical mapping: tool name → toolset.
81
+ * Every registered tool must appear exactly once.
82
+ * Tests enforce: union === all 152 tools, no entry for an unregistered name.
83
+ *
84
+ * After moves (review 2026-09-18):
85
+ * ofac_screen_entity vetting → core (govcon users screen SAM exclusions + OFAC together)
86
+ * gsa_benchmark_labor_rates pricing → core (proposal pricing; needed in core govcon workflows)
87
+ * hts_lookup geo → pricing (tariff/duty data; groups with cost/rate tools)
88
+ *
89
+ * Resulting counts: core=60, sled=13, vetting=16, disclosure=8, regulatory=9,
90
+ * pricing=15, health=17, safety=3, geo=8, cyber=3. Total=152.
91
+ */
92
+ export const TOOL_TOOLSET_MAP: Readonly<Record<string, ToolsetName>> = {
93
+ // ── core (60) ──────────────────────────────────────────────────────────────
94
+ // SAM.gov: discovery, attachments, wage-determinations, exclusions, integrity
95
+ sam_search_opportunities: "core",
96
+ sam_search_shaping: "core",
97
+ sam_get_opportunity: "core",
98
+ sam_fetch_description: "core",
99
+ sam_attachment_url: "core",
100
+ sam_fetch_attachment_text: "core",
101
+ sam_lookup_organization: "core",
102
+ sam_lookup_notice_fields: "core",
103
+ sam_check_exclusions: "core",
104
+ sam_get_wage_rates: "core",
105
+ sam_search_wage_determinations: "core",
106
+ sam_integrity_lookup: "core",
107
+ // Grants.gov
108
+ grants_search: "core",
109
+ grants_get_opportunity: "core",
110
+ // USAspending (29)
111
+ usas_search_awards: "core",
112
+ usas_search_individual_awards: "core",
113
+ usas_search_subagency_spending: "core",
114
+ usas_lookup_agency: "core",
115
+ usas_search_awards_by_recipient: "core",
116
+ usas_search_subawards: "core",
117
+ usas_search_recompetes: "core",
118
+ usas_search_expiring_contracts: "core",
119
+ usas_get_award_detail: "core",
120
+ usas_analyze_incumbent: "core",
121
+ usas_spending_over_time: "core",
122
+ usas_search_psc_spending: "core",
123
+ usas_search_state_spending: "core",
124
+ usas_search_cfda_spending: "core",
125
+ usas_search_federal_account_spending: "core",
126
+ usas_search_agency_spending: "core",
127
+ usas_get_agency_profile: "core",
128
+ usas_get_agency_awards_summary: "core",
129
+ usas_get_agency_budget_function: "core",
130
+ usas_search_recipients: "core",
131
+ usas_get_recipient_profile: "core",
132
+ usas_autocomplete_naics: "core",
133
+ usas_autocomplete_recipient: "core",
134
+ usas_disaster_spending: "core",
135
+ usas_glossary: "core",
136
+ usas_list_disaster_codes: "core",
137
+ usas_list_toptier_agencies: "core",
138
+ usas_naics_hierarchy: "core",
139
+ usas_search_teaming_partners: "core",
140
+ // FPDS
141
+ fpds_search_awards: "core",
142
+ // GAO
143
+ gao_protest_lookup: "core",
144
+ // FAR / DFARS
145
+ far_search: "core",
146
+ far_clause_lookup: "core",
147
+ far_compliance_matrix: "core",
148
+ // eCFR
149
+ ecfr_search: "core",
150
+ ecfr_list_titles: "core",
151
+ ecfr_get_section: "core",
152
+ // Federal Register
153
+ fed_register_search_documents: "core",
154
+ fed_register_get_document: "core",
155
+ fed_register_list_agencies: "core",
156
+ fed_register_public_inspection: "core",
157
+ // SBA
158
+ sba_size_standard: "core",
159
+ // OFAC (moved from vetting — govcon users screen SAM exclusions + OFAC together)
160
+ ofac_screen_entity: "core",
161
+ // GSA labor-rate benchmarks (moved from pricing — needed in core proposal-pricing)
162
+ gsa_benchmark_labor_rates: "core",
163
+ // Always-loaded utilities (every profile; mapped to core for 1-set invariant)
164
+ api_key_status: "core",
165
+ feedback: "core",
166
+
167
+ // ── sled (13) ──────────────────────────────────────────────────────────────
168
+ // State/local procurement + spend
169
+ opengov_list_governments: "sled",
170
+ opengov_search_solicitations: "sled",
171
+ bonfire_list_organizations: "sled",
172
+ bonfire_search_opportunities: "sled",
173
+ arcgis_hub_discover_datasets: "sled",
174
+ arcgis_feature_query: "sled",
175
+ socrata_discover_datasets: "sled",
176
+ socrata_query: "sled",
177
+ ckan_discover_datasets: "sled",
178
+ ckan_query: "sled",
179
+ tableau_view_csv: "sled",
180
+ open_checkbook_search: "sled",
181
+ search_gov_domains: "sled",
182
+
183
+ // ── vetting (16) ───────────────────────────────────────────────────────────
184
+ // Partner due-diligence (ofac_screen_entity moved to core)
185
+ fac_search_audits: "vetting",
186
+ fac_get_findings: "vetting",
187
+ fdic_search_institutions: "vetting",
188
+ fdic_institution_financials: "vetting",
189
+ fdic_institution_history: "vetting",
190
+ fdic_bank_failures: "vetting",
191
+ fdic_branch_deposits: "vetting",
192
+ fdic_industry_summary: "vetting",
193
+ fdic_risk_ratios: "vetting",
194
+ echo_search_facilities: "vetting",
195
+ echo_facility_report: "vetting",
196
+ epa_tri_facilities: "vetting",
197
+ courtlistener_search_opinions: "vetting",
198
+ nonprofit_search: "vetting",
199
+ nonprofit_financials: "vetting",
200
+ lda_search_filings: "vetting",
201
+
202
+ // ── disclosure (8) ─────────────────────────────────────────────────────────
203
+ // SEC EDGAR
204
+ edgar_company_concept: "disclosure",
205
+ edgar_company_facts: "disclosure",
206
+ edgar_company_filings: "disclosure",
207
+ edgar_daily_filing_index: "disclosure",
208
+ edgar_filing_index: "disclosure",
209
+ edgar_full_text_search: "disclosure",
210
+ edgar_lookup_cik: "disclosure",
211
+ edgar_xbrl_frames: "disclosure",
212
+
213
+ // ── regulatory (9) ─────────────────────────────────────────────────────────
214
+ // Regulations.gov, Congress.gov, GovInfo
215
+ regulations_search_documents: "regulatory",
216
+ regulations_search_comments: "regulatory",
217
+ regulations_search_dockets: "regulatory",
218
+ regulations_get_docket: "regulatory",
219
+ congress_search_bills: "regulatory",
220
+ congress_get_bill: "regulatory",
221
+ govinfo_list_collections: "regulatory",
222
+ govinfo_search_packages: "regulatory",
223
+ govinfo_get_package: "regulatory",
224
+
225
+ // ── pricing (15) ───────────────────────────────────────────────────────────
226
+ // GSA per-diem, BLS, Treasury, BEA, Census business-patterns, FRED, DOL, USITC HTS
227
+ // (gsa_benchmark_labor_rates moved to core; hts_lookup moved in from geo)
228
+ gsa_perdiem_rates: "pricing",
229
+ bls_oews_wages: "pricing",
230
+ bls_qcew: "pricing",
231
+ bls_timeseries: "pricing",
232
+ treasury_avg_interest_rates: "pricing",
233
+ treasury_debt_to_penny: "pricing",
234
+ treasury_monthly_statement: "pricing",
235
+ treasury_query_dataset: "pricing",
236
+ bea_regional_data: "pricing",
237
+ census_business_patterns: "pricing",
238
+ fred_search_series: "pricing",
239
+ fred_series_observations: "pricing",
240
+ dol_list_datasets: "pricing",
241
+ dol_get_dataset: "pricing",
242
+ hts_lookup: "pricing",
243
+
244
+ // ── health (17) ────────────────────────────────────────────────────────────
245
+ // CMS, NPPES, NIH, NSF, ClinicalTrials.gov, openFDA
246
+ cms_medicare_provider_services: "health",
247
+ cms_hospital_compare: "health",
248
+ cms_facility_directory: "health",
249
+ cms_dmepos_suppliers: "health",
250
+ cms_revoked_providers: "health",
251
+ cms_query_dataset: "health",
252
+ cms_search_datasets: "health",
253
+ nppes_lookup_provider: "health",
254
+ nih_reporter_search_projects: "health",
255
+ nsf_search_awards: "health",
256
+ nsf_get_award: "health",
257
+ clinicaltrials_search_studies: "health",
258
+ clinicaltrials_get_study: "health",
259
+ clinicaltrials_facet_counts: "health",
260
+ openfda_enforcement: "health",
261
+ openfda_device_clearances: "health",
262
+ openfda_drug_approvals: "health",
263
+
264
+ // ── safety (3) ─────────────────────────────────────────────────────────────
265
+ // NHTSA, CPSC
266
+ nhtsa_recalls: "safety",
267
+ nhtsa_complaints: "safety",
268
+ cpsc_recalls: "safety",
269
+
270
+ // ── geo (8) ────────────────────────────────────────────────────────────────
271
+ // Census geocode, FEMA, NWS, CBP, data.gov catalog
272
+ // (hts_lookup moved to pricing)
273
+ census_geocode_address: "geo",
274
+ census_geographies_by_coordinates: "geo",
275
+ fema_disaster_declarations: "geo",
276
+ fema_search_hazard_mitigation: "geo",
277
+ fema_search_public_assistance: "geo",
278
+ nws_active_alerts: "geo",
279
+ cbp_border_wait_times: "geo",
280
+ datagov_search_datasets: "geo",
281
+
282
+ // ── cyber (3) ──────────────────────────────────────────────────────────────
283
+ // NVD CVE, CISA KEV, NIST 800-53
284
+ cve_lookup: "cyber",
285
+ cisa_kev_lookup: "cyber",
286
+ nist_800_53_controls: "cyber",
287
+ };
288
+
289
+ export type ResolveResult = {
290
+ /** Tools that are loaded (by name). */
291
+ loaded: Set<string>;
292
+ /** Valid toolset names that were requested. */
293
+ sets: string[];
294
+ /** Names that were not recognised as valid toolset names. */
295
+ unknown: string[];
296
+ /** True if we fell back to all because no valid name survived. */
297
+ fellBack: boolean;
298
+ };
299
+
300
+ /**
301
+ * Parse MCP_SAM_GOV_TOOLSETS and return the set of tool names to expose.
302
+ *
303
+ * Pure: takes the env value and the full tool-name list as inputs, reads
304
+ * nothing from `process.env`, and has no side-effects. Call sites (server.ts)
305
+ * are responsible for logging any warnings to stderr.
306
+ *
307
+ * `feedback` and `api_key_status` are always injected into `loaded` regardless
308
+ * of the requested profile (ALWAYS_LOADED_TOOLS invariant).
309
+ *
310
+ * @param envValue The raw value of MCP_SAM_GOV_TOOLSETS (undefined / empty = all).
311
+ * @param toolNames The complete list of registered tool names (from TOOLS).
312
+ */
313
+ export function resolveToolsets(
314
+ envValue: string | undefined,
315
+ toolNames: readonly string[],
316
+ ): ResolveResult {
317
+ const allLoaded = new Set(toolNames);
318
+
319
+ // Unset or empty → all.
320
+ if (!envValue || !envValue.trim()) {
321
+ return { loaded: allLoaded, sets: ["all"], unknown: [], fellBack: false };
322
+ }
323
+
324
+ // Split on commas and/or whitespace, normalise to lower-case.
325
+ const tokens = envValue
326
+ .split(/[\s,]+/)
327
+ .map((t) => t.toLowerCase().trim())
328
+ .filter(Boolean);
329
+
330
+ // "all" anywhere → all.
331
+ if (tokens.includes("all")) {
332
+ return { loaded: allLoaded, sets: ["all"], unknown: [], fellBack: false };
333
+ }
334
+
335
+ const validSet = new Set<string>(ALL_TOOLSET_NAMES);
336
+ const validNames: string[] = [];
337
+ const unknownNames: string[] = [];
338
+
339
+ for (const token of tokens) {
340
+ if (validSet.has(token)) {
341
+ if (!validNames.includes(token)) validNames.push(token);
342
+ } else {
343
+ unknownNames.push(token);
344
+ }
345
+ }
346
+
347
+ // No valid names → fall back to all.
348
+ if (validNames.length === 0) {
349
+ return {
350
+ loaded: allLoaded,
351
+ sets: ["all"],
352
+ unknown: unknownNames,
353
+ fellBack: true,
354
+ };
355
+ }
356
+
357
+ // Build the loaded set: union of all tools in the requested toolsets.
358
+ const requestedSets = new Set(validNames);
359
+ const loaded = new Set<string>();
360
+ for (const toolName of toolNames) {
361
+ const ts = TOOL_TOOLSET_MAP[toolName];
362
+ if (ts && requestedSets.has(ts)) {
363
+ loaded.add(toolName);
364
+ }
365
+ }
366
+
367
+ // Always inject always-loaded tools (feedback, api_key_status) regardless
368
+ // of the profile. They are already in core's mapping but must appear in every
369
+ // profile's loaded set.
370
+ for (const toolName of toolNames) {
371
+ if (ALWAYS_LOADED_TOOLS.has(toolName)) {
372
+ loaded.add(toolName);
373
+ }
374
+ }
375
+
376
+ return {
377
+ loaded,
378
+ sets: validNames,
379
+ unknown: unknownNames,
380
+ fellBack: false,
381
+ };
382
+ }
383
+
384
+ /**
385
+ * Filter a tool list to only those in the loaded set.
386
+ * Exported so server.ts can call this and tests can import and verify the logic
387
+ * without mutating dist/server.js.
388
+ */
389
+ export function filterToolsFor<T extends { name: string }>(
390
+ tools: readonly T[],
391
+ loaded: ReadonlySet<string>,
392
+ ): T[] {
393
+ return tools.filter((t) => loaded.has(t.name));
394
+ }
395
+
396
+ /**
397
+ * Build the structured tool_not_loaded error envelope.
398
+ * Exported so server.ts can call this and tests can import and verify the logic
399
+ * without mutating dist/server.js.
400
+ *
401
+ * @param toolName The name of the tool that was called but not loaded.
402
+ * @param loadedSets The currently loaded toolset names (from ResolveResult.sets).
403
+ */
404
+ export function toolNotLoadedEnvelope(
405
+ toolName: string,
406
+ loadedSets: readonly string[],
407
+ ): {
408
+ ok: false;
409
+ error: { kind: "tool_not_loaded"; message: string; retryable: false };
410
+ } {
411
+ const toolset = TOOL_TOOLSET_MAP[toolName] ?? "unknown";
412
+ // Suggest the union of currently loaded sets + the needed set (not just the
413
+ // needed set alone, which would silently drop the user's current profile).
414
+ const isAll = loadedSets.length === 1 && loadedSets[0] === "all";
415
+ let suggestion: string;
416
+ if (isAll) {
417
+ // All tools are already loaded — the tool simply doesn't exist.
418
+ suggestion = `MCP_SAM_GOV_TOOLSETS=all`;
419
+ } else if (toolset === "unknown") {
420
+ suggestion = `MCP_SAM_GOV_TOOLSETS=all`;
421
+ } else {
422
+ // Build union: current loaded sets + the required set, deduped.
423
+ const unionSets = [...loadedSets];
424
+ if (!unionSets.includes(toolset)) unionSets.push(toolset);
425
+ suggestion = `MCP_SAM_GOV_TOOLSETS=${unionSets.join(",")}`;
426
+ }
427
+ const error = {
428
+ kind: "tool_not_loaded" as const,
429
+ message:
430
+ `Tool '${toolName}' belongs to the '${toolset}' toolset, which is not loaded. ` +
431
+ `Set ${suggestion} to enable it.`,
432
+ retryable: false as const,
433
+ };
434
+ return { ok: false as const, error };
435
+ }