@cliwant/mcp-sam-gov 1.13.2 → 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/server.ts CHANGED
@@ -104,11 +104,19 @@ import {
104
104
  } from "./meta.js";
105
105
  import { pathToFileURL, fileURLToPath } from "node:url";
106
106
  import { realpathSync } from "node:fs";
107
+ import {
108
+ resolveToolsets,
109
+ TOOL_TOOLSET_MAP,
110
+ ALL_TOOLSET_NAMES,
111
+ TOOLSET_HINTS,
112
+ filterToolsFor,
113
+ toolNotLoadedEnvelope,
114
+ } from "./toolsets.js";
107
115
 
108
116
  const SERVER_NAME = "mcp-sam-gov";
109
117
  // Kept in lockstep with package.json / manifest.json / server.json.
110
118
  // Keep in sync with package.json "version" (asserted at release; see CHANGELOG).
111
- const SERVER_VERSION = "1.13.2";
119
+ const SERVER_VERSION = "1.14.0";
112
120
 
113
121
  // ─── Tool input schemas (Zod) ────────────────────────────────────
114
122
 
@@ -3098,7 +3106,7 @@ const TableauViewCsvInput = z.object({
3098
3106
  const ArcgisFeatureQueryInput = z.object({
3099
3107
  service: z
3100
3108
  .enum(arcgisFeature.ARCGIS_SERVICES.map((s) => s.key) as [string, ...string[]])
3101
- .describe("The curated ArcGIS layer (SSRF allowlist enum). DC OCP PASS: 'dc_pass_solicitations' (live solicitations ~25k), 'dc_pass_contracts', 'dc_pass_purchase_orders', 'dc_pass_payments'. Other US local govs: 'asheville_purchase_orders'/'asheville_po_summary' (Asheville NC), 'bellevue_vendor_payments'/'bellevue_awarded_contracts' (Bellevue WA), 'miamidade_purchase_orders_2025'/'miamidade_purchase_orders_2017' (Miami-Dade FL, current/2017), 'suffolk_county_ny_contracts_2018' (Suffolk County NY), 'matsu_borough_ak_checkbook' (Matanuska-Susitna Borough AK), 'lasvegas_checkbook' (Las Vegas NV ~373k), 'baltimore_checkbook' (Baltimore City MD ~367k), 'naperville_vendor_payments' (Naperville IL ~127k), 'worcester_ma_checkbook_fy25' (Worcester MA FY25), 'lasvegas_purchasing_contracts' (Las Vegas NV contract register), 'txdot_construction_projects' (Texas DOT, awarded construction company ~85k), 'akdot_construction_awards'/'akdot_aashtoware_proposals' (Alaska DOT&PF bid awards/proposals), 'iowadot_public_bid_awards' (Iowa DOT public bid), 'okdot_cirb_contract_status' (Oklahoma DOT CIRB contract status), 'topeka_checkbook_aggregate' (Topeka KS checkbook FY2015–2023 ~332k), 'nddot_flex_setaside_road'/'nddot_flex_partner_road'/'nddot_flex_setaside_bridge'/'nddot_flex_partner_bridge' (North Dakota DOT federal flex-funding awards to local public agencies — counties/townships/cities, NOT vendor contracts; a proxy because ND's checkbook/procurement portal is not keyless-reachable). 27 curated services (state DOT bid/award registers: TX/AK/IA/OK + ND DOT flex-funding awards + municipal checkbooks/contracts)."),
3109
+ .describe("Service key (SSRF allowlist; 27 services). DC OCP PASS (solicitations/contracts/purchase_orders/payments). US local govs: Asheville NC, Bellevue WA, Miami-Dade FL×2, Suffolk County NY, Mat-Su AK, Las Vegas NV×2, Baltimore MD, Naperville IL, Worcester MA, Topeka KS (FY2015–23); TX/AK/IA/OK DOT bid/award registers. ND DOT flex-funding to local agencies (nddot_flex×4 — NOT vendor contracts)."),
3102
3110
  where: z
3103
3111
  .string()
3104
3112
  .min(1)
@@ -3532,7 +3540,7 @@ const HtsLookupInput = z.object({
3532
3540
  // SECOND POST-batch getJson-port consumer (after NIH). SSRF surface = a compile-
3533
3541
  // time-CONSTANT host+path; seriesids ride in the module-built POST body. `series`
3534
3542
  // is a FROZEN 9-key curated enum (the SSRF value guard + the units-label source);
3535
- // `seriesId` is the raw passthrough, charclass-validated ^[A-Z0-9]{1,20}$. Years
3543
+ // `seriesId` is the raw passthrough, charclass-validated ^[A-Z0-9]{1,25}$. Years
3536
3544
  // are bounded ints (1900..currentYear+1); the span is clamped to the tier cap
3537
3545
  // (v1 ~10y) BEFORE the fetch + disclosed. An OPTIONAL free BLS_API_KEY rides ONLY
3538
3546
  // in the POST body (v2, ~500/day) — never a URL/header/label/_meta/log.
@@ -3545,11 +3553,11 @@ const BlsTimeseriesInput = z.object({
3545
3553
  "One or more CURATED series enum keys (typo-proof; each carries a meaning + units label): cpi_u_all (CPI-U all items NSA, index), cpi_u_core (CPI-U core NSA, index), ppi_final_demand (PPI final demand NSA, index), eci_total_comp (ECI total comp — ★12-MO % CHANGE, not an index), eci_wages (ECI wages — ★12-MO % CHANGE), unemployment_rate (SA, percent), labor_force_participation (SA, percent), employment_total_nonfarm (SA, thousands of persons), avg_hourly_earnings (SA, dollars/hour). NSA CPI-U is the escalation/EPA-clause reference. At least one of series/seriesId is required; both may be combined.",
3546
3554
  ),
3547
3555
  seriesId: z
3548
- .array(z.string().regex(/^[A-Z0-9]{1,20}$/))
3556
+ .array(z.string().regex(/^[A-Z0-9]{1,25}$/))
3549
3557
  .max(bls.BLS_SERIES_KEYS.length + 50)
3550
3558
  .optional()
3551
3559
  .describe(
3552
- "One or more RAW BLS series IDs (power-user passthrough for the un-curatable space — OEWS area×occupation, local-area unemployment LAUCN…, SA/regional CPI variants). Charclass ^[A-Z0-9]{1,20}$ (uppercase alnum; punctuation/whitespace/lowercase rejected — SSRF + 'verify the ID' honesty). A raw ID has units:null (consult BLS). A nonexistent/typo'd ID returns BLS success + empty data (the ambiguity is disclosed, not asserted as 'no data'). At least one of series/seriesId is required.",
3560
+ "One or more RAW BLS series IDs (power-user passthrough for the un-curatable space — OEWS area×occupation, local-area unemployment LAUCN…, SA/regional CPI variants). Charclass ^[A-Z0-9]{1,25}$ (uppercase alnum; punctuation/whitespace/lowercase rejected — SSRF + 'verify the ID' honesty). OEWS IDs are 25 chars (e.g. OEUN000000000000015125201). A raw ID has units:null (consult BLS). A nonexistent/typo'd ID returns BLS success + empty data (the ambiguity is disclosed, not asserted as 'no data'). At least one of series/seriesId is required.",
3553
3561
  ),
3554
3562
  startYear: z
3555
3563
  .number()
@@ -4573,7 +4581,7 @@ const LdaSearchFilingsInput = z.object({
4573
4581
  .string()
4574
4582
  .min(1)
4575
4583
  .optional()
4576
- .describe("NOTE: the keyless /filings/ endpoint has NO server-side government-entity filter — the LDA API silently ignores it, so this value is NOT applied (reported in _meta.filtersDropped, never as a narrowed total). Government entities are nested per lobbying activity (each filing's lobbyingActivities[].governmentEntities); to find who lobbied an agency, narrow by registrantName/clientName/issue and inspect those nested entities. Retained for discoverability of the limitation."),
4584
+ .describe("NOTE: /filings/ has NO server-side government-entity filter — the LDA API silently ignores this field (reported in _meta.filtersDropped, never as a narrowed total). Government entities are nested per activity in lobbyingActivities[].governmentEntities; narrow by registrantName/clientName/issue and inspect those nested entities. Retained for discoverability."),
4577
4585
  issue: z
4578
4586
  .string()
4579
4587
  .min(1)
@@ -5684,8 +5692,7 @@ export const TOOLS: ToolDef[] = [
5684
5692
  // kev.listed to null (never false), and a kevOnly filter during an outage THROWS.
5685
5693
  defineTool({
5686
5694
  name: "cve_lookup",
5687
- description:
5688
- "Look up NIST NVD CVE records (keyless; services.nvd.nist.gov CVE API 2.0) — exact by `cveId` (CVE-YYYY-NNNN) OR search by `keyword`/`cpeName`/`cvssV3Severity`/a publication or last-modified date range — each row JOINED with its CISA KEV (Known Exploited Vulnerabilities) status. THE B2G unlock for FedRAMP/CMMC/SBOM IT-compliance: CVSS severity AND whether CISA mandates remediation by a date, in one row. Returns { results:[{ cveId, vulnStatus, rejected, published, lastModified, description, cvssMetrics:[{version,source,type,baseScore,baseSeverity,vectorString,exploitabilityScore,impactScore}], primaryCvss:{version,baseScore,baseSeverity,type}|null, cwes, references, kev }] } + honest _meta. Optional `kevOnly` (KEV-listed rows only), `resultsPerPage` (≤2000, def 50), `startIndex`. CVSS HONESTY: every metrics key matching ^cvssMetric (V2/V30/V31/V40) is surfaced as its own cvssMetrics[] element — versions are NEVER conflated and ssvcV203/non-CVSS keys are excluded; V2 baseSeverity reads from the metric level; primaryCvss is the highest-version metric, preferring type:'Primary' but FALLING BACK to the highest Secondary (a real CNA score is never dropped), null ONLY when no CVSS exists (Rejected/Awaiting) — base scores are null-never-0. KEV HONESTY: kev is {listed:true,dateAdded,dueDate,ransomware,requiredAction,catalogVersion} | {listed:false,note} | {listed:null,status:'unavailable'}; a not-listed result carries the not-in-KEV≠safe caveat (absence is NOT a clearance); if the KEV catalog cannot load, kev.listed degrades to NULL (never false) with fieldsUnavailable:['kev'], and a kevOnly filter during that outage THROWS (a KEV-membership filter is unanswerable without the catalog). PAGINATION is from NVD's EXACT totalResults, never page length. A genuine totalResults:0 is an honest found:false; a 403/429 rate breach THROWS rate_limited with the NVD_API_KEY tier disclosure; 404/5xx/timeout/off-host-redirect THROW (never a fake-empty). An OPTIONAL free NVD_API_KEY (env; https://nvd.nist.gov/developers/request-an-api-key) lifts the rate and is sent ONLY in the apiKey header — never a URL/label/_meta/log.",
5695
+ description: "Look up NIST NVD CVE records (keyless; services.nvd.nist.gov CVE API 2.0) — exact by `cveId` OR search by `keyword`/`cpeName`/`cvssV3Severity`/date range — each row JOINED with its CISA KEV status. Returns { results:[{ cveId, vulnStatus, rejected, published, lastModified, description, cvssMetrics:[{version,source,type,baseScore,baseSeverity,vectorString,exploitabilityScore,impactScore}], primaryCvss:{version,baseScore,baseSeverity,type}|null, cwes, references, kev }] } + honest _meta. Optional `kevOnly`, `resultsPerPage` (≤2000, def 50), `startIndex`. CVSS HONESTY: every ^cvssMetric key (V2/V30/V31/V40) is its own element — versions never conflated; V2 baseSeverity reads from metric level; primaryCvss is highest-version, type:Primary preferred but FALLS BACK to highest Secondary (real CNA score never dropped), null ONLY when no CVSS exists — base scores null-never-0. KEV HONESTY: kev is {listed:true,dateAdded,dueDate,ransomware,requiredAction,catalogVersion} | {listed:false,note} | {listed:null,status:'unavailable'}; not-listed ≠ safe (absence is NOT a clearance); if KEV catalog cannot load, kev.listed degrades to NULL (never false) with fieldsUnavailable:['kev']; a kevOnly filter during KEV outage THROWS. PAGINATION from NVD EXACT totalResults (never page length). Genuine totalResults:0 → honest found:false; 403/429 → rate_limited THROWS with NVD_API_KEY tier disclosure; 404/5xx/timeout/off-host THROW. Optional free NVD_API_KEY (env) lifts the rate — sent ONLY in the apiKey header.",
5689
5696
  inputSchema: CveLookupInput,
5690
5697
  handler: (input) => nvd.cveLookup(input),
5691
5698
  }),
@@ -5698,16 +5705,14 @@ export const TOOLS: ToolDef[] = [
5698
5705
  }),
5699
5706
  defineTool({
5700
5707
  name: "nist_800_53_controls",
5701
- description:
5702
- "Look up NIST SP 800-53 Rev 5 security & privacy CONTROLS (keyless) — the requirement backbone for FedRAMP / CMMC / RMF compliance work. Retrieve a control by `controlId` (exact, e.g. 'AC-2', 'SC-7', 'AC-2(1)'), a `family` (2-letter code 'AC'/'SC'/'IA' or a name substring 'Access Control'), and/or a `keyword` (case-insensitive substring over title + statement); `limit`/`offset` pagination. Each row: { id (e.g. 'AC-2'), family (e.g. 'AC — Access Control'), title, status ('withdrawn' | null), statement (the labelled requirement prose; NULL for a WITHDRAWN control, never ''), guidance (discussion), incorporatedInto:[control ids that superseded a withdrawn control, e.g. AC-13 → ['AC-2','AU-6']], enhancements:[{id,title}] (e.g. AC-2(1)) }. Complements cve_lookup + cisa_kev_lookup (the vulnerability side) with the CONTROL/requirement side. HONESTY: source is NIST's OFFICIAL OSCAL catalog published at github.com/usnistgov/oscal-content (authoritative first-party data served from GitHub, not a .gov API host — provenance disclosed in _meta); the exact OSCAL version + last-modified are surfaced in _meta (the catalog is fetched live from the MOVING 'main' branch, so control text can shift between point releases, e.g. 5.1.1 → 5.2.0 — cite the version, not just 'Rev 5'); a WITHDRAWN control (status:'withdrawn') has statement:null and is NOT an active requirement (see incorporatedInto for what replaced it); the catalog has no query API so filtering is CLIENT-SIDE and totalAvailable is the EXACT match count; this is the REQUIREMENT text only — applicability depends on the system's FIPS-199 impact baseline (Low/Moderate/High), which the catalog does not encode (disclosed); a download failure or an implausibly-truncated catalog (< 15 families) THROWS (never a fake-empty 'control not found').",
5708
+ description: "Look up NIST SP 800-53 Rev 5 security & privacy CONTROLS (keyless) — the requirement backbone for FedRAMP / CMMC / RMF compliance work. Complements cve_lookup + cisa_kev_lookup. Retrieve by `controlId` (exact, e.g. 'AC-2', 'SC-7', 'AC-2(1)'), `family` (2-letter 'AC'/'SC'/'IA' or name substring), and/or `keyword` (case-insensitive substring over title + statement); `limit`/`offset` pagination. Each row: { id, family, title, status ('withdrawn'|null), statement (requirement prose; NULL for a WITHDRAWN control, never ''), guidance (discussion), incorporatedInto:[ids that superseded a withdrawn control], enhancements:[{id,title}] }. HONESTY: source is NIST's OFFICIAL OSCAL catalog at github.com/usnistgov/oscal-content (authoritative first-party data served from GitHub, not a .gov API host — provenance disclosed in _meta); the exact OSCAL version + last-modified are surfaced in _meta (catalog fetched live from the MOVING 'main' branch, so control text can shift between point releases — cite the version); a WITHDRAWN control has statement:null and is NOT an active requirement (see incorporatedInto for what replaced it); filtering is CLIENT-SIDE and totalAvailable is the EXACT match count; applicability depends on the system's FIPS-199 impact baseline (Low/Moderate/High), which the catalog does not encode; a download failure or implausibly-truncated catalog (< 15 families) THROWS (never fake-empty).",
5703
5709
  inputSchema: NistControlsInput,
5704
5710
  handler: (input) => nistControls.searchControls(input),
5705
5711
  }),
5706
5712
  // ━━━ NPPES NPI Registry — Healthcare-Provider Vetting (1) ━━━ ADR-0036
5707
5713
  defineTool({
5708
5714
  name: "nppes_lookup_provider",
5709
- description:
5710
- "Keyless CMS/HHS NPPES NPI Registry lookup — the authoritative PUBLIC registry of every US healthcare provider (individual NPI-1 + organization NPI-2), for VA/HHS/CMS subcontractor/provider/teaming due-diligence (validate an NPI, confirm taxonomy/specialty, enumeration status, practice state, org/name match). Host npiregistry.cms.hhs.gov/api (version=2.1). Mode is inferred from `number` (no mode flag). EXACT-NPI mode (`number` given): the NPI is CMS-Luhn-validated client-side (Luhn over 80840+first-9) ⇒ a typo'd NPI is invalid_input, NEVER a fake 'does not exist'; ★the wire query carries `number` (+version) ALONE — any co-supplied filter (last_name/state/…) is DROPPED from the wire and checked CLIENT-SIDE (disclosed in data.filterMatch:{field:bool} + data.filtersDropped), because NPPES AND-combines a number with filters and a mismatch would falsely zero a real active provider into found:false. SEARCH mode: required-one of { first_name, last_name, organization_name, taxonomy_description, city, postal_code } (state + enumeration_type are REFINERS ONLY — rejected alone); a trailing '*' wildcard on a name/org field needs ≥2 leading literal chars. Returns EXACT-mode { found, provider:{ number, enumerationType, active, status, basic{…individual OR org fields, null-never-fabricated…}, taxonomies[{code,desc,primary,state,license,taxonomyGroup}], addresses[{purpose,address1,city,state,postalCode,telephone,fax,countryCode}], practiceLocations[…same, SEPARATE from addresses], identifiers[], otherNames[], endpoints[], createdEpoch, lastUpdatedEpoch }, filterMatch? } OR SEARCH-mode { providers:[…] } + honest _meta. HONESTY: active = basic.status==='A' (a deactivated/absent NPI is NOT active); epochs are ms numeric STRINGS → number|null (null-never-0); addresses[] and practiceLocations[] are kept SEPARATE (a provider can practice in a state that appears ONLY in practiceLocations); NPPES exposes NO match total, so a full page ⇒ totalAvailable is a disclosed LOWER BOUND (totalIsLowerBound) + a ~1,200-row-per-query reach cap (limit ≤ 200, skip ≤ 1,000 — OUR policy, a PER-QUERY cap only; cross-query enumeration is not architecturally prevented). A genuine {result_count:0} ⇒ honest found:false/empty; a {Errors:[…]} 200 body (no results key) ⇒ THROWS invalid_input (never a fake empty); any 4xx/5xx/timeout/off-host-redirect ⇒ THROWS; result_count !== results.length ⇒ schema_drift. ★NOT a fitness/exclusion/licensure/sanctions determination — cross-check SAM exclusions + OFAC; individual (NPI-1) records may surface personal/home addresses + phone/fax verbatim with NO enrichment. The caveat + reach-cap disclosure ride EVERY response.",
5715
+ description: "Keyless CMS/HHS NPPES NPI Registry — every US healthcare provider (NPI-1 individual + NPI-2 organization). EXACT-NPI mode (when `number` supplied): NPI is CMS-Luhn-validated — typo'd NPI is invalid_input, NEVER a fake 'does not exist'; wire carries `number`+version ALONE — co-supplied filters are DROPPED from wire and checked CLIENT-SIDE (filterMatch:{field:bool} + filtersDropped) because NPPES AND-combines number+filters and a mismatch would falsely zero a real active provider. SEARCH mode: required-one of {first_name, last_name, organization_name, taxonomy_description, city, postal_code} (state + enumeration_type are REFINERS ONLY — rejected alone); trailing '*' wildcard needs ≥2 leading literal chars. Returns EXACT-mode { found, provider:{number, enumerationType, active, basic, taxonomies, addresses, practiceLocations, identifiers, otherNames, endpoints, createdEpoch, lastUpdatedEpoch}, filterMatch? } OR SEARCH { providers:[…] } + honest _meta. HONESTY: active = basic.status==='A'; epochs are ms numeric STRINGS → number|null; addresses[] and practiceLocations[] kept SEPARATE (a provider may appear in practiceLocations ONLY); NPPES exposes NO match total — full page → totalAvailable is a LOWER BOUND (totalIsLowerBound) + reach cap (limit ≤ 200, skip ≤ 1,000). Genuine {result_count:0} → honest found:false; {Errors:[…]} 200 body THROWS; 4xx/5xx/timeout THROW; count mismatch → schema_drift. ★NOT a fitness/exclusion/licensure/sanctions determination — cross-check SAM + OFAC; NPI-1 records may surface personal/home addresses + phone/fax verbatim.",
5711
5716
  inputSchema: NppesLookupInput,
5712
5717
  handler: (input) => nppes.lookupProvider(input),
5713
5718
  }),
@@ -5721,23 +5726,20 @@ export const TOOLS: ToolDef[] = [
5721
5726
  }),
5722
5727
  defineTool({
5723
5728
  name: "cms_query_dataset",
5724
- description:
5725
- "Query a CMS Open Payments DKAN datastore distribution by datasetId + index (keyless; openpaymentsdata.cms.gov) — the healthcare industry-financial-relationship / COI-vetting + market-intelligence lane NPPES (provider identity) cannot answer. GET /api/1/datastore/query/{datasetId}/{index} with server-side `conditions` filters, an EXACT `count`, offset/limit pagination, and a `properties` projection. Returns { datasetId, index, results (mode), fields:[{name,type,mysqlType,description}] (from the DKAN schema), rows:[…verbatim…] } + honest _meta. A confirmed target: 2025 Research Payment Data 'f0d1de67-6852-4093-a036-c9328c256a05' index 0 (count 931959; + a recipient_state='CA' condition → 92097). ★HONESTY: `count` is the EXACT grand total (P1) → totalAvailable=count + real offset pagination (NOT a page-length lower bound); `conditions` are server-side and self-policing — a valid column narrows the count, a BAD column ⇒ HTTP 400 ⇒ invalid_input, so filtersDropped is ALWAYS empty (no silent-drop path, P4); limit ≤ 500 is the HARD API cap (a higher limit ⇒ invalid_input, no silent clamp); every column is text, so amounts (total_amount_of_payment_usdollars, …) arrive as STRINGS surfaced verbatim (a missing amount is null-never-0, P3). ★results:false = a COUNT/SCHEMA-discovery mode: no rows, pagination disabled (no livelock), but the EXACT count + every column's schema returned (count=true is ALWAYS on the wire — not a caller toggle). A genuine {count:0} ⇒ honest empty; a 400 (bad column/limit) / 404 (bad datasetId/index) / HTML (SPA/WAF) / 5xx / timeout / a missing schema anchor or non-array results (in results:true) ⇒ THROW (never a fake empty). ★SSRF: datasetId (36-char lowercase UUID) + index interpolate into the URL PATH (validated before interpolation). ★PII: Open Payments is PUBLIC transparency-BY-LAW data (in-scope per the NPPES precedent) naming physicians + amounts verbatim — bounded to targeted vetting (offset ≤ 2000 reach cap), NO enrichment, NO covered_recipient_npi→NPPES auto-join. NOT a conflict-of-interest finding / fitness / exclusion determination — cross-check SAM exclusions + OFAC + the OIG-LEIE. The caveat + reach-cap disclosure ride EVERY response.",
5729
+ description: "Query a CMS Open Payments DKAN datastore distribution by datasetId + index (keyless; openpaymentsdata.cms.gov). Returns { datasetId, index, results (mode), fields:[{name,type,mysqlType,description}], rows:[…verbatim…] } + honest _meta. ★HONESTY: `count` is the EXACT grand total → totalAvailable=count + real offset pagination (NOT a page-length lower bound). `conditions` are server-side self-policing — BAD column → HTTP 400 → invalid_input; filtersDropped is ALWAYS empty (no silent-drop path). limit ≤ 500 is the HARD API cap (higher → invalid_input, no silent clamp). Every column is text, amounts arrive as STRINGS verbatim (null-never-0). ★results:false = COUNT/SCHEMA-discovery mode: no rows, pagination disabled (no livelock), EXACT count + column schema returned. Genuine {count:0} → honest empty; 400/404/HTML/5xx/timeout/missing schema/non-array → THROW. ★SSRF: datasetId (36-char UUID) + index are validated before URL interpolation. ★PII: Open Payments is PUBLIC transparency-BY-LAW data — bounded to targeted vetting (offset ≤ 2000 reach cap), NO enrichment, NO covered_recipient_npi→NPPES auto-join. NOT a conflict-of-interest / fitness / exclusion determination — cross-check SAM + OFAC + OIG-LEIE. The caveat + reach-cap disclosure ride EVERY response.",
5726
5730
  inputSchema: CmsQueryDatasetInput,
5727
5731
  handler: (input) => cms.queryDataset(input),
5728
5732
  }),
5729
5733
  // ━━━ FAC Federal Audit Clearinghouse — Single Audit audit-risk vetting (2) ━━━ ADR-0038
5730
5734
  defineTool({
5731
5735
  name: "fac_search_audits",
5732
- description:
5733
- "Search entity Single Audit summaries from the Federal Audit Clearinghouse (keyless via the api.data.gov DEMO_KEY; api.fac.gov PostgREST `general` table) — the SUBCONTRACTOR / teaming AUDIT-RISK vetting entry point (2 CFR 200 Subpart F / Single Audit Act; every entity expending ≥$750K/yr in federal awards). Structured filters (all optional, AND-combined): `auditeeUei` (12-char SAM UEI — the PRIMARY join key to SAM/USAspending/EDGAR, → auditee_uei), `auditeeState` (2-letter → auditee_state), `auditYear` (int → audit_year), `totalExpendedMin`/`totalExpendedMax` (USD → total_amount_expended gte/lte). `limit` (≤100, def 25), `offset`. Returns { audits:[{ report_id, auditee_uei, audit_year, auditee_name, auditee_ein, auditee_state, auditee_city, total_amount_expended, fac_accepted_date }] } + honest _meta. Feed a row's report_id (or the UEI) to fac_get_findings for the audit-RISK flags. ★PII: a HARDCODED select-allowlist surfaces ONLY entity + audit-summary fields and DELIBERATELY EXCLUDES the auditee's personal-contact columns (email/phone/certifying-official name) — the vetting subject is the ENTITY; there is NO caller `select`/column param. HONESTY: totalAvailable is the EXACT Content-Range total (a response header under Prefer:count=exact; a '*'/absent/non-numeric denominator ⇒ totalAvailable:null + a page-fullness hedge, NEVER 0); total_amount_expended is null-never-0 (a missing amount is null, never 0); a bad column ⇒ PostgREST 400 ⇒ invalid_input (filtersDropped is ALWAYS empty); a genuine [] ⇒ honest empty; 400/403/5xx/timeout/HTML/non-array THROW (206 = success, never a fake empty). NOT a debarment/exclusion/fitness determination — an audit finding is the auditor's opinion; cross-check SAM exclusions + OFAC. Keyless-first via DEMO_KEY (~10 req/hr shared ceiling; set DATA_GOV_API_KEY for production — never logged).",
5736
+ description: "Search entity Single Audit summaries from the Federal Audit Clearinghouse (keyless via api.data.gov DEMO_KEY; api.fac.gov PostgREST `general` table) — the SUBCONTRACTOR / teaming AUDIT-RISK vetting entry point (2 CFR 200 Subpart F / Single Audit Act; every entity expending ≥$750K/yr in federal awards). Filters (all optional, AND-combined): `auditeeUei` (12-char SAM UEI — PRIMARY join key to SAM/USAspending/EDGAR), `auditeeState` (2-letter), `auditYear` (int), `totalExpendedMin`/`totalExpendedMax` (USD). `limit` (≤100, def 25), `offset`. Returns { audits:[{ report_id, auditee_uei, audit_year, auditee_name, auditee_ein, auditee_state, auditee_city, total_amount_expended, fac_accepted_date }] } + honest _meta. Feed report_id (or UEI) to fac_get_findings for the audit-RISK flags. ★PII: a HARDCODED select-allowlist surfaces ONLY entity + audit-summary fields and DELIBERATELY EXCLUDES personal-contact columns — NO caller `select`/column param. HONESTY: totalAvailable is the EXACT Content-Range total (a response header under Prefer:count=exact; '*'/absent/non-numeric denominator → totalAvailable:null + page-fullness hedge, NEVER 0); total_amount_expended is null-never-0; a bad column → PostgREST 400 → invalid_input (filtersDropped ALWAYS empty); genuine [] → honest empty; 400/403/5xx/timeout/HTML/non-array THROW. NOT a debarment/exclusion/fitness determination — cross-check SAM + OFAC. Keyless-first via DEMO_KEY (~10 req/hr shared; set DATA_GOV_API_KEY for production — never logged).",
5734
5737
  inputSchema: FacSearchAuditsInput,
5735
5738
  handler: (input) => fac.searchAudits(input),
5736
5739
  }),
5737
5740
  defineTool({
5738
5741
  name: "fac_get_findings",
5739
- description:
5740
- "Drill into the audit-RISK findings for an entity from the Federal Audit Clearinghouse (keyless via the api.data.gov DEMO_KEY; api.fac.gov PostgREST `findings` table) — the risk-detail step after fac_search_audits. At least ONE of `auditeeUei` (12-char UEI → auditee_uei) or `reportId` (→ report_id, from a fac_search_audits row) is REQUIRED (an empty query is refused, never a whole-table scan); optional `auditYear` (int), `limit` (≤100, def 50), `offset`. Returns { findings:[{ report_id, auditee_uei, audit_year, award_reference, reference_number, is_material_weakness, is_modified_opinion, is_questioned_costs, is_repeat_finding, is_significant_deficiency, is_other_findings, is_other_matters, type_requirement, prior_finding_ref_numbers, riskFlags:{materialWeakness, modifiedOpinion, questionedCosts, repeatFinding, significantDeficiency, otherFindings, otherMatters} }] } + honest _meta. ★RISK-FLAG HONESTY: the is_* flags are surfaced VERBATIM as the auditor reported them (\"Y\"/\"N\") PLUS a typed riskFlags tri-state (\"Y\"→true / \"N\"→false / blank/absent/other → null=UNKNOWN) — a null flag is NEVER rendered as false/\"no material weakness\" (the false-CLEAR class). ★EMPTY ≠ CLEAN: an empty findings list does NOT confirm a clean audit — the entity may not have filed a Single Audit (below the $750K threshold), the audit may predate FAC coverage, or the UEI may be wrong; a disclosure note fires on any empty result telling you to confirm an ACCEPTED audit exists via fac_search_audits. ★PII: a HARDCODED select-allowlist (NO caller column param) surfaces only entity + audit-risk fields — no personal contact. totalAvailable is the EXACT Content-Range total ('*'/absent ⇒ null + hedge, never 0); a bad column ⇒ 400 ⇒ invalid_input; 400/403/5xx/timeout/HTML/non-array THROW (206 = success). NOT a debarment/determination — cross-check SAM exclusions + OFAC + the specific finding text. Keyless-first via DEMO_KEY (~10 req/hr; set DATA_GOV_API_KEY — never logged).",
5742
+ description: "Drill into audit-RISK findings for an entity from the Federal Audit Clearinghouse (keyless via api.data.gov DEMO_KEY; api.fac.gov PostgREST `findings` table) — the risk-detail step after fac_search_audits. At least ONE of `auditeeUei` (12-char UEI) or `reportId` is REQUIRED (empty query refused); optional `auditYear`, `limit` (≤100, def 50), `offset`. Returns { findings:[{ report_id, auditee_uei, audit_year, award_reference, reference_number, is_material_weakness, is_modified_opinion, is_questioned_costs, is_repeat_finding, is_significant_deficiency, is_other_findings, is_other_matters, type_requirement, prior_finding_ref_numbers, riskFlags:{materialWeakness, modifiedOpinion, questionedCosts, repeatFinding, significantDeficiency, otherFindings, otherMatters} }] } + honest _meta. ★RISK-FLAG HONESTY: is_* flags surfaced VERBATIM (\"Y\"/\"N\") PLUS typed riskFlags tri-state (\"Y\"→true / \"N\"→false / blank/absent → null=UNKNOWN) — null NEVER rendered as false (the false-CLEAR class). ★EMPTY ≠ CLEAN: empty findings does NOT confirm a clean audit — the entity may not have filed a Single Audit (below the $750K threshold), may predate FAC coverage, or UEI wrong; a disclosure note fires on any empty result; on empty, confirm an ACCEPTED audit via fac_search_audits. ★PII: HARDCODED select-allowlist, entity + audit-risk fields only. totalAvailable is EXACT Content-Range total ('*'/absent → null + hedge, never 0); 400/403/5xx/timeout/HTML/non-array THROW. NOT a debarment/determination — cross-check SAM + OFAC. DEMO_KEY ~10 req/hr — set DATA_GOV_API_KEY.",
5741
5743
  inputSchema: FacGetFindingsInput,
5742
5744
  handler: (input) => fac.getFindings(input),
5743
5745
  }),
@@ -5816,22 +5818,19 @@ export const TOOLS: ToolDef[] = [
5816
5818
  }),
5817
5819
  defineTool({
5818
5820
  name: "edgar_filing_index",
5819
- description:
5820
- "Bulk cross-filer SEC filing index for a quarter (keyless, from the www.sec.gov EDGAR full-index master.idx). Reads the WHOLE quarter's index (every filer's every filing — CIK|Company|Form|Date|Filename, ~370K rows), FULL-SCANS it, and returns offset-paginated filings matching CLIENT-SIDE filters with the EXACT total. Input `year` (>=1993, <= current year), `quarter` (1..4); optional `formType` (exact form, e.g. '8-K'), `cik` (numeric, leading-zero-safe), `companyContains` (LITERAL case-insensitive substring), `dateFrom`/`dateTo` (ISO YYYY-MM-DD), `limit` (<=1000, def 100), `offset`. Returns { year, quarter, indexFile, returned, totalAvailable, filings:[{ cik, cikPadded, companyName, formType, dateFiled, filename, filingUrl }] }. This is the BULK-ENUMERATION primitive (the per-filer edgar tools need a CIK you already hold; this sweeps a whole quarter by form/date/company, e.g. 'every 8-K in 2024 Q1'). HONESTY: totalAvailable is the EXACT match count over the full quarter scan — never a page length, never a byte-capped subset (SEC ignores HTTP Range); a 0-match result is a genuine EXACT ZERO (complete:true), NOT a truncation; a bounds-valid but unpublished quarter returns HTTP 403 and is surfaced as an AMBIGUOUS both-causes error (quarter-not-published OR the 10 req/s rate-block), never a bare rate-limit and never a fake-empty; a non-index / all-malformed body is refused as schema_drift; a future year / bad quarter is rejected pre-fetch (invalid_input, 0 fetch). The CURRENT quarter grows daily (totalAvailable is exact AS-OF-snapshot). filingUrl is a resolvable archive URL. NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — there is no authoritative CIK↔UEI join.",
5821
+ description: "Bulk sweep — per-filer edgar tools need a CIK. Bulk cross-filer SEC filing index for a quarter (keyless; www.sec.gov EDGAR full-index master.idx). Reads the WHOLE quarter's index (~370K rows: every filer's every filing — CIK|Company|Form|Date|Filename), full-scans it, and returns offset-paginated filings matching CLIENT-SIDE filters with the EXACT total. Input: `year` (≥1993, ≤current year), `quarter` (1..4); optional `formType` (exact, e.g. '8-K'), `cik` (numeric, leading-zero-safe), `companyContains` (LITERAL case-insensitive substring), `dateFrom`/`dateTo` (ISO YYYY-MM-DD), `limit` (≤1000, def 100), `offset`. Returns { year, quarter, indexFile, returned, totalAvailable, filings:[{ cik, cikPadded, companyName, formType, dateFiled, filename, filingUrl }] }. HONESTY: totalAvailable is the EXACT match count over the full quarter scan — never a page length, never a byte-capped subset (SEC ignores HTTP Range). A 0-match result is a genuine EXACT ZERO (complete:true), NOT a truncation. A bounds-valid but unpublished quarter returns HTTP 403 and is surfaced as an AMBIGUOUS error (quarter-not-published OR the 10 req/s rate-block) — never a bare rate-limit and never a fake-empty. A non-index or all-malformed body → schema_drift. A future year / bad quarter → invalid_input pre-fetch. The CURRENT quarter grows daily (totalAvailable is exact as-of-snapshot). filingUrl is a resolvable archive URL. NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — no authoritative CIK↔UEI join.",
5821
5822
  inputSchema: EdgarFilingIndexInput,
5822
5823
  handler: (input) => edgar.filingIndex(input),
5823
5824
  }),
5824
5825
  defineTool({
5825
5826
  name: "edgar_daily_filing_index",
5826
- description:
5827
- "Per-DAY cross-filer SEC filing index (keyless, from the www.sec.gov EDGAR daily-index master.YYYYMMDD.idx). The per-day sibling of edgar_filing_index (~30× smaller): reads ONE calendar day's index (every filer's every filing that day — CIK|Company|Form|Date|File Name, ~8K rows), FULL-SCANS it, and returns offset-paginated filings matching CLIENT-SIDE filters with the EXACT total. Answers the monitoring/alerting question the quarterly tool cannot ('every 8-K filed on 2024-01-03', 'watch a CIK day-by-day'). Input `date` (required ISO YYYY-MM-DD, >=1994-01-01, not future); optional `formType` (exact form, e.g. '8-K'), `cik` (numeric, leading-zero-safe), `companyContains` (LITERAL case-insensitive substring), `limit` (<=1000, def 100), `offset`. Returns { found, date, year, quarter, indexFile, returned, totalAvailable, filings:[{ cik, cikPadded, companyName, formType, dateFiled, filename, filingUrl }] }. HONESTY: totalAvailable is the EXACT match count over the full day scan — never a page length, never a byte-capped subset (SEC ignores HTTP Range). The daily-index's pervasive-403 empty model is disambiguated via the quarter's index.json existence oracle, RECENCY-AWARE: a day NEWER than the newest published index (weekend/holiday/not-yet-disseminated recent trading day) ⇒ found:false, complete:FALSE, retryable not-yet-disseminated note (NEVER a confident empty); an unlisted day INSIDE the covered range (a real weekend/holiday) ⇒ found:false, complete:true genuine-absent; a LISTED day whose .idx 403s ⇒ honest rate_limited; the oracle itself inconclusive ⇒ ambiguous both-causes upstream_unavailable. A non-real/future date is rejected pre-fetch (invalid_input, 0 fetch); a non-index / all-malformed body is refused as schema_drift. dateFiled is normalized to ISO from the compact YYYYMMDD column. NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — there is no authoritative CIK↔UEI join.",
5827
+ description: "Per-day sibling of edgar_filing_index. Per-day cross-filer SEC filing index (keyless; www.sec.gov EDGAR daily-index master.YYYYMMDD.idx) — reads ONE calendar day's index (~8K rows), full-scans it, and returns offset-paginated filings matching CLIENT-SIDE filters with the EXACT total. Answers the monitoring/alerting question ('every 8-K filed on 2024-01-03'). Input: `date` (required ISO YYYY-MM-DD, ≥1994-01-01, not future); optional `formType` (exact), `cik` (numeric), `companyContains` (LITERAL case-insensitive), `limit` (≤1000, def 100), `offset`. Returns { found, date, year, quarter, indexFile, returned, totalAvailable, filings:[{ cik, cikPadded, companyName, formType, dateFiled, filename, filingUrl }] }. HONESTY: totalAvailable is EXACT match count — never a page length. The daily-index pervasive-403 model is disambiguated via the quarter's index.json existence oracle, RECENCY-AWARE: a day NEWER than the newest published index → found:false, complete:FALSE, retryable not-yet-disseminated note (NEVER a confident empty); an unlisted day INSIDE the covered range (real weekend/holiday) → found:false, complete:true; a LISTED day whose .idx 403s → honest rate_limited; oracle inconclusive → ambiguous upstream_unavailable. A non-real/future date → invalid_input pre-fetch; non-index/all-malformed body → schema_drift. dateFiled normalized from compact YYYYMMDD to ISO. NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — no authoritative CIK↔UEI join.",
5828
5828
  inputSchema: EdgarDailyFilingIndexInput,
5829
5829
  handler: (input) => edgar.dailyFilingIndex(input),
5830
5830
  }),
5831
5831
  defineTool({
5832
5832
  name: "edgar_company_concept",
5833
- description:
5834
- "One filer × one XBRL concept × the COMPLETE reported time-series (keyless, from data.sec.gov companyconcept). The focused financial-TREND / entity-vetting primitive BETWEEN edgar_company_facts (many curated concepts for one filer) and edgar_xbrl_frames (one concept across ALL filers for one period) — 'track THIS filer's Assets/Revenues/NetIncomeLoss OVER TIME, and was it ever revised?'. Input `cikOrTicker` (CIK or resolvable ticker/name), `concept` (EXACT alnum XBRL tag, e.g. 'Assets'), optional `taxonomy` (us-gaap|dei|ifrs-full, def us-gaap), `unit` (CLIENT-SIDE key filter), `form`/`fy` (client-side), `canonicalOnly` (def false), `limit`/`offset`. Returns { found, cik, entityName, taxonomy, concept, label, description, unitsAvailable:[{unit,count}], rows:[{ unit, start, end, val, accn, fy, fp, form, filed, frame, canonical }] }. HONESTY: (M1) period identity is the (start,end) PAIR — every row carries `start` (null for INSTANT concepts, the ISO date for DURATION/flow concepts); the SAME `end` with a DIFFERENT `start` is a different-duration fact (a 3-month quarter vs the 12-month year), NOT a revision — a revision is only multiple rows sharing the same (start,end) with a differing accn/filed/val. DEFAULT returns ALL rows incl. the amendment/restatement history + a per-row `canonical` (frame-tagged = SEC's consolidated value); `canonicalOnly:true` dedups to one canonical row per (unit,start,end), fully disclosed, never a silent drop. Every row is unit-tagged (a USD amount is NEVER conflated with a share count); unitsAvailable discloses ALL units with their RAW counts even under a unit filter; val is null-never-0. A bad CIK/taxonomy/concept ⇒ upstream 404 ⇒ found:false (NEVER a fabricated val:0); a 5xx/timeout/non-JSON/units-shape-drift THROWS; a `unit` not present ⇒ honest empty + the available-units note (unit is CLIENT-SIDE, not a path segment). cik/taxonomy/concept are validated path segments (regex+enum, re-checked pre-fetch) — no injection surface. NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — there is no authoritative CIK↔UEI join.",
5833
+ description: "One filer × one XBRL concept × complete reported time-series (keyless; data.sec.gov companyconcept), including amendment/restatement history. Sits between edgar_company_facts (many concepts, one filer) and edgar_xbrl_frames (one concept, all filers). start=null for INSTANT concepts. Input: `cikOrTicker`, `concept` (EXACT alnum XBRL tag, e.g. 'Assets'), optional `taxonomy` (us-gaap|dei|ifrs-full, def us-gaap), `unit` (CLIENT-SIDE filter), `form`/`fy` (client-side), `canonicalOnly` (def false), `limit`/`offset`. Returns { found, cik, entityName, taxonomy, concept, label, description, unitsAvailable:[{unit,count}], rows:[{unit, start, end, val, accn, fy, fp, form, filed, frame, canonical}] }. HONESTY M1: period identity is the (start,end) PAIR — the SAME `end` with a DIFFERENT `start` is a different-duration fact (3-month vs 12-month), NOT a revision; a revision is multiple rows sharing the same (start,end) with differing accn/filed/val. DEFAULT returns ALL rows including restatement history + per-row `canonical`; `canonicalOnly:true` dedupes to one canonical row per (unit,start,end), fully disclosed, never a silent drop. Every row is unit-tagged; unitsAvailable discloses ALL units even under a unit filter; val is null-never-0. A bad CIK/taxonomy/concept → 404 → found:false (NEVER fabricated val:0); 5xx/timeout/non-JSON/shape-drift THROWS; `unit` not present → honest empty + available-units note (unit is CLIENT-SIDE, not a path segment). cik/taxonomy/concept are validated path segments (no injection). NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — no authoritative CIK↔UEI join.",
5835
5834
  inputSchema: EdgarCompanyConceptInput,
5836
5835
  handler: (input) => edgar.companyConcept(input),
5837
5836
  }),
@@ -5882,30 +5881,26 @@ export const TOOLS: ToolDef[] = [
5882
5881
  }),
5883
5882
  defineTool({
5884
5883
  name: "fdic_bank_failures",
5885
- description:
5886
- "Historical FDIC-insured bank failures & assistance transactions (keyless FDIC BankFind, api.fdic.gov/banks/failures) — B2G counterparty / entity due-diligence: a failed or FDIC-assisted institution is a red flag, and CERT links a failure back to fdic_search_institutions / fdic_institution_financials. Exact-key filters: `state` (2-letter → PSTALP — NOTE the /failures state field is PSTALP, NOT STALP), `failYear` (→ FAILYR; e.g. 2023 → the 5 real 2023 failures incl. Silicon Valley Bank & First Republic Bank), `cert` (→ CERT, the STABLE entity key). `limit` (≤1000, def 100), `offset` (≤100000), `sortBy` (allowlisted enum FAILDATE/COST/QBFASSET/QBFDEP/NAME/FAILYR, def FAILDATE), `sortOrder` (def DESC → most-recent first). Returns { failures:[{ name, cert, failDate, failYear, city, state, resolutionType, resolutionFund, estimatedLossUSD, depositsUSD, assetsUSD, id }] }. NO name/city filter — FDIC's /failures `search` param is IGNORED (it returns the whole dataset), so name/city are SHOWN in each row but NOT searchable; to find a specific bank's failure, resolve its CERT via fdic_search_institutions then filter here by `cert`. HONESTY: totalAvailable is the EXACT meta.total (stable across offset — never the page length); failDate is normalized from FDIC's M/D/YYYY to ISO YYYY-MM-DD (an unrecognized value is surfaced raw + disclosed, never nulled/fabricated); COST/QBFDEP/QBFASSET are $thousands normalized to whole USD ×1000 (null-never-0 — a genuine 0 = a fully-assisted no-loss stays 0, a NEGATIVE COST = a net DIF recovery/gain not a loss, absent → null); the ONLY honest empty is meta.total:0/data:[] ⇒ complete:true/total:0, every other envelope (400 errors[]/404/non-JSON/missing meta or data) THROWS (never a fake empty); the point-in-time snapshot build time is disclosed. NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
5884
+ description: "Historical FDIC-insured bank failures & assistance transactions (keyless; api.fdic.gov/banks/failures). CERT links a failure back to fdic_search_institutions / fdic_institution_financials. Filters: `state` (2-letter → PSTALP — NOTE: /failures state field is PSTALP, NOT STALP), `failYear` (→FAILYR), `cert` (→CERT, the STABLE entity key). `limit` (≤1000), `offset` (≤100000), `sortBy` (FAILDATE/COST/QBFASSET/QBFDEP/NAME/FAILYR, def FAILDATE), `sortOrder` (def DESC). Returns { failures:[{ name, cert, failDate, failYear, city, state, resolutionType, resolutionFund, estimatedLossUSD, depositsUSD, assetsUSD, id }] }. NO name/city filter — FDIC /failures `search` param is IGNORED (returns the whole dataset); to find a specific bank's failure, resolve its CERT via fdic_search_institutions. HONESTY: totalAvailable is EXACT meta.total (stable across offset). failDate normalized from FDIC's M/D/YYYY to ISO YYYY-MM-DD (unrecognized → surfaced raw + disclosed, never nulled). COST/QBFDEP/QBFASSET are $thousands normalized to whole USD ×1000 (null-never-0 — genuine 0 = a no-loss assisted transaction stays 0; NEGATIVE COST = a net DIF recovery/gain, not a loss; absent → null). ONLY honest empty is meta.total:0/data:[] → complete:true/total:0; any other envelope (400 errors[]/404/non-JSON/missing meta or data) THROWS (never fake-empty). The point-in-time snapshot build time is disclosed. NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
5887
5885
  inputSchema: FdicBankFailuresInput,
5888
5886
  handler: (input) => fdic.bankFailures(input),
5889
5887
  }),
5890
5888
  defineTool({
5891
5889
  name: "fdic_institution_history",
5892
- description:
5893
- "Institution-level STRUCTURAL-CHANGE event log for FDIC-insured banks (keyless FDIC BankFind, api.fdic.gov/banks/history) — the full lineage of mergers, absorptions, consolidations, failures, name/location/charter/regulator changes, branch open/close, trust-power grants & FRS-membership changes. Completes the FDIC entity cluster (directory + financials + failures + history). Killer feature: CERT-linked MERGER LINEAGE — a merger/failure row carries the acquiring / outgoing / surviving institution's CERT + name, each linking back to fdic_search_institutions / fdic_institution_financials / fdic_bank_failures. Exact-key filters (all optional, AND-combined): `cert` (→ CERT, the STABLE entity key & PRIMARY lookup; e.g. 3510 → Bank of America's 13,794 rows), `changeCode` (→ CHANGECODE; e.g. 223 = merger, 211 = failure, 721 = branch closing, 520 = location change), `effYear` (→ EFFYEAR), `state` (2-letter → PSTALP — NOTE the /history state field is PSTALP, NOT STALP). `limit` (≤1000, def 100), `offset` (≤100000), `sortBy` (allowlisted enum EFFDATE/PROCDATE/CHANGECODE/TRANSNUM, def EFFDATE), `sortOrder` (def DESC → newest change first). Returns { history:[{ cert, instName, state, changeCode, changeDescription, effectiveDate, processDate, effYear, transNum, acquirerCert, acquirerName, outgoingCert, outgoingName, survivingCert, survivingName, id }] }. NO name/city filter — FDIC's /history `search` param returns 0 for INSTNAME (a false-empty), so names are SHOWN in each row but NOT searchable; to find a specific bank's history, resolve its CERT via fdic_search_institutions then filter here by `cert`. HONESTY: totalAvailable is the EXACT meta.total (stable across offset — never the page length); changeDescription is FDIC's OWN co-served CHANGECODE_DESC passed through verbatim (the numeric changeCode is authoritative — never a hand-map); effectiveDate/processDate are normalized from FDIC's YYYY-MM-DDT00:00:00 to ISO YYYY-MM-DD (an unrecognized value is surfaced raw + disclosed, never nulled/fabricated); the acquirer/outgoing/surviving CERTs are null on a non-merger event (null-never-0 — a real absence, never a fabricated 0; *_UNINUM's 0 sentinel is NOT surfaced); the ONLY honest empty is meta.total:0/data:[] ⇒ complete:true/total:0, every other envelope (400 errors[]/404/non-JSON/missing meta or data) THROWS (never a fake empty); the point-in-time snapshot build time is disclosed. NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
5890
+ description: "Institution-level STRUCTURAL-CHANGE event log for FDIC-insured banks (keyless; api.fdic.gov/banks/history) — mergers, absorptions, consolidations, failures, name/location/charter/regulator changes, branch open/close, trust-power grants & FRS-membership. CERT-linked merger lineage: each row carries acquiring/outgoing/surviving institution CERT + name, linking to fdic_search_institutions / fdic_bank_failures. Filters (all optional, AND-combined): `cert` (→CERT, PRIMARY lookup), `changeCode` (→CHANGECODE; e.g. 223=merger, 211=failure, 721=branch closing), `effYear` (→EFFYEAR), `state` (2-letter → PSTALP — NOTE: /history uses PSTALP, NOT STALP). `limit` (≤1000), `offset` (≤100000), `sortBy` (EFFDATE/PROCDATE/CHANGECODE/TRANSNUM), `sortOrder`. Returns { history:[{ cert, instName, state, changeCode, changeDescription, effectiveDate, processDate, effYear, transNum, acquirerCert, acquirerName, outgoingCert, outgoingName, survivingCert, survivingName, id }] }. NO name/city filter — FDIC /history `search` returns 0 for INSTNAME (a false-empty); resolve CERT via fdic_search_institutions first. HONESTY: totalAvailable is EXACT meta.total; changeDescription is FDIC's CHANGECODE_DESC verbatim (changeCode is authoritative — never hand-mapped); effectiveDate/processDate normalized to ISO YYYY-MM-DD (unrecognized → surfaced raw); acquirer/outgoing/surviving CERTs null on non-merger events (null-never-0); ONLY honest empty is meta.total:0/data:[] → complete:true/total:0; any other envelope THROWS (never fake-empty). NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
5894
5891
  inputSchema: FdicInstitutionHistoryInput,
5895
5892
  handler: (input) => fdic.institutionHistory(input),
5896
5893
  }),
5897
5894
  defineTool({
5898
5895
  name: "fdic_industry_summary",
5899
- description:
5900
- "FDIC industry & state banking-sector ANNUAL AGGREGATES — the FDIC's own roll-ups (keyless FDIC BankFind, api.fdic.gov/banks/summary). The FIRST aggregate/statistical FDIC tool (the other 4 are per-ENTITY, keyed on CERT): total assets, deposits, net income, equity & net interest income + structural counts (institutions, offices, branches, employees) for the whole US banking industry OR one state/territory in one year, split by charter class. Answers 'how big is the US (or a state's) banking industry this year, and how many institutions?' — a question the entity tools cannot express without summing thousands of rows. Exact-key filters (all optional, AND-combined): `year` (→ YEAR; e.g. 2023 → 121 rows), `state` (2-or-3-letter → STALP — NOTE the /summary state field is STALP, NOT PSTALP; accepts a jurisdiction code TX/CA/DC/GU/PR… OR a ROLL-UP code USA/US/OT/PI), `charterClass` (CB = commercial banks, SI = savings institutions; omit for both — there is NO combined row). `limit` (≤1000, def 100), `offset` (≤100000), `sortBy` (allowlisted enum YEAR/ASSET/DEP/NETINC/BANKS, def YEAR), `sortOrder` (def DESC → newest year / largest first). Returns { summary:[{ year, charterClass, charterClassCode, geography, stateCode, stateFips, scope, isRollup, institutionCount, officeCount, branchCount, employeeCount, totalAssetsUSD, totalDepositsUSD, netIncomeUSD, totalEquityUSD, netInterestIncomeUSD, id }] }. ★ROLL-UP HONESTY: each row crosses charter × geography; STALP ∈ {USA,US,OT,PI} are GEOGRAPHIC AGGREGATES (scope national_total/national_states_dc/territories_total/pacific_islands, isRollup:true), every other STALP is a jurisdiction (isRollup:false) — NEVER sum a roll-up row with jurisdiction rows or across scopes (national_total = national_states_dc + territories_total; a geography's total = its CB row + its SI row), read the national_total (USA) row directly for one national figure; a roll-up is NOT a state. ★NIM is net interest INCOME (a $ sum surfaced as netInterestIncomeUSD), NOT the margin ratio; this endpoint has NO ratio fields (ROA/ROE — derive from netIncomeUSD/totalAssetsUSD/totalEquityUSD). NO name/city filter — FDIC's /summary `search` param is ignored (returns the whole year); drill to institutions via fdic_search_institutions. HONESTY: totalAvailable is the EXACT meta.total (stable across offset — never the page length); money (ASSET/DEP/NETINC/EQ/NIM) is $thousands → whole USD ×1000 (null-never-0 — a genuine 0 like American Samoa's zero commercial banks stays 0, absent → null), counts (BANKS/OFFICES/BRANCHES/employees) pass through un-scaled (a count ×1000 is a fabrication); a non-int year is rejected pre-fetch (a malformed year is a live HTTP-200 total:0 false-empty); the ONLY honest empty is meta.total:0/data:[] ⇒ complete:true/total:0, every other envelope (400 errors[]/404/non-JSON/missing meta or data) THROWS (never a fake empty); the point-in-time snapshot build time is disclosed. NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
5896
+ description: "FDIC banking-sector ANNUAL AGGREGATES — total assets, deposits, net income, equity & net interest income + institution/office/branch/employee counts for the whole US OR one state, split by charter class (keyless; api.fdic.gov/banks/summary). Filters (all optional): `year` (→YEAR), `state` (→STALP — NOTE: /summary uses STALP, NOT PSTALP; accepts TX/CA/DC/GU/PR or ROLL-UP codes USA/US/OT/PI), `charterClass` (CB=commercial, SI=savings; omit for both). `limit` (≤1000), `offset` (≤100000), `sortBy` (YEAR/ASSET/DEP/NETINC/BANKS), `sortOrder`. Returns { summary:[{ year, charterClass, charterClassCode, geography, stateCode, stateFips, scope, isRollup, institutionCount, officeCount, branchCount, employeeCount, totalAssetsUSD, totalDepositsUSD, netIncomeUSD, totalEquityUSD, netInterestIncomeUSD, id }] }. ★ROLL-UP HONESTY: STALP ∈ {USA,US,OT,PI} are GEOGRAPHIC AGGREGATES (isRollup:true). NEVER sum a roll-up row with jurisdiction rows or across scopes — USA is the one national figure; a roll-up is NOT a state. ★netInterestIncomeUSD is net interest INCOME ($ sum), NOT the margin ratio; NO ratio fields (ROA/ROE); derive from netIncomeUSD/totalAssetsUSD/totalEquityUSD. NO name/city filter — FDIC /summary `search` is ignored; drill via fdic_search_institutions. HONESTY: totalAvailable is EXACT meta.total; money ($thousands → whole USD ×1000, null-never-0; genuine 0 stays 0; absent → null); counts pass through unscaled; non-int year rejected pre-fetch; ONLY honest empty is meta.total:0/data:[] → complete:true/total:0; any other envelope THROWS. NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
5901
5897
  inputSchema: FdicIndustrySummaryInput,
5902
5898
  handler: (input) => fdic.industrySummary(input),
5903
5899
  }),
5904
5900
  // ━━━ FDIC BankFind Suite — WITHIN-SOURCE DEPTH: counterparty risk ratios + branch deposits (2) ━━━ ADR-0040
5905
5901
  defineTool({
5906
5902
  name: "fdic_risk_ratios",
5907
- description:
5908
- "FDIC counterparty RISK RATIOS for ONE institution by certificate number (keyless FDIC BankFind, api.fdic.gov/banks/financials) — the SOUNDNESS lane the balance-sheet tools cannot express: profitability (ROA/pretax ROA/ROE), net interest margin, efficiency ratio, asset quality (net charge-offs to loans), capital adequacy (leverage, tier-1 risk-based, total risk-based ratios) + the tier-1 capital LEVEL. Input `cert` (REQUIRED FDIC certificate number, from fdic_search_institutions), `reportDate` (optional YYYYMMDD quarter-end → REPDTE; omit for the full quarterly time-series), `limit` (≤1000, def 100), `offset` (≤100000), `sortBy` (allowlisted enum REPDTE/ROA/ROE/RBCRWAJ/EEFFR, def REPDTE), `sortOrder` (def DESC → newest quarter first). Returns { cert, ratios:[{ cert, reportDate, cblrFramework, returnOnAssetsPct, preTaxReturnOnAssetsPct, returnOnEquityPct, netInterestMarginPct, efficiencyRatioPct, netChargeOffsToLoansPct, leverageRatioPct, tier1RiskBasedCapitalRatioPct, totalRiskBasedCapitalRatioPct, tier1CapitalUSD, id }] }. ★UNITS-IN-THE-KEY: every *Pct field is an FDIC-published PERCENTAGE surfaced VERBATIM (no scaling, no recompute) — do NOT read it as a dollar amount or ×1000-scale it; tier1CapitalUSD is a DOLLAR amount (FDIC publishes it in $thousands, normalized ×1000). ★NULL-NEVER-0: a not-reported ratio is null (never 0% — a false 'no return / no capital'). ★CBLR (community-bank-leverage) banks (cblrFramework:true) do NOT report the risk-based capital ratios — FDIC returns a literal 0 for the total risk-based ratio, which this tool maps to null for BOTH tier1RiskBasedCapitalRatioPct and totalRiskBasedCapitalRatioPct (a null there is a normal framework artifact, read alongside leverageRatioPct — NOT a 0% capital red flag). No ratio is recomputed; each is exactly FDIC's published Call-Report figure. HONESTY: totalAvailable is the EXACT meta.total (stable across offset); the ONLY honest empty is meta.total:0/data:[] ⇒ complete:true/total:0, every other envelope (400 errors[]/404/non-JSON/missing meta or data) THROWS (never a fake empty); the snapshot build time is disclosed. NOTE: reported regulatory metrics, NOT a soundness rating or failure prediction; FDIC keys on CERT, not SAM UEI/DUNS.",
5903
+ description: "FDIC risk ratios for ONE institution by certificate number (keyless; api.fdic.gov/banks/financials) — profitability (ROA/pretax ROA/ROE), net interest margin, efficiency ratio, asset quality (net charge-offs to loans), capital adequacy (leverage, tier-1 risk-based, total risk-based ratios) + tier-1 capital level. Input: `cert` (REQUIRED FDIC certificate, from fdic_search_institutions), `reportDate` (optional YYYYMMDD quarter-end → REPDTE; omit for full quarterly time-series), `limit` (≤1000), `offset`, `sortBy` (REPDTE/ROA/ROE/RBCRWAJ/EEFFR), `sortOrder`. Returns { cert, ratios:[{ cert, reportDate, cblrFramework, returnOnAssetsPct, preTaxReturnOnAssetsPct, returnOnEquityPct, netInterestMarginPct, efficiencyRatioPct, netChargeOffsToLoansPct, leverageRatioPct, tier1RiskBasedCapitalRatioPct, totalRiskBasedCapitalRatioPct, tier1CapitalUSD, id }] }. ★UNITS: every *Pct field is FDIC-published PERCENTAGE verbatim (no scaling); tier1CapitalUSD is $thousands × 1000. ★NULL-NEVER-0: not-reported ratio is null (never 0). ★CBLR banks (cblrFramework:true) do NOT report risk-based capital ratios — FDIC returns literal 0 for totalRiskBased only; this tool maps 0→null for BOTH tier1RiskBased and totalRiskBased; null is a framework artifact, not a 0% red flag — read alongside leverageRatioPct. No ratio is recomputed; each is FDIC's published Call-Report figure verbatim. HONESTY: totalAvailable is EXACT meta.total; ONLY honest empty is meta.total:0/data:[] → complete:true/total:0; any other envelope THROWS. NOTE: regulatory metrics, NOT a soundness rating. FDIC keys on CERT, not SAM UEI/DUNS.",
5909
5904
  inputSchema: FdicRiskRatiosInput,
5910
5905
  handler: (input) => fdic.riskRatios(input),
5911
5906
  }),
@@ -5919,8 +5914,7 @@ export const TOOLS: ToolDef[] = [
5919
5914
  // ━━━ USITC Harmonized Tariff Schedule — keyless import-tariff / duty-rate lookup (1) ━━━ ADR-0039
5920
5915
  defineTool({
5921
5916
  name: "hts_lookup",
5922
- description:
5923
- "Look up US import-tariff classification + duty rates from the USITC Harmonized Tariff Schedule (keyless; hts.usitc.gov/reststop/search) — the IMPORT-TARIFF / supply-chain PRICE lane a product-reseller / supply-chain bidder needs to price a hardware or commodity contract (extends the THIN Price lane with a NON-labor cost input, a sibling of gsa_benchmark_labor_rates). A single `query` serves BOTH modes: a KEYWORD (e.g. 'laptop', 'cotton shirt') OR an HTS number (e.g. '8471.30' / '8471.30.01.00') — both ride the `keyword=` search. Returns { query, lines:[{ htsno, statisticalSuffix, indent, description, units, columnOneGeneral, specialPreferential, columnTwo, additionalDuties, footnotes, quotaQuantity, effectivePeriod, status, isChapter99 }] } + honest _meta. ★DUTY-RATE HONESTY (the crux): columnOneGeneral (Column-1 General), specialPreferential (Special/preferential/FTA), and columnTwo (Column-2) are AUTHORITATIVE VERBATIM TEXT surfaced as strings — 'Free', a percentage ('35%'), a specific rate ('0.47¢/kg'), a compound/range, or null — NEVER coerced to a number (a coerced 0/NaN would fabricate a false 'duty-free'); an empty Special ('') → null = NO special-program rate published (NEVER read as Free). ★HIERARCHY (M1): a lookup returns rows across levels; the rate is stated ONCE at a shallower level (usually the 6/8-digit subheading) and inherits DOWNWARD to the blank statistical-suffix lines — to find a specific line's rate, read UP to the nearest ANCESTOR line (shallower indent, same htsno prefix) with a non-empty rate; a blank deepest line is NOT no/unknown duty. ★ADDITIONAL DUTIES (S1): the per-line `additionalDuties` is frequently null even when Section 301/232 duties apply — the real additional duty rides the Chapter-99 rows (isChapter99:true, htsno beginning '99') returned alongside the base line + the footnotes; they STACK on the base rate. ★COMPLETENESS (M2): the endpoint returns the FULL match array with NO server-side total and NO working pagination (offset is IGNORED) → totalAvailable is the EXACT served array length and paging is CLIENT-SIDE; there is no fixed cap (a single-char/common fragment can return 10,000–16,000+ rows / several MB), so `query` must be ≥3 non-whitespace chars (a 1–2 char query is rejected invalid_input before the fetch). `limit` (≤200, def 50), `offset`. A no-match ⇒ honest empty; a 404/5xx/timeout/non-array/HTML(→schema_drift) ⇒ THROWS (never a fake empty); a transient 400 on the validated query ⇒ upstream_unavailable (retryable). NOT a binding CBP classification ruling and NOT a landed-cost quote — the duty owed depends on country of origin + trade program + Section 301/232 / Chapter-99 additional duties + footnotes; confirm via CBP (CROSS / eRulings). The not-a-ruling caveat rides EVERY response.",
5917
+ description: "Look up US import-tariff classification + duty rates from the USITC Harmonized Tariff Schedule (keyless; hts.usitc.gov/reststop/search). A single `query` serves BOTH modes: KEYWORD (e.g. 'laptop') OR HTS number (e.g. '8471.30'). Returns { query, lines:[{ htsno, statisticalSuffix, indent, description, units, columnOneGeneral, specialPreferential, columnTwo, additionalDuties, footnotes, quotaQuantity, effectivePeriod, status, isChapter99 }] } + honest _meta. ★DUTY-RATE HONESTY: columnOneGeneral, specialPreferential, and columnTwo are AUTHORITATIVE VERBATIM TEXT — 'Free', '35%', '0.47¢/kg', compound/range, or null — NEVER coerced to a number (0/NaN fabricates a false 'duty-free'); empty Special ('') → null = NO special rate (NEVER read as Free). ★HIERARCHY: rate stated at a shallower level and inherits downward; read to the nearest ancestor with a non-empty rate; blank deepest ≠ 'no duty'. ★ADDITIONAL DUTIES: `additionalDuties` is frequently null even when Section 301/232 duties apply — the real additional duty rides Chapter-99 rows (isChapter99:true) and STACKS on the base rate. ★COMPLETENESS: endpoint returns the FULL match array; offset IGNORED → totalAvailable is the EXACT served array length; paging CLIENT-SIDE. query must be ≥3 non-whitespace chars (1-2 → invalid_input). limit (≤200, def 50), offset. No-match → honest empty; 404/5xx/timeout/non-array/HTML → schema_drift THROW; transient 400 → upstream_unavailable. NOT a binding CBP ruling or landed-cost quote — duty owed depends on country of origin + trade program + Ch-99 stacking; confirm via CBP (CROSS/eRulings).",
5924
5918
  inputSchema: HtsLookupInput,
5925
5919
  handler: (input) => usitc.htsLookup(input),
5926
5920
  }),
@@ -5935,8 +5929,7 @@ export const TOOLS: ToolDef[] = [
5935
5929
  // rides ONLY in the POST body (v2) — never a URL/header/label/_meta/log.
5936
5930
  defineTool({
5937
5931
  name: "bls_timeseries",
5938
- description:
5939
- "Fetch US Bureau of Labor Statistics time series — the PRICING / ESCALATION layer (keyless; api.bls.gov Public Data API v1, POST/JSON batch). CPI-U & ECI drive federal contract escalation / economic-price-adjustment (EPA) clauses; PPI benchmarks materials pricing; CES employment/wages give labor-rate context (next to gsa_benchmark_labor_rates + sam wage determinations). Inputs (at least one of series/seriesId REQUIRED; both combinable): `series` — a FROZEN 9-key CURATED enum (typo-proof; each carries meaning + units): cpi_u_all/cpi_u_core (CPI-U index, NSA — the escalation reference), ppi_final_demand (PPI index), eci_total_comp/eci_wages (★12-MONTH % CHANGE, NOT an index — a consumer misreads 3.4 as an index level otherwise), unemployment_rate/labor_force_participation (percent, SA), employment_total_nonfarm (thousands of persons, SA), avg_hourly_earnings (dollars/hour, SA). `seriesId` — raw BLS IDs (charclass ^[A-Z0-9]{1,20}$; the OEWS/local-area/regional passthrough; units:null for a raw ID). `startYear`/`endYear` (1900..currentYear+1; default a ~10-year window; span CLAMPED to the tier cap ~10y and disclosed). Returns { series:[{ seriesId, key, meaning, units, observations:[{ year, period, periodName, value, valueUnavailable, footnotes, latest }], observationCount, coveredRange }] } + honest _meta. HONESTY: each `value` is PARSED number|null — the BLS \"-\" unavailable marker (e.g. the 2025 lapse-in-appropriations gap) → null NEVER 0, with valueUnavailable:true + the footnote reason on the observation AND lifted into _meta.notes (a data gap is DISCLOSED, never a silent null and never a fabricated 0); a genuine \"0\" stays 0. A non-SUCCESS status THROWS (never a fake-empty): REQUEST_NOT_PROCESSED (the v1 ~25/day limit) ⇒ rate_limited with the tier disclosure; REQUEST_FAILED ⇒ upstream_unavailable/invalid_input surfacing message[]. A non-JSON 200 or a SUCCESS body missing Results.series ⇒ schema_drift. An empty data[] on SUCCESS ⇒ observations:[] + an ambiguity note (a curated key = a genuine empty range; a raw seriesId = EITHER genuine-empty OR a nonexistent/typo'd ID — verify it). Every response discloses the active tier (v1 keyless ~25/day, 25 series/query, ~10y span | v2 with a free BLS_API_KEY ~500/day) + the per-series units caveat. An OPTIONAL free BLS_API_KEY (env; https://data.bls.gov/registrationEngine/) lifts to v2 and is sent ONLY in the request body — never a URL/header/log.",
5932
+ description: "Fetch one or more BLS time-series over a year range (keyless v1 default; optional free BLS_API_KEY lifts to v2; api.bls.gov POST). At least one of `series` (curated enum key) or `seriesId` (raw ^[A-Z0-9]{1,25}$; covers 25-char OEWS IDs) is required; both may be combined. Optional `startYear`/`endYear` (defaults to active tier's span cap). Returns { series:[{ seriesId, key, meaning, units, observations:[{year, period, periodName, value:number|null, valueUnavailable:bool, footnotes:[{code,text}], latest:bool}], observationCount, coveredRange:{from,to} }] } + honest _meta. HONESTY: BLS '-' unavailable marker → value:null (NEVER 0); valueUnavailable:true on the observation + footnote reason lifted into _meta.notes so the gap is DISCLOSED, never silent. Each series carries its own units label — an ECI '…A' series is a 12-month PERCENT CHANGE, NOT an index level; CPI/PPI are index levels; CES nonfarm employment is thousands of persons. Do NOT compare values across series without reading each units label. status !== 'REQUEST_SUCCEEDED' THROWS: REQUEST_NOT_PROCESSED (v1 daily limit) → rate_limited retryable; REQUEST_FAILED → upstream_unavailable. Series count refused over active tier cap (v1: 25 series/~10yr; v2: 50 series/~20yr) — overflow is NEVER silently dropped. Span is CLAMPED to tier cap before the fetch and disclosed. totalAvailable is null (batch fetch has no upstream total). A typo'd seriesId returns an empty series with 'Invalid Series' upstream message — NOT a real available series. BLS_API_KEY rides ONLY in the POST body, never URL/label/_meta/log.",
5940
5933
  inputSchema: BlsTimeseriesInput,
5941
5934
  handler: (input) => bls.timeseries(input),
5942
5935
  }),
@@ -5944,13 +5937,12 @@ export const TOOLS: ToolDef[] = [
5944
5937
  // The LEVEL layer next to bls_timeseries's ESCALATION layer: mean/median annual &
5945
5938
  // hourly wages + employment by SOC occupation × geography — the highest-value B2G
5946
5939
  // BLS slice (labor-rate benchmarking) that bls_timeseries structurally cannot reach
5947
- // (OEWS IDs are 25 chars > the raw-seriesId 20-char cap). BUILDS the 25-char series
5948
- // ID INTERNALLY from validated structured inputs; REUSES the same POST/JSON transport
5940
+ // (OEWS IDs are 25 chars; bls_timeseries raw-seriesId cap is now widened to 25). BUILDS
5941
+ // the 25-char series ID INTERNALLY from validated structured inputs; REUSES the same POST/JSON transport
5949
5942
  // + parseBlsBody status-throw + mapObservation ("-"→null-never-0) + tier/key seam.
5950
5943
  defineTool({
5951
5944
  name: "bls_oews_wages",
5952
- description:
5953
- "Benchmark US occupational wages & employment from BLS OEWS (Occupational Employment & Wage Statistics) — the LEVEL layer for labor-rate benchmarking (keyless; api.bls.gov Public Data API, POST/JSON batch). The actual mean/median annual & hourly wage a labor category commands, by area — next to gsa_benchmark_labor_rates (GSA CALC), sam wage determinations, and bls_timeseries (the CPI/ECI escalation layer). OEWS series IDs are 25 chars (area×occupation×industry×datatype), EXCEEDING bls_timeseries's raw-seriesId cap, so this tool BUILDS the ID INTERNALLY from validated structured inputs. Inputs (at least one of occupation/soc REQUIRED; all arrays batch into ONE POST — the cartesian product area×occupation×datatype is capped at the active tier's series cap and refused over-cap WITH THE COUNT NAMED, never silently truncated): `occupation` — a CURATED 16-key SOC enum (typo-proof; e.g. software_developer=15-1252, civil_engineer=17-2051, management_analyst=13-1111); `soc` — raw 6-digit HYPHENLESS SOC codes for the ~830-SOC long tail (use 151252, not 15-1252); `area` — default [\"national\"]; each is \"national\", a 2-letter USPS state code (CA/TX/DC…), or a 5-digit CBSA metro code (19100 = Dallas-Fort Worth); `datatype` — default [\"annual_mean\"]: annual_mean/annual_median (dollars/year), hourly_mean/hourly_median (dollars/hour), employment (count jobs). NO year input — OEWS is ANNUAL and the API serves only the latest release; the tool requests a recent window internally and DISCLOSES the reference year. Returns { results:[{ area:{type,code,label}, occupation:{soc,key,label}, measure:{key,code,units}, value:number|null, valueUnavailable, referenceYear, referencePeriod, footnotes, seriesId }] } + honest _meta. HONESTY: (H1) OEWS is an ANNUAL point-in-time snapshot (reference May <year>, period A01), NOT monthly/current-quarter — disclosed every call; (H2) a built ID that returns empty/absent ⇒ value:null, valueUnavailable:FALSE (the occupation is not surveyed/estimated there OR the cell is suppressed for confidentiality) + the not-published note + the surfaced upstream \"Series does not exist\" message + the ID in fieldsUnavailable — NEVER a fabricated 0; a PRESENT \"-\" in-band value ⇒ null + valueUnavailable:true + footnote; (H3) each row's measure.units labels the datatype (never read an employment count as a wage); (H4) the API returns real numerics (no top-code); a non-SUCCESS status THROWS (REQUEST_NOT_PROCESSED ⇒ rate_limited with the tier disclosure; a non-JSON 200 ⇒ schema_drift). Every response discloses the active tier. An OPTIONAL free BLS_API_KEY lifts to v2 and is sent ONLY in the request body — never a URL/header/log.",
5945
+ description: "BLS OEWS occupational wage/employment benchmarking by area × occupation × datatype (keyless; api.bls.gov). Builds validated 25-char series IDs from structured inputs and batches them into one POST — NO year input (OEWS serves only the latest annual release). Inputs: `occupation` (curated enum, e.g. 'software_developer') or `soc` (raw 6-digit SOC, ^[0-9]{6}$ NO hyphen — at least ONE required); `area` (default 'national', 2-letter USPS state, or 5-digit CBSA metro code); `datatype` (default 'annual_mean'; annual_mean/annual_median/hourly_mean/hourly_median/employment). Returns { results:[{ area:{type,code,label}, occupation:{soc,key,label}, measure:{key,code,units}, value:number|null, valueUnavailable:bool, referenceYear, referencePeriod, footnotes, seriesId }] } + honest _meta. ★H1: OEWS is an ANNUAL point-in-time snapshot (reference May <year>); BLS API serves ONLY the most recent release, may lag ~1 year. NOT monthly/current-quarter. ★H2: a built-ID with no published value (occupation not surveyed or suppressed in that area) → value:null (NOT a tool error). ★H3: measure.units is set from datatype — annual_mean/annual_median=dollars/year; hourly_mean/hourly_median=dollars/hour; employment=count. NEVER mislabeled. ★H4: the API returns real numerics (no '#' top-code). area×occupation×datatype is capped at the tier's series cap (v1 25 / v2 50) and refused over-cap with the count named — never silently truncated; REQUEST_NOT_PROCESSED → rate_limited THROWS. Active tier (v1 keyless ~25/day or v2 BLS_API_KEY ~500/day) and series-cap limits disclosed.",
5954
5946
  inputSchema: BlsOewsWagesInput,
5955
5947
  handler: (input) => bls.oewsWages(input),
5956
5948
  }),
@@ -5967,8 +5959,7 @@ export const TOOLS: ToolDef[] = [
5967
5959
  // → honest empty. NEW gate key "bls_qcew"; NO BLS_API_KEY on this keyless path.
5968
5960
  defineTool({
5969
5961
  name: "bls_qcew",
5970
- description:
5971
- "BLS QCEW (Quarterly Census of Employment & Wages) — county×NAICS MARKET-SIZE / wages / location-quotient (keyless; data.bls.gov/cew Open Data Access CSV, a SECOND un-rate-limited BLS domain — NOT the ~25/day api.bls.gov timeseries API). Answers the market-size / competition-density question no other tool can: for ONE area_fips (county/state/metro/US) OR ONE NAICS × quarter — establishment COUNT (market size / competitor density), county×NAICS employment, average weekly wage (labor cost), and the LOCATION QUOTIENT (lq_* = concentration vs the national average; >1.00 = more concentrated / higher competition density). Inputs: `mode` (REQUIRED {area,industry}); `area` (area_fips ^[0-9A-Za-z]{1,6}$ — REQUIRED path segment for mode=area, else an optional client-side narrow); `industry` (NAICS ^[0-9]{1,6}$ DIGIT-ONLY — REQUIRED path segment for mode=industry, else an optional narrow; a hyphenated 31-33 404s, use the digit aggregate); `year` (REQUIRED 1990..current), `quarter` (REQUIRED 1|2|3|4); client-side `ownership`(own_code)/`aggregationLevel`(agglvl_code)/`sizeCode`; `limit` (≤1000, def 50)/`offset`. Wire: GET data.bls.gov/cew/data/api/{year}/{quarter}/{mode}/{code}.csv. Returns { found, mode, area|industry, year, quarter, rows:[{ area_fips, own_code, industry_code, agglvl_code, size_code, base:{ disclosed, disclosureCode, qtrly_estabs, month1/2/3_emplvl, total_qtrly_wages, taxable_qtrly_wages, qtrly_contributions, avg_wkly_wage }, locationQuotient:{ disclosed, disclosureCode, lq_qtrly_estabs, lq_… }, overTheYear:{ disclosed, disclosureCode, oty_qtrly_estabs_chg, oty_…_pct_chg } }] } + honest _meta. ★DISCLOSURE-SUPPRESSION HONESTY (the crux): each row carries THREE disclosure codes (base/lq/oty), each governing its block. QCEW encodes a SUPPRESSED (confidential) employment/wage value as a literal 0 — so under 'N' the confidential emplvl/wage/avg-wkly fields map to null (WITHHELD, never a fabricated $0), while the establishment COUNT (qtrly_estabs / lq_qtrly_estabs) AND its over-the-year change (oty_qtrly_estabs_chg / _pct_chg) stay DISCLOSED (real); under '-' the WHOLE block incl. the estabs field(s) → null; under blank a genuine reported/NEGATIVE 0 SURVIVES (the disclosed federal taxable=0/contrib=0 and the oty_*_chg=0 'no change'). NEVER a blanket 0→null. A null carries disclosed:false + the raw disclosureCode; a suppression note fires whenever any page row is suppressed. HONESTY: totalAvailable is the EXACT filtered row count (fetch-once + client-side limit/offset — QCEW does not paginate; never the page length); a per-tuple HTTP 404 ⇒ honest empty (found:false, the HTML 404 body NEVER parsed as CSV); a 5xx/timeout ⇒ THROW; a 200 non-CSV / a renamed/±column header / a wrong field-count row ⇒ schema_drift THROW (symmetric drift guard). The file MIXES aggregation levels + ownerships — a do-NOT-sum-across-agglvl/ownership note rides every response. PUBLIC AGGREGATE stats (the suppression mechanism keeps small-cell data non-identifying — no PII). Keyless, un-rate-limited; NO BLS_API_KEY is read.",
5962
+ description: "BLS QCEW (Quarterly Census of Employment & Wages) — county×NAICS market-size / wages / location-quotient (keyless; data.bls.gov/cew Open Data Access CSV, un-rate-limited). Inputs: `mode` (REQUIRED {area,industry}); `area` (area_fips ^[0-9A-Za-z]{1,6}$ — REQUIRED for mode=area); `industry` (NAICS ^[0-9]{1,6}$ DIGIT-ONLY — REQUIRED for mode=industry; hyphenated 31-33 404s, use digit aggregate); `year` (REQUIRED), `quarter` (REQUIRED 1|2|3|4); client-side `ownership`/`aggregationLevel`/`sizeCode`; `limit`/`offset`. Returns { found, mode, rows:[{ area_fips, own_code, industry_code, agglvl_code, size_code, base:{disclosed, disclosureCode, qtrly_estabs, month1/2/3_emplvl, total_qtrly_wages, taxable_qtrly_wages, qtrly_contributions, avg_wkly_wage}, locationQuotient:{disclosed, disclosureCode, lq_…}, overTheYear:{disclosed, disclosureCode, oty_…} }] }. ★DISCLOSURE HONESTY: each row has three disclosure codes (base/lq/oty). QCEW encodes SUPPRESSED values as literal 0 — under 'N': confidential emplvl/wage/avg-wkly → null (WITHHELD), estab count + oty-estab change stay DISCLOSED; under '-': WHOLE block → null; under blank: genuine reported/NEGATIVE 0 SURVIVES. NEVER blanket 0→null; null carries disclosed:false + raw disclosureCode; suppression note fires on any suppressed row. HONESTY: totalAvailable is EXACT filtered row count (fetch-once; QCEW does not paginate); per-tuple HTTP 404 → honest empty; 5xx/timeout THROW; 200 non-CSV/renamed header/wrong field-count → schema_drift THROW. Do-NOT-sum-across-agglvl/ownership note rides every response.",
5972
5963
  inputSchema: BlsQcewInput,
5973
5964
  handler: (input) => bls.qcew(input),
5974
5965
  }),
@@ -6030,8 +6021,7 @@ export const TOOLS: ToolDef[] = [
6030
6021
  // (silent no-op). The 15,000-record retrieval window is disclosed, not hidden.
6031
6022
  defineTool({
6032
6023
  name: "nih_reporter_search_projects",
6033
- description:
6034
- "Search awarded NIH RePORTER research-GRANT projects (keyless; api.reporter.nih.gov v2, POST/JSON — the FIRST non-GET getJson-port consumer) — the NEW federal research-funding recipient-enrichment axis (who receives NIH research money, by organization / state, joinable to SAM/USAspending via primary_uei). Structured, LIVE-CONFIRMED-narrowing criteria ONLY, AND-combined in a module-built body (NO raw passthrough): orgStates (UPPERCASE 2-letter USPS enum — the SSRF + silent-zero guard; a lowercase/unknown code silently returns zeros), orgNames (≤512 each, ≤20), fiscalYears (int array 1985..currentYear+1, ≤20), limit (1..500, def 50), offset (0..14,999, def 0). Returns { projects:[{ projectNum, projectTitle, fiscalYear, awardAmount, organization:{ name, state, primaryUei, primaryDuns, ueis, duns }, principalInvestigators, contactPiName, fundingIc }] } + honest _meta. HONESTY: (M2) records are RESEARCH GRANTS, NOT procurement contracts — primary_uei joins to SAM/USAspending recipients but the award nature differs (disclosed in every _meta.notes); totalAvailable = the EXACT meta.total (NEVER the page size, NEVER a lower bound); NIH caps keyless retrieval at the first 15,000 records (offset 0..14,999) — offset ≥ 15,000 ⇒ invalid_input, and past the window the count stays exact while records are UNREACHABLE (disclosed in a note; nextOffset is never a dead-end). Disclose-not-refuse: an unscoped query still returns the first page + the exact total + a narrow-your-criteria note. agencyIcCodes is intentionally NOT a filter (NIH silently drops it — it would be a false 'applied'). Genuine-empty (total:0) ⇒ complete:true/total:0; an outage/5xx/timeout THROWS; a 400 (bad offset/limit/type) ⇒ invalid_input; a 200 body that isn't {meta,results} or a non-numeric meta.total ⇒ schema_drift (never a fake empty). awardAmount is number|null (a real $0 award is 0, an absent amount is null).",
6024
+ description: "Search awarded NIH RePORTER research-grant projects (keyless; api.reporter.nih.gov v2, POST/JSON), joinable to SAM/USAspending via primary_uei. LIVE-CONFIRMED-narrowing criteria ONLY, AND-combined: orgStates (UPPERCASE 2-letter USPS — lowercase/unknown code silently returns zeros), orgNames (≤512 chars each, ≤20 names), fiscalYears (int array, 1985..currentYear+1, ≤20), limit (1..500), offset (0..14,999). Returns { projects:[{ projectNum, projectTitle, fiscalYear, awardAmount, organization:{name, state, primaryUei, primaryDuns, ueis, duns}, principalInvestigators, contactPiName, fundingIc }] } + honest _meta. HONESTY: records are RESEARCH GRANTS, NOT procurement contracts — primaryUei joins SAM/USAspending but the award nature differs (disclosed in every _meta.notes). totalAvailable = EXACT meta.total (NEVER the page size, NEVER a lower bound). NIH caps keyless retrieval at the first 15,000 records (offset 0..14,999); offset ≥ 15,000 → invalid_input; past the window the count stays exact while records are UNREACHABLE (disclosed in a note). An unscoped query returns the first page + exact total + narrow-your-criteria note. agencyIcCodes is NOT a filter (NIH silently drops it — would be a false 'applied'). awardAmount is number|null (genuine $0 is 0, absent is null). Genuine total:0 → complete:true/total:0; outage/5xx THROWS; 400 (bad offset/limit/type) → invalid_input; 200 not {meta,results} or non-numeric meta.total → schema_drift.",
6035
6025
  inputSchema: NihSearchProjectsInput,
6036
6026
  handler: (input) => nih.searchProjects(input),
6037
6027
  }),
@@ -6047,8 +6037,7 @@ export const TOOLS: ToolDef[] = [
6047
6037
  // HTTP 200 loud-fails (never a fake empty); grant≠contract in every response.
6048
6038
  defineTool({
6049
6039
  name: "nsf_search_awards",
6050
- description:
6051
- "Search awarded NSF research-GRANT awards (keyless; api.nsf.gov/services/v1/awards.json) — the NEW federal research-funding recipient-enrichment axis (who receives NSF research money, by organization / UEI / PI / state, joinable to SAM/USAspending via ueiNumber/parentUeiNumber). The grant-SIBLING of nih_reporter_search_projects on a different agency. LIVE-CONFIRMED-narrowing filters ONLY, module-built into a URLSearchParams query (NO raw passthrough): keyword (free text; MULTI-WORD is OR-tokenized — 'machine learning' = machine OR learning, disclosed in _meta.notes), awardeeStateCode (UPPERCASE 2-letter USPS enum — the SSRF + silent-zero guard; a non-state typo silently returns 0), awardeeName, ueiNumber (12-char UEI — an EXACT SAM/USAspending join), parentUeiNumber (parent-org roll-up), pdPIName, dateStart/dateEnd (STRICT mm/dd/yyyy on the award ACTION date — a wrong format is silently mis-parsed), limit (1..100, def 25 → rpp), offset (0..9999). Returns { awards:[{ id, title, agency, cfdaNumber, transType, awardee:{ name, city, stateCode, ueiNumber, parentUeiNumber }, performanceSite, principalInvestigator:{ fullName, firstName, lastName, middleInitial, email, id }, coPrincipalInvestigators, programOfficer, amounts:{ fundsObligatedAmt, estimatedTotalAmt, fundsObligatedByYear }, dates, program, activeAward, historicalAward }] } (abstract EXCLUDED — use nsf_get_award) + honest _meta. HONESTY: NSF Awards are RESEARCH GRANTS, NOT procurement contracts (ueiNumber joins to SAM/USAspending but the award nature differs — disclosed every response); totalAvailable = the EXACT metadata.totalCount below 10,000 and SATURATES at 10,000 (an ES track_total_hits cap ⇒ totalIsLowerBound:true + a note — the true total is ≥10,000 and only the first 10,000 are retrievable); NSF caps keyless retrieval at offset+rpp ≤ 10,000 (offset ≥ 10,000 ⇒ invalid_input; the outgoing rpp is clamped so a page never crosses the window). fundsObligatedAmt/estimatedTotalAmt arrive as STRINGS → number|null (a real $0 is 0, absent is null). Genuine-empty (totalCount:0) ⇒ complete:true/total:0; a serviceNotification at HTTP 200 (bad param / deep offset) ⇒ invalid_input/upstream_unavailable THROWS (never a fake empty); an outage/5xx/timeout THROWS; a 200 body that isn't {response:{award,metadata}} or a non-numeric totalCount ⇒ schema_drift. Feed a row's id to nsf_get_award for the full record + abstractText.",
6040
+ description: "Search awarded NSF research-grant awards (keyless; api.nsf.gov/services/v1/awards.json), joinable to SAM/USAspending via ueiNumber/parentUeiNumber. Filters: keyword (MULTI-WORD is OR-tokenized — 'machine learning' = machine OR learning, disclosed), awardeeStateCode (UPPERCASE 2-letter USPS — non-state typo silently returns 0), awardeeName, ueiNumber (12-char UEI — EXACT SAM/USAspending join), parentUeiNumber, pdPIName, dateStart/dateEnd (strict mm/dd/yyyy — wrong format silently mis-parsed), limit (1..100), offset (0..9999). Returns { awards:[{ id, title, agency, cfdaNumber, transType, awardee:{name, city, stateCode, ueiNumber, parentUeiNumber}, principalInvestigator, coPrincipalInvestigators, programOfficer, amounts:{fundsObligatedAmt, estimatedTotalAmt, fundsObligatedByYear}, dates, program, activeAward, historicalAward }] } (abstract EXCLUDED — use nsf_get_award) + honest _meta. HONESTY: NSF Awards are RESEARCH GRANTS, NOT procurement contracts (ueiNumber joins SAM/USAspending but the award nature differs — disclosed every response). totalAvailable = EXACT metadata.totalCount below 10,000; SATURATES at 10,000 (ES track_total_hits cap → totalIsLowerBound:true + note; first 10,000 only retrievable). NSF caps retrieval at offset+rpp ≤ 10,000 (offset ≥ 10,000 → invalid_input). fundsObligatedAmt/estimatedTotalAmt: STRINGS → number|null (genuine $0 is 0, absent is null). Genuine totalCount:0 → complete:true/total:0; serviceNotification at HTTP 200 → THROWS; outage/5xx THROWS; 200 not {response:{award,metadata}} or non-numeric totalCount → schema_drift.",
6052
6041
  inputSchema: NsfSearchAwardsInput,
6053
6042
  handler: (input) => nsf.searchAwards(input),
6054
6043
  }),
@@ -6073,8 +6062,7 @@ export const TOOLS: ToolDef[] = [
6073
6062
  // AND-tokenized (disclosed); trial≠federal-award caveat in every response.
6074
6063
  defineTool({
6075
6064
  name: "clinicaltrials_search_studies",
6076
- description:
6077
- "Search federally-registered clinical-research studies with LEAD-SPONSOR / COLLABORATOR / ORGANIZATION / FUNDING-SOURCE entity enrichment (keyless; clinicaltrials.gov/api/v2/studies) — the trial-REGISTRATION axis of the research-funding entity layer (the sponsor/collaborator NAMES overlap the pharma/biotech/university/agency entities in NIH RePORTER / NSF Awards / SAM / USAspending). LIVE-CONFIRMED-narrowing filters ONLY, module-built into a URLSearchParams query (NO raw passthrough): query.term (broad free-text), sponsor (→query.spons — a fuzzy sponsor NAME search), condition (→query.cond), location (→query.locn), overallStatus (a frozen 14-value enum → filter.overallStatus), funderType (a frozen 4-value enum nih/fed/industry/other → aggFilters — the FEDERAL-funding axis), pageSize (1..1000, def 20), pageToken (the OPAQUE cursor). Returns { studies:[{ nctId, briefTitle, orgStudyId, organization:{ name, class }, leadSponsor:{ name, class }, collaborators:[{ name, class }], fundingClass, overallStatus, startDate, studyType, phases, conditions }] } (briefSummary EXCLUDED — use clinicaltrials_get_study) + honest _meta. HONESTY: countTotal=true is ALWAYS sent ⇒ totalAvailable = the EXACT filter-respecting UNCAPPED total (NEVER studies.length; a missing/non-number totalCount ⇒ schema_drift; a genuine 0 ⇒ 0, never null); pagination is an OPAQUE cursor (offset/nextOffset null; nextCursor = nextPageToken passed back verbatim as pageToken; terminal = token absent; a bad token ⇒ HTTP 400 THROWS). funderType is re-validated IN the handler — an UNLISTED value silently returns totalCount:0 at HTTP 200 (a fake-empty trap) ⇒ invalid_input pre-fetch (0 fetch); funderType is an OVERLAPPING facet (counts MUST NOT be summed). A MULTI-WORD query.term/sponsor/condition is AND-conjunctive (ALL tokens must co-occur — disclosed). A registered trial is NOT a federal award and leadSponsor.name is FREE TEXT (not a UEI) ⇒ a NOMINAL name match only (disclosed every response). Genuine-empty (totalCount:0, no token) ⇒ complete:true/total:0; a bad overallStatus/pageToken/nctId ⇒ HTTP 400/404 THROWS; an outage/5xx ⇒ THROWS (never a fake empty). Feed a row's nctId to clinicaltrials_get_study for the full record + briefSummary.",
6065
+ description: "Search federally-registered clinical-research studies with LEAD-SPONSOR / COLLABORATOR / FUNDING-SOURCE enrichment (keyless; clinicaltrials.gov/api/v2/studies). Filters: query.term (broad free-text), sponsor (→query.spons, fuzzy NAME search), condition (→query.cond), location (→query.locn), overallStatus (frozen 14-value enum), funderType (frozen 4-value enum nih/fed/industry/other), pageSize (1..1000), pageToken (OPAQUE cursor). Returns { studies:[{ nctId, briefTitle, orgStudyId, organization:{name,class}, leadSponsor:{name,class}, collaborators:[{name,class}], fundingClass, overallStatus, startDate, studyType, phases, conditions }] } (briefSummary EXCLUDED — use clinicaltrials_get_study) + honest _meta. HONESTY: countTotal=true ALWAYS sent → totalAvailable = EXACT uncapped total (NEVER studies.length; missing/non-number totalCount → schema_drift). Pagination is OPAQUE cursor (nextCursor = nextPageToken back as pageToken; terminal = token absent; bad token → HTTP 400 THROWS). funderType re-validated in handler — UNLISTED value silently returns totalCount:0 at HTTP 200 (fake-empty trap) → invalid_input pre-fetch. funderType is an OVERLAPPING facet (counts MUST NOT be summed). MULTI-WORD query.term/sponsor/condition is AND-conjunctive (all tokens must co-occur — disclosed). Registered trial is NOT a federal award; leadSponsor.name is FREE TEXT (not a UEI) → NOMINAL name match only — disclosed every response. Genuine totalCount:0 → complete:true/total:0; bad overallStatus/pageToken → HTTP 400/404 THROWS; outage/5xx THROWS. Feed nctId to clinicaltrials_get_study.",
6078
6066
  inputSchema: ClinicaltrialsSearchStudiesInput,
6079
6067
  handler: (input) => clinicaltrials.searchStudies(input),
6080
6068
  }),
@@ -6096,8 +6084,7 @@ export const TOOLS: ToolDef[] = [
6096
6084
  // every response. NO free-text ⇒ no tokenization.
6097
6085
  defineTool({
6098
6086
  name: "clinicaltrials_facet_counts",
6099
- description:
6100
- "Aggregate/statistical view: EXACT per-value STUDY counts over the WHOLE ClinicalTrials.gov registry for one or more whitelisted ENUM fields (keyless; clinicaltrials.gov/api/v2/stats/field/values) — the DISTRIBUTION sibling of clinicaltrials_search_studies (which gives the exact FILTERED total for a query). Input `fields`: 1..11 ENUM fields (deduped) — OverallStatus, StudyType, Phase, LeadSponsorClass (★ the funding-SOURCE-class distribution: NIH/FED/OTHER_GOV/INDUSTRY/OTHER/NETWORK/INDIV/UNKNOWN/AMBIG — richer than, and distinct from, the search tool's 4-value funderType filter), Sex, DesignAllocation, DesignPrimaryPurpose, DesignInterventionModel, DesignMasking, DesignObservationalModel, DesignTimePerspective. Module-built comma-joined into fields=<…> (NO raw passthrough). Returns { facets:[{ field, fieldPath, valueType, uniqueValuesCount, missingStudiesCount, returned, truncated, overlapping, values:[{ value, studiesCount }] }] } + honest _meta. HONESTY: each studiesCount/uniqueValuesCount is EXACT (typeof-checked to a NUMBER before num() — a non-number ⇒ schema_drift, NEVER a silent 0); a non-ENUM shape for a whitelisted field (e.g. a BOOLEAN {trueCount,falseCount}) ⇒ schema_drift (never read as empty). [M1] _meta.totalAvailable/returned count DISTINCT FIELD VALUES across the requested facet(s), NOT studies (a mandatory unit note points to facets[].values[].studiesCount / clinicaltrials_search_studies for a study count). These counts cover the ENTIRE registry and are NOT filterable — /stats/field/values rejects query.*/filter.*/countTotal/pageSize (HTTP 400) — a scope note cross-links the search tool for filtered totals. The returned<uniqueValuesCount⇒truncated invariant discloses the endpoint's hard 250-value cap the instant it binds (never for these v1 ENUM fields — all complete). Phase is ARRAY-valued (a study can carry several) ⇒ overlapping:true + a not-a-partition note (counts MUST NOT be summed); scalar fields partition the registry minus missingStudiesCount. A high missingStudiesCount ⇒ a note that the shown buckets cover a MINORITY of the registry. MANDATORY CAVEAT every response: a facet count is a distribution over trial REGISTRATIONS, NOT federal awards; LeadSponsorClass is the funding-SOURCE class, not a UEI-keyed award join. An unlisted field ⇒ invalid_input pre-fetch (0 fetch); a 404/400/5xx ⇒ THROWS (never a fake-empty distribution).",
6087
+ description: "Aggregate EXACT per-value study counts over the WHOLE ClinicalTrials.gov registry for 1..11 whitelisted ENUM fields (keyless; clinicaltrials.gov/api/v2/stats/field/values). Input `fields` (deduped): OverallStatus, StudyType, Phase, LeadSponsorClass (NIH/FED/OTHER_GOV/INDUSTRY/OTHER/NETWORK/INDIV/UNKNOWN/AMBIG — richer than the 4-value funderType in the search tool), Sex, DesignAllocation, DesignPrimaryPurpose, DesignInterventionModel, DesignMasking, DesignObservationalModel, DesignTimePerspective. Returns { facets:[{ field, fieldPath, valueType, uniqueValuesCount, missingStudiesCount, returned, truncated, overlapping, values:[{value, studiesCount}] }] } + honest _meta. HONESTY: each studiesCount/uniqueValuesCount is EXACT (typeof NUMBER — non-number → schema_drift); non-ENUM shape for a whitelisted field → schema_drift. _meta.totalAvailable/returned count DISTINCT FIELD VALUES, NOT studies — see facets[].values[].studiesCount / clinicaltrials_search_studies for study counts. Counts cover the ENTIRE registry and are NOT filterable — /stats/field/values rejects query.*/filter.*/pageSize (HTTP 400). returned<uniqueValuesCount → truncated (hard cap 250). Phase is ARRAY-valued (overlapping:true, MUST NOT sum counts); scalar fields partition the registry minus missingStudiesCount. High missingStudiesCount → buckets cover a MINORITY of the registry. MANDATORY CAVEAT: facet counts are distributions over trial REGISTRATIONS, NOT federal awards; LeadSponsorClass is the funding-SOURCE class, not a UEI-keyed award join. Unlisted field → invalid_input pre-fetch; 404/400/5xx → THROWS.",
6101
6088
  inputSchema: ClinicaltrialsFacetCountsInput,
6102
6089
  handler: (input) => clinicaltrials.facetCounts(input),
6103
6090
  }),
@@ -6197,8 +6184,7 @@ export const TOOLS: ToolDef[] = [
6197
6184
  // is a planned, separately-guarded addition).
6198
6185
  defineTool({
6199
6186
  name: "arcgis_hub_discover_datasets",
6200
- description:
6201
- "Discover ArcGIS Hub datasets by keyword — the SLED/GIS open-data layer that Socrata and CKAN do NOT cover (keyless; hub.arcgis.com/api/v3/datasets). Much US state/local/regional/tribal open data (GIS, infrastructure, permits, zoning, boundaries, procurement) is published on ArcGIS Hub. Input `query` (→q, REQUIRED, ≥2 non-whitespace chars — a broad whole-Hub scan is refused), `openDataOnly` (default TRUE → filter[openData]=true, the B2G-relevant designated-open-data subset; false broadens to all shared items), `limit` (1..100, def 20 → page[size]), `offset` (0-based → page[start]=offset+1). Returns { query, openDataOnly, datasets:[{ id, name, description, owner, orgName, source, region, type, sector, keywords, downloadable, hasApi, created, modified, landingPage, itemId }] } + honest _meta. ★PROVENANCE (the crux — a DIFFERENT trust posture from our other sources): ArcGIS Hub is a GLOBAL, OPEN publishing platform — results include NON-US and NON-GOVERNMENTAL publishers. This is a DISCOVERY aid, NOT a curated official-source allowlist (unlike socrata_query): the per-row owner/orgName/source/region are surfaced VERBATIM so you can VET the publisher before relying on the data, and the global-platform caveat rides EVERY response. DISCOVERY ONLY — metadata + links; to read rows follow the dataset on its own ArcGIS endpoint (a guarded row-query tool is a planned addition). HONESTY: totalAvailable = the EXACT Hub match count (meta.total, NEVER data.length — P1); pagination is a 0-based offset (nextOffset when more remain); every scalar null-never-empty, booleans null-preserving, counts null-never-0; a genuine no-match ⇒ complete:true/returned:0; a 429 ⇒ rate_limited / 5xx/timeout ⇒ upstream_unavailable THROWS (never a fake empty); a 200 non-JSON / non-array data ⇒ schema_drift.",
6187
+ description: "Discover ArcGIS Hub datasets by keyword — the SLED/GIS open-data layer that Socrata and CKAN do NOT cover (keyless; hub.arcgis.com/api/v3/datasets). Input `query` (→q, REQUIRED, ≥2 non-whitespace chars — broad whole-Hub scan refused), `openDataOnly` (default TRUE → filter[openData]=true, the B2G-relevant designated-open-data subset; false broadens to all shared items), `limit` (1..100, def 20 → page[size]), `offset` (0-based → page[start]=offset+1). Returns { query, openDataOnly, datasets:[{ id, name, description, owner, orgName, source, region, type, sector, keywords, downloadable, hasApi, created, modified, landingPage, itemId }] } + honest _meta. ★PROVENANCE (the crux): ArcGIS Hub is a GLOBAL, OPEN publishing platform — results include NON-US and NON-GOVERNMENTAL publishers. This is a DISCOVERY aid, NOT a curated official-source allowlist (unlike socrata_query): the per-row owner/orgName/source/region are surfaced VERBATIM so you can VET the publisher, and the global-platform caveat rides EVERY response. DISCOVERY ONLY — metadata + links; to read rows, arcgis_feature_query covers only its curated allowlist; other datasets must be followed on their own endpoint. HONESTY: totalAvailable = EXACT Hub match count (meta.total, NEVER data.length); pagination is 0-based offset; scalars null-never-empty, booleans null-preserving; genuine no-match → complete:true/returned:0; 429 → rate_limited / 5xx/timeout → upstream_unavailable THROWS; 200 non-JSON / non-array → schema_drift.",
6202
6188
  inputSchema: ArcgisHubDiscoverInput,
6203
6189
  handler: (input) => arcgisHub.discoverDatasets(input),
6204
6190
  }),
@@ -6314,8 +6300,7 @@ export const TOOLS: ToolDef[] = [
6314
6300
  // vintage enum is the (benchmark,vintage) UNION ([M2]); GEOIDs stay strings.
6315
6301
  defineTool({
6316
6302
  name: "census_geocode_address",
6317
- description:
6318
- "Resolve a one-line US address → its matched address(es) + the Census GEOGRAPHIES that drive set-aside / place-of-performance analysis (US Census Geocoder, keyless; geocoding.geo.census.gov/geocoder/geographies/onelineaddress) — the NEW territory/geospatial domain. Input `address` (≤500 chars), optional `benchmark` (default Public_AR_Current) / `vintage` (default Current_Current). Returns { matches:[{ matchedAddress, coordinates:{x,y}, tigerLineId, addressComponents, geographies:{ state, county, congressionalDistrict, censusTract, censusBlock, place, cbsaOrCsa, stateLegislativeUpper, stateLegislativeLower } }], matchCount, vintageResolved } + honest _meta. Each geography = { layerKey (the RAW vintage-versioned key, e.g. '119th Congressional Districts'), geoid (a STRING — leading zeros survive: '0102'), name }. HONESTY: genuine-empty (addressMatches:[]) ⇒ matchCount:0/complete:true (NOT an error; verify spelling + add city/state/ZIP); MULTIPLE matches are ALL surfaced (each with its own geographies) + a note; a historical vintage can return >1 layer per type (e.g. 111th+113th Congressional Districts with DISTINCT GEOIDs for a redistricted place) ⇒ BOTH surfaced (chosen + alternates[]) + a mandatory note (NEVER silently dropped); the resolved benchmark/vintage is echoed + a 'Current is a MOVING vintage' note; an invalid/missing benchmark/vintage ⇒ HTTP 400 THROWS (never a fake empty); an outage/5xx ⇒ THROWS. MANDATORY CAVEAT every response: these are a NOMINAL input, NOT an authoritative HUBZone / Opportunity-Zone / set-aside determination (those require SBA's HUBZone map / Treasury's OZ-tract list). Feed censusTract.geoid / county.geoid onward to those authoritative sources.",
6303
+ description: "Resolve a one-line US address → matched address(es) + the Census GEOGRAPHIES for set-aside / place-of-performance analysis (US Census Geocoder, keyless; geocoding.geo.census.gov/geocoder/geographies/onelineaddress). Input: `address` (≤500 chars), optional `benchmark` (default Public_AR_Current), `vintage` (default Current_Current). Returns { matches:[{ matchedAddress, coordinates:{x,y}, tigerLineId, addressComponents, geographies:{ state, county, congressionalDistrict, censusTract, censusBlock, place, cbsaOrCsa, stateLegislativeUpper, stateLegislativeLower } }], matchCount, vintageResolved }. Each geography = { layerKey (raw vintage-versioned key), geoid (STRING — leading zeros survive, e.g. '0102'), name }. HONESTY: genuine empty (addressMatches:[]) → matchCount:0/complete:true (NOT an error; verify spelling + add city/state/ZIP). MULTIPLE matches are ALL surfaced (each with its own geographies) + a note. A historical vintage can return >1 layer per type with DISTINCT GEOIDs → BOTH surfaced (chosen + alternates[]) + a mandatory note (NEVER silently dropped). The resolved benchmark/vintage is echoed + a 'Current is a MOVING vintage' note. Invalid/missing benchmark/vintage → HTTP 400 THROWS (never fake-empty); outage/5xx THROWS. MANDATORY CAVEAT every response: these are a NOMINAL input, NOT an authoritative HUBZone / Opportunity-Zone / set-aside determination — feed censusTract.geoid / county.geoid to SBA's HUBZone map / Treasury's OZ-tract list.",
6319
6304
  inputSchema: CensusGeocodeAddressInput,
6320
6305
  handler: (input) => census.geocodeAddress(input),
6321
6306
  }),
@@ -6335,8 +6320,7 @@ export const TOOLS: ToolDef[] = [
6335
6320
  // a negative number / never 0). The 2D-array body is parsed by header name.
6336
6321
  defineTool({
6337
6322
  name: "census_business_patterns",
6338
- description:
6339
- "Market sizing by NAICS × geography — establishments, employment, and annual payroll from the US Census County Business Patterns (CBP) API (api.census.gov/data/{year}/cbp). ★REQUIRES a free CENSUS_API_KEY: the Census Data API has NO keyless tier, so without the key this tool THROWS an honest config error (get one at https://api.census.gov/data/key_signup.html; call api_key_status to see every source's key requirement). Input: optional `naics` (2–6 digit NAICS-2017, e.g. '5415'; omit to aggregate all sectors), `geography` (us|state|county, default us; county REQUIRES `state`), `state` (2-digit FIPS, e.g. '06'), `year` (default '2023'), optional `limit` (client-side top-N; CBP has no server pagination). Returns { rows:[{ name, geoId, naicsCode, naicsLabel, establishments, employees, annualPayrollUsd, state }] } + honest _meta. HONESTY: establishments/employees are integer counts and annualPayrollUsd is annual US dollars (×1000 from the source's $1,000-unit PAYANN); large-negative suppression sentinels map to null — NEVER a negative number and NEVER 0 (a genuine 0 stays 0; note CBP primarily uses noise-infusion + suppression flags, surfaced as reported — see the tool's suppression note); geoId/naicsCode/state are STRINGS (leading zeros survive). CBP returns the COMPLETE geography set for the filter (no pagination) ⇒ totalAvailable = the row count, complete:true. A missing/invalid key ⇒ invalid_input (a 302 to the Missing-Key page); a header-only body ⇒ honest empty (returned:0); a 5xx ⇒ THROWS; a 200 non-JSON ⇒ schema_drift. The key rides ONLY in the &key= query param — never logged or echoed.",
6323
+ description: "Market sizing by NAICS × geography — establishments, employment, and annual payroll from the US Census County Business Patterns (CBP) API (api.census.gov/data/{year}/cbp). ★REQUIRES a free CENSUS_API_KEY: the Census Data API has NO keyless tier, so without the key this tool THROWS an honest config error (get one at https://api.census.gov/data/key_signup.html; call api_key_status to check). Input: optional `naics` (2–6 digit NAICS-2017, e.g. '5415'; omit to aggregate all sectors), `geography` (us|state|county, default us; county REQUIRES `state`), `state` (2-digit FIPS, e.g. '06'), `year` (default '2023'), optional `limit` (client-side top-N). Returns { rows:[{ name, geoId, naicsCode, naicsLabel, establishments, employees, annualPayrollUsd, state }] } + honest _meta. HONESTY: establishments/employees are integer counts and annualPayrollUsd is annual US dollars (×1000 from source's $1,000-unit PAYANN); large-negative suppression sentinels map to null — NEVER a negative number and NEVER 0 (genuine 0 stays 0; CBP primarily uses noise-infusion + suppression flags, surfaced as reported); geoId/naicsCode/state are STRINGS (leading zeros survive). CBP returns the COMPLETE geography set (no pagination) → totalAvailable = row count, complete:true. Missing/invalid key → invalid_input (302 to Missing-Key page); header-only body → honest empty; 5xx → THROWS; 200 non-JSON → schema_drift. Key rides ONLY in the &key= query param.",
6340
6324
  inputSchema: CensusBusinessPatternsInput,
6341
6325
  handler: (input) => censusEconomic.businessPatterns(input),
6342
6326
  }),
@@ -6361,8 +6345,7 @@ export const TOOLS: ToolDef[] = [
6361
6345
  // SPECIFIC ANNUAL VINTAGE (surfaced in a _meta note; update yearly).
6362
6346
  defineTool({
6363
6347
  name: "cms_medicare_provider_services",
6364
- description:
6365
- "Look up Medicare Part-B provider utilization — for a given provider (NPI) or state, the HCPCS services rendered, beneficiaries served, and submitted / Medicare-allowed / Medicare-paid amounts (CMS 'Medicare Physician & Other Practitioners — by Provider and Service', keyless; data.cms.gov data-API). The demand-side complement to nppes_lookup_provider (who providers ARE → what they BILL) for healthcare-market / competitor / teaming due-diligence. Input: `npi` (10-digit) OR `state` (2-letter) — at least ONE is REQUIRED (the table is 9.78M rows; an all-empty query is refused; providerType/hcpcsCode alone are NOT enough to scope); optional `providerType` (exact CMS specialty, e.g. 'Family Practice'), `hcpcsCode` (e.g. '97110', 'G0463'), `size` (1–100, default 25), `offset`. Returns { services:[{ npi, providerName, credentials, providerType, city, state, zip, hcpcsCode, hcpcsDescription, totalBeneficiaries, totalServices, avgSubmittedCharge, avgMedicareAllowed, avgMedicarePayment }] } + honest _meta. ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (…/data-viewer/stats → found_rows, e.g. VA=278254), NEVER the returned-rows length; if that count fails, totalAvailable is null + a disclosing note (never length-faked). offset/size pagination (hasMore = offset+returned < total). Aggregate/payment values are numeric-string → number|null (a genuine 0 stays 0, absent → null, never 0-faked); NPI/HCPCS/names are null-never-empty-string. A genuine no-match ⇒ honest empty (returned:0); a 4xx ⇒ invalid_input/not_found; a 5xx ⇒ THROWS; a 200 non-array/non-JSON ⇒ schema_drift. These are public PROVIDER-level AGGREGATE figures (no patient identifiers) for ONE annual vintage (the dataset year is disclosed in _meta) — a utilization snapshot, NOT a fraud/quality/fitness determination. KEYLESS — no key is sent.",
6348
+ description: "Medicare Part-B provider utilization — HCPCS services rendered, beneficiaries served, and submitted / Medicare-allowed / Medicare-paid amounts (CMS 'Medicare Physician & Other Practitioners — by Provider and Service', keyless; data.cms.gov data-API). Input: `npi` (10-digit) OR `state` (2-letter) — at least ONE is REQUIRED (the table is 9.78M rows; an all-empty query is refused; providerType/hcpcsCode alone are NOT enough to scope). Optional `providerType` (exact CMS specialty, e.g. 'Family Practice'), `hcpcsCode` (e.g. '97110'), `size` (1–100, def 25), `offset`. Returns { services:[{ npi, providerName, credentials, providerType, city, state, zip, hcpcsCode, hcpcsDescription, totalBeneficiaries, totalServices, avgSubmittedCharge, avgMedicareAllowed, avgMedicarePayment }] } + honest _meta. ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (…/data-viewer/stats → found_rows); if that count fails, totalAvailable is null + a disclosing note (never length-faked). hasMore = offset+returned < total. Aggregate/payment values: numeric-string → number|null (genuine 0 stays 0, absent → null, never 0-faked); NPI/HCPCS/names are null-never-empty-string. Genuine no-match → honest empty; 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array/non-JSON → schema_drift. PUBLIC PROVIDER-LEVEL AGGREGATE figures (no patient identifiers) for ONE annual vintage (dataset year disclosed in _meta) — utilization snapshot, NOT a fraud/quality/fitness determination.",
6366
6349
  inputSchema: CmsMedicareProviderServicesInput,
6367
6350
  handler: (input) => cmsUtilization.providerServices(input),
6368
6351
  }),
@@ -6375,8 +6358,7 @@ export const TOOLS: ToolDef[] = [
6375
6358
  // AND-combined server-side. REQUIRE state OR facilityName (never scanned unscoped).
6376
6359
  defineTool({
6377
6360
  name: "cms_hospital_compare",
6378
- description:
6379
- "Look up Medicare-certified hospitals by US state and/or facility-name fragment — location, type, ownership, emergency-services flag, and CMS star rating (CMS Hospital Compare 'Hospital General Information', keyless; data.cms.gov provider-data datastore-query API, ~5,432 hospitals). A healthcare-facility directory / market-map lane (WHERE hospitals are and HOW CMS rates them). Input: `state` (2-letter, EXACT) OR `facilityName` (a name fragment, case-insensitive substring/contains match) — at least ONE is REQUIRED (an all-empty query is refused; hospitalType alone is NOT enough to scope); optional `hospitalType` (substring, e.g. 'Acute', 'Critical Access'), `size` (1–100, default 25), `offset`. Returns { hospitals:[{ facilityId, facilityName, address, city, state, zip, county, phone, hospitalType, ownership, emergencyServices, overallRating }] } + honest _meta. ★HONESTY: totalAvailable is the response's EXACT top-level `count` for the filter set (VA=96), NEVER the returned-rows length; offset/size pagination (hasMore = offset+returned < count). overallRating is CMS's 1–5 star rating as a number; 'Not Available'/blank/non-numeric ⇒ null (NEVER 0). emergencyServices normalizes 'Yes'⇒true / 'No'⇒false / else null (never a fabricated false). IDs/names/addresses are null-never-empty-string. A genuine no-match ⇒ honest empty (returned:0); a 4xx ⇒ invalid_input/not_found; a 5xx ⇒ THROWS; a 200 non-array body or one missing count/results ⇒ schema_drift. Filters are applied SERVER-SIDE (AND-combined) — nothing is silently dropped. This is a summary star rating, NOT a clinical-quality or fitness determination. KEYLESS — no key is sent.",
6361
+ description: "Look up Medicare-certified hospitals by state and/or facility-name fragment — location, type, ownership, emergency-services flag, and CMS star rating (CMS Hospital Compare 'Hospital General Information', keyless; data.cms.gov provider-data datastore-query API, ~5,432 hospitals). Input: `state` (2-letter, EXACT) OR `facilityName` (case-insensitive substring) — at least ONE is REQUIRED (all-empty query refused; hospitalType alone is NOT enough to scope); optional `hospitalType` (substring, e.g. 'Acute', 'Critical Access'), `size` (1–100, default 25), `offset`. Returns { hospitals:[{ facilityId, facilityName, address, city, state, zip, county, phone, hospitalType, ownership, emergencyServices, overallRating }] } + honest _meta. ★HONESTY: totalAvailable is the response's EXACT top-level `count` for the filter set, NEVER the returned-rows length. overallRating is CMS's 1–5 star rating; 'Not Available'/blank/non-numeric → null (NEVER 0). emergencyServices normalizes 'Yes'→true / 'No'→false / else null. IDs/names/addresses are null-never-empty-string. Genuine no-match → honest empty; 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array or missing count/results → schema_drift. Filters applied SERVER-SIDE (AND-combined). Summary star rating, NOT a clinical-quality or fitness determination.",
6380
6362
  inputSchema: CmsHospitalCompareInput,
6381
6363
  handler: (input) => cmsHospital.hospitalCompare(input),
6382
6364
  }),
@@ -6389,8 +6371,7 @@ export const TOOLS: ToolDef[] = [
6389
6371
  // per dataset → coalesced (null if none — never empty-string, never fabricated).
6390
6372
  defineTool({
6391
6373
  name: "cms_facility_directory",
6392
- description:
6393
- "Look up Medicare/Medicaid-certified healthcare FACILITIES by type — nursing homes, home health agencies, hospices, or dialysis facilities — with their name, address, city, state, zip, and ownership (CMS provider-data, keyless; data.cms.gov datastore-query API, four datasets). A healthcare-facility directory / market-map lane that generalizes cms_hospital_compare beyond hospitals. Input: `facilityType` (REQUIRED enum — 'nursing_home' ~14,695 / 'home_health' ~12,460 / 'hospice' ~6,852 / 'dialysis' ~7,490; selects the dataset id via a constant map, the value never enters the URL path), optional `state` (2-letter, EXACT), `facilityName` (a name fragment, case-insensitive substring/contains match against the dataset's primary-name column), `size` (1–100, default 25), `offset`. Returns { facilities:[{ name, address, city, state, zip, facilityType, ownership }] } + honest _meta. ★HONESTY: totalAvailable is the response's EXACT top-level `count` for the filter set, NEVER the returned-rows length; offset/size pagination (hasMore = offset+returned < count). name/address/ownership column names DIFFER across the four datasets, so each is COALESCED over per-dataset candidates (name: provider_name/facility_name/legal_business_name; address: address/provider_address/address_line_1; ownership: ownership_type/type_of_ownership/profit_or_nonprofit) — a field absent in the chosen dataset is null (unknown), NEVER an empty string and NEVER fabricated. facilityType is echoed on each row. A genuine no-match ⇒ honest empty (returned:0); an invalid facilityType ⇒ invalid_input (blocked by the enum); a 4xx ⇒ invalid_input/not_found; a 5xx ⇒ THROWS; a 200 non-array body or one missing count/results ⇒ schema_drift. Filters are applied SERVER-SIDE (AND-combined) — nothing is silently dropped. This is a facility directory, NOT a clinical-quality or fitness determination. KEYLESS — no key is sent.",
6374
+ description: "Medicare/Medicaid-certified healthcare facilities by type — nursing homes, home health agencies, hospices, or dialysis facilities — with name, address, city, state, zip, and ownership (CMS provider-data, keyless; data.cms.gov datastore-query API, four datasets). Input: `facilityType` (REQUIRED enum — 'nursing_home' ~14,695 / 'home_health' ~12,460 / 'hospice' ~6,852 / 'dialysis' ~7,490; selects the dataset id via a constant map, the value NEVER enters the URL path). Optional `state` (2-letter, EXACT), `facilityName` (case-insensitive substring), `size` (1–100, def 25), `offset`. Returns { facilities:[{ name, address, city, state, zip, facilityType, ownership }] } + honest _meta. ★HONESTY: totalAvailable is the response's EXACT top-level `count` for the filter set — NEVER the returned-rows length; hasMore = offset+returned < count. name/address/ownership column names DIFFER across the four datasets → each is COALESCED over per-dataset candidates (name: provider_name/facility_name/legal_business_name; address: address/provider_address/address_line_1; ownership: ownership_type/type_of_ownership/profit_or_nonprofit) — a field absent in the chosen dataset is null (NEVER an empty string, NEVER fabricated). facilityType is echoed on each row. Filters applied SERVER-SIDE (AND-combined) — nothing silently dropped. Genuine no-match → honest empty; invalid facilityType → invalid_input (enum-blocked); 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array or missing count/results → schema_drift. NOT a clinical-quality or fitness determination.",
6394
6375
  inputSchema: CmsFacilityDirectoryInput,
6395
6376
  handler: (input) => cmsFacility.facilityDirectory(input),
6396
6377
  }),
@@ -6402,8 +6383,7 @@ export const TOOLS: ToolDef[] = [
6402
6383
  // scanned unscoped). The dataset UUID is a SPECIFIC ANNUAL VINTAGE (update yearly).
6403
6384
  defineTool({
6404
6385
  name: "cms_dmepos_suppliers",
6405
- description:
6406
- "Look up Medicare DMEPOS (Durable Medical Equipment, Devices & Supplies) SUPPLIERS — for a given supplier (NPI) or state, the supplier's identity plus aggregate Medicare figures: HCPCS codes billed, beneficiaries served, claims, services, and submitted / Medicare-allowed / Medicare-paid amounts (CMS 'Medicare DMEPOS — by Supplier', keyless; data.cms.gov data-API). The supply-side complement to cms_medicare_provider_services for healthcare-market / competitor / teaming due-diligence on equipment suppliers. Input: `npi` (10-digit) OR `state` (2-letter) — at least ONE is REQUIRED (an all-empty query is refused; the supplier table is never scanned unscoped); optional `size` (1–100, default 25), `offset`. Returns { suppliers:[{ npi, supplierName, credentials, entityType, city, state, zip, totalHcpcsCodes, totalBeneficiaries, totalClaims, totalServices, submittedCharges, medicareAllowed, medicarePayment }] } + honest _meta. ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (…/data-viewer/stats → found_rows), NEVER the returned-rows length; if that count fails, totalAvailable is null + a disclosing note (never length-faked). offset/size pagination (hasMore = offset+returned < total). Aggregate/payment values are numeric-string → number|null (a genuine 0 stays 0, absent → null, never 0-faked); NPI/entityType/names are null-never-empty-string; supplierName joins Last_Name_Org + First_Name ('Last, First' for individuals, the org name alone for organizations). A genuine no-match ⇒ honest empty (returned:0); a 4xx ⇒ invalid_input/not_found; a 5xx ⇒ THROWS; a 200 non-array/non-JSON ⇒ schema_drift. These are public SUPPLIER-level AGGREGATE figures (no patient identifiers) for ONE annual vintage (disclosed in _meta) — a utilization snapshot, NOT a fraud/quality/fitness determination. KEYLESS — no key is sent.",
6386
+ description: "Look up Medicare DMEPOS (Durable Medical Equipment) SUPPLIERS — supplier identity plus aggregate Medicare figures: HCPCS codes billed, beneficiaries served, claims, services, submitted / Medicare-allowed / Medicare-paid amounts (CMS 'Medicare DMEPOS — by Supplier', keyless; data.cms.gov data-API). The supply-side complement to cms_medicare_provider_services. Input: `npi` (10-digit) OR `state` (2-letter) — at least ONE is REQUIRED (all-empty query refused); optional `size` (1–100, default 25), `offset`. Returns { suppliers:[{ npi, supplierName, credentials, entityType, city, state, zip, totalHcpcsCodes, totalBeneficiaries, totalClaims, totalServices, submittedCharges, medicareAllowed, medicarePayment }] } + honest _meta. ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (…/data-viewer/stats → found_rows), NEVER the returned-rows length; if that count fails, totalAvailable is null + a disclosing note (never length-faked). Aggregate/payment values are numeric-string → number|null (genuine 0 stays 0, absent → null); NPI/entityType/names are null-never-empty-string; supplierName coalesces Last_Name_Org + First_Name. Genuine no-match → honest empty; 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array/non-JSON → schema_drift. PUBLIC SUPPLIER-LEVEL AGGREGATE figures (no patient identifiers) for ONE annual vintage — NOT a fraud/quality/fitness determination.",
6407
6387
  inputSchema: CmsDmeposSuppliersInput,
6408
6388
  handler: (input) => cmsSupplier.dmeposSuppliers(input),
6409
6389
  }),
@@ -6415,8 +6395,7 @@ export const TOOLS: ToolDef[] = [
6415
6395
  // P1 pattern; filter VALUES ride via URLSearchParams (bracket key + value encoded).
6416
6396
  defineTool({
6417
6397
  name: "cms_revoked_providers",
6418
- description:
6419
- "Search CMS's PUBLIC 'Revoked Medicare Providers & Suppliers' list — the legally-published register of Medicare enrollment revocations, with the revoked provider/supplier's identity, provider type, revocation reason, effective date, and re-enrollment-bar expiration (CMS 'Revoked Providers and Suppliers', keyless; data.cms.gov data-API, ~7,059 rows). A vetting / due-diligence lane in the SAME class as the OFAC / SAM-exclusions lists — for screening a counterparty before teaming or subcontracting. Input (ALL optional — the ~7K-row list is safe to page unfiltered): `npi` (10-digit → NPI), `state` (2-letter → STATE_CD, exact), `lastName` (→ LAST_NAME, exact), `size` (1–100, default 25), `offset`. Returns { revocations:[{ enrollmentId, npi, name, state, providerType, revocationReason, revocationEffectiveDate, reenrollmentBarExpiration }] } + honest _meta (which notes this is CMS's public revocation/exclusion list — a due-diligence signal, NOT a current-eligibility, guilt, or fitness determination). ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (…/data-viewer/stats → found_rows), NEVER the returned-rows length; if that count fails, totalAvailable is null + a disclosing note (never length-faked). offset/size pagination (hasMore = offset+returned < total). name coalesces ORG_NAME (organizations) else FIRST_NAME + LAST_NAME (individuals) — null if none, never a fabricated empty; NPI/reasons/dates are strings (null-never-empty-string). A genuine no-match ⇒ honest empty (returned:0); a 4xx ⇒ invalid_input/not_found; a 5xx ⇒ THROWS; a 200 non-array/non-JSON ⇒ schema_drift. KEYLESS — no key is sent.",
6398
+ description: "Search CMS's PUBLIC 'Revoked Medicare Providers & Suppliers' list — the legally-published register of Medicare enrollment revocations, with the revoked provider's identity, provider type, revocation reason, effective date, and re-enrollment-bar expiration (CMS 'Revoked Providers and Suppliers', keyless; data.cms.gov data-API, ~7,059 rows). A vetting lane in the same class as OFAC / SAM-exclusions lists. Input (ALL optional — the ~7K-row list is safe to page unfiltered): `npi` (10-digit), `state` (2-letter, EXACT), `lastName` (→ LAST_NAME, exact), `size` (1–100, default 25), `offset`. Returns { revocations:[{ enrollmentId, npi, name, state, providerType, revocationReason, revocationEffectiveDate, reenrollmentBarExpiration }] } + honest _meta (which notes this is CMS's public revocation list — a due-diligence signal, NOT a current-eligibility, guilt, or fitness determination). ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (found_rows), NEVER the returned-rows length (null + note if count fails). name coalesces ORG_NAME else FIRST_NAME + LAST_NAME; NPI/reasons/dates null-never-empty. Genuine no-match → honest empty; 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array/non-JSON → schema_drift. KEYLESS.",
6420
6399
  inputSchema: CmsRevokedProvidersInput,
6421
6400
  handler: (input) => cmsSupplier.revokedProviders(input),
6422
6401
  }),
@@ -6448,8 +6427,7 @@ export const TOOLS: ToolDef[] = [
6448
6427
  // empty, never a throw. The optional key rides &api_key= ONLY.
6449
6428
  defineTool({
6450
6429
  name: "openfda_enforcement",
6451
- description:
6452
- "Search openFDA recall/enforcement records — drug/device/food product recalls with the recalling firm, product, reason, FDA classification (Class I/II/III), status, and geography (openFDA /{category}/enforcement.json; api.fda.gov). KEYLESS (an OPTIONAL free OPENFDA_API_KEY only RAISES the rate limit — keyless works at ~1000 requests/day; it NEVER throws for a missing key; get one at https://open.fda.gov/apis/authentication/; call api_key_status to see every source's key requirement). Input: `category` (drug|device|food, default drug), and STRUCTURED filters — `firm` (→recalling_firm), `product` (→product_description), `reason` (→reason_for_recall), `classification` (Class I|II|III), `status` (e.g. Ongoing/Terminated/Completed), `state` (2-letter, e.g. 'CA') — the tool safely assembles + escapes these into the openFDA search= Lucene string (NO raw passthrough — injection-safe), plus `limit` (1..100, default 25) and `skip` (offset ≥0). Returns { recalls:[{ recallingFirm, productDescription, reasonForRecall, classification, status, state, city, recallInitiationDate, recallNumber, voluntaryMandated, distributionPattern }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination via hasMore/nextOffset — never results.length); every scalar (dates included, recall_initiation_date is a YYYYMMDD string) is null-never-empty-string. ★A no-match query returns openFDA HTTP 404 NOT_FOUND ⇒ an HONEST EMPTY (returned:0, totalAvailable:0), NOT an error; a 400 syntax error ⇒ invalid_input surfacing openFDA's message; a 5xx ⇒ THROWS; a 200 non-JSON ⇒ schema_drift. The optional key rides ONLY the &api_key= query param — never logged or echoed.",
6430
+ description: "Search openFDA recall/enforcement records — drug/device/food product recalls with the recalling firm, product, reason, FDA classification (Class I/II/III), status, and geography (api.fda.gov/{category}/enforcement.json). KEYLESS — an OPTIONAL free OPENFDA_API_KEY only raises the rate limit; keyless works at ~1000 requests/day and NEVER throws for a missing key. Input: `category` (drug|device|food, default drug), structured filters — `firm` (→recalling_firm), `product` (→product_description), `reason` (→reason_for_recall), `classification` (Class I|II|III), `status` (e.g. Ongoing/Terminated), `state` (2-letter) — safely assembled + escaped into openFDA search= Lucene string (NO raw passthrough), plus `limit` (1..100, def 25) and `skip`. Returns { recalls:[{ recallingFirm, productDescription, reasonForRecall, classification, status, state, city, recallInitiationDate, recallNumber, voluntaryMandated, distributionPattern }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination via hasMore/nextOffset — never results.length). Every scalar (recall_initiation_date is a YYYYMMDD string) is null-never-empty-string. ★A no-match query returns openFDA HTTP 404 NOT_FOUND → HONEST EMPTY (returned:0, totalAvailable:0), NOT an error; 400 → invalid_input surfacing openFDA's message; 5xx → THROWS; 200 non-JSON → schema_drift. Optional key rides ONLY the &api_key= query param.",
6453
6431
  inputSchema: OpenfdaEnforcementInput,
6454
6432
  handler: (input) => openfda.enforcement(input),
6455
6433
  }),
@@ -6463,15 +6441,13 @@ export const TOOLS: ToolDef[] = [
6463
6441
  // empty, never a throw. The optional key rides &api_key= ONLY.
6464
6442
  defineTool({
6465
6443
  name: "openfda_device_clearances",
6466
- description:
6467
- "Search openFDA 510(k) DEVICE CLEARANCES — the FDA's premarket-notification (510(k)) clearances for medical devices, with the applicant/manufacturer, device name, clearance number (K-number), decision (date + description), clearance type, product code, advisory committee, and geography (openFDA /device/510k.json; api.fda.gov). KEYLESS (an OPTIONAL free OPENFDA_API_KEY only RAISES the rate limit — keyless works at ~1000 requests/day; it NEVER throws for a missing key; get one at https://open.fda.gov/apis/authentication/; call api_key_status to see every source's key requirement). Input: STRUCTURED filters — `applicant` (→applicant), `deviceName` (→device_name), `productCode` (→product_code), `clearanceType` (→clearance_type, e.g. Traditional/Special/Abbreviated), `kNumber` (→k_number, e.g. 'K123456'), `state` (2-letter, e.g. 'CA') — the tool safely assembles + escapes these into the openFDA search= Lucene string (NO raw passthrough — injection-safe), plus `limit` (1..100, default 25) and `skip` (offset ≥0). Returns { clearances:[{ applicant, deviceName, kNumber, decisionDate, decisionDescription, clearanceType, productCode, advisoryCommittee, state }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination via hasMore/nextOffset — never results.length); every scalar (dates included, decision_date is a YYYY-MM-DD string) is null-never-empty-string. ★A no-match query returns openFDA HTTP 404 NOT_FOUND ⇒ an HONEST EMPTY (returned:0, totalAvailable:0), NOT an error; a 400 syntax error ⇒ invalid_input surfacing openFDA's message; a 5xx ⇒ THROWS; a 200 non-JSON ⇒ schema_drift. The optional key rides ONLY the &api_key= query param — never logged or echoed.",
6444
+ description: "Search openFDA 510(k) DEVICE CLEARANCES — FDA premarket-notification clearances for medical devices, with the applicant/manufacturer, device name, clearance number (K-number), decision (date + description), clearance type, product code, advisory committee, and geography (openFDA /device/510k.json; api.fda.gov). KEYLESS (optional free OPENFDA_API_KEY only raises the rate limit — keyless works at ~1000 requests/day; NEVER throws for a missing key). Input: STRUCTURED filters — `applicant`, `deviceName`, `productCode`, `clearanceType` (e.g. Traditional/Special/Abbreviated), `kNumber` (e.g. 'K123456'), `state` (2-letter) — safely escaped into the openFDA search= Lucene string (NO raw passthrough), plus `limit` (1..100, default 25) and `skip` (offset ≥0). Returns { clearances:[{ applicant, deviceName, kNumber, decisionDate, decisionDescription, clearanceType, productCode, advisoryCommittee, state }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination via hasMore/nextOffset — never results.length); every scalar is null-never-empty-string; decision_date is a YYYY-MM-DD string. ★A no-match query returns openFDA HTTP 404 → HONEST EMPTY (returned:0/total:0), NOT an error; 400 → invalid_input surfacing openFDA's message; 5xx → THROWS; 200 non-JSON → schema_drift. Optional key rides ONLY in &api_key= param.",
6468
6445
  inputSchema: OpenfdaDeviceClearancesInput,
6469
6446
  handler: (input) => openfdaDevice.deviceClearances(input),
6470
6447
  }),
6471
6448
  defineTool({
6472
6449
  name: "openfda_drug_approvals",
6473
- description:
6474
- "Search openFDA Drugs@FDA DRUG APPROVALS — FDA-approved drug applications (NDA/ANDA/BLA) with the sponsor, application number, each approved product (brand + generic/active-ingredient name, dosage form, route, marketing status), and the submission/approval history (openFDA /drug/drugsfda.json; api.fda.gov). Answers 'what drugs did sponsor X get approved, and which are still marketed' — pharma vendor product/approval intelligence. KEYLESS (an OPTIONAL free OPENFDA_API_KEY only RAISES the rate limit — keyless works at ~1000 requests/day; NEVER throws for a missing key; api_key_status lists every source's key requirement). Input: STRUCTURED filters — `sponsorName` (→sponsor_name), `brandName` (→products.brand_name), `activeIngredient` (→products.active_ingredients.name), `applicationNumber` (→application_number) — safely escaped into the openFDA search= Lucene string (NO raw passthrough — injection-safe), plus `limit` (1..100, default 25) and `skip` (offset ≥0). Returns { applications:[{ applicationNumber, sponsorName, products:[{ brandName, genericIngredients:[{name,strength}], dosageForm, route, marketingStatus }], submissions:[{ submissionType, submissionNumber, submissionStatus, submissionStatusDate, submissionClass }] }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination — never results.length); every scalar is null-never-empty-string; a 'Discontinued' marketingStatus is NOT an approval revocation (disclosed in _meta). ★A no-match query returns openFDA HTTP 404 NOT_FOUND ⇒ an HONEST EMPTY (returned:0/total:0), NOT an error; a 400 ⇒ invalid_input surfacing openFDA's message; a 5xx ⇒ THROWS; a 200 non-JSON ⇒ schema_drift. The optional key rides ONLY the &api_key= query param — never logged or echoed.",
6450
+ description: "Search openFDA Drugs@FDA DRUG APPROVALS — FDA-approved drug applications (NDA/ANDA/BLA) with the sponsor, application number, approved products (brand + generic/active-ingredient name, dosage form, route, marketing status), and submission/approval history (openFDA /drug/drugsfda.json; api.fda.gov). KEYLESS (optional free OPENFDA_API_KEY only raises the rate limit — ~1000 req/day keyless; NEVER throws for a missing key). Input: STRUCTURED filters — `sponsorName`, `brandName`, `activeIngredient`, `applicationNumber` — safely escaped into the openFDA search= Lucene string (NO raw passthrough), plus `limit` (1..100, default 25) and `skip` (offset ≥0). Returns { applications:[{ applicationNumber, sponsorName, products:[{ brandName, genericIngredients:[{name,strength}], dosageForm, route, marketingStatus }], submissions:[{ submissionType, submissionNumber, submissionStatus, submissionStatusDate, submissionClass }] }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination — never results.length); every scalar is null-never-empty-string; a 'Discontinued' marketingStatus is NOT an approval revocation (disclosed in _meta). ★A no-match query returns openFDA HTTP 404 → HONEST EMPTY (returned:0/total:0), NOT an error; 400 → invalid_input; 5xx → THROWS; 200 non-JSON → schema_drift. Optional key rides ONLY in &api_key= param.",
6475
6451
  inputSchema: OpenfdaDrugApprovalsInput,
6476
6452
  handler: (input) => openfdaDrugsfda.drugApprovals(input),
6477
6453
  }),
@@ -6504,8 +6480,7 @@ export const TOOLS: ToolDef[] = [
6504
6480
  // (disclosed) rather than silently fetch the entire dataset.
6505
6481
  defineTool({
6506
6482
  name: "cpsc_recalls",
6507
- description:
6508
- "Look up U.S. CPSC consumer-product RECALLS — the recall title, hazard description, remedy, affected products, manufacturers, retailers, injuries, and country of manufacture (CPSC SaferProducts /RestWebServices/Recall; www.saferproducts.gov). The consumer-goods / import product-safety lane alongside nhtsa_recalls (vehicles) and openfda (medical). KEYLESS — no API key is required or accepted. Inputs (ALL optional): `dateStart`/`dateEnd` (YYYY-MM-DD recall date range), `productName` (substring), `manufacturer` (substring), `recallNumber` (a specific CPSC recall number). Returns { recalls:[{ recallNumber, recallDate, title, description, url, products:[names], numberOfUnits, manufacturers:[names], retailers:[names], hazards:[descriptions], remedies:[descriptions], injuries:[names], manufacturerCountries:[names] }] } + honest _meta. HONESTY: the CPSC response is a bare array with NO count field and NO pagination — it returns the COMPLETE matching set, so totalAvailable = the number of returned recalls and complete:true (never a fabricated total). ★With NO filter given, results are bounded to a DEFAULT ~90-day recent window (RecallDateStart, disclosed in _meta.notes) rather than a silent whole-dataset fetch. An empty result ⇒ an HONEST EMPTY (returned:0), NOT an error; a 4xx ⇒ invalid_input; a 5xx/timeout ⇒ THROWS; a 200 non-JSON OR a non-array body ⇒ schema_drift. Nested arrays are flattened to name/description strings (an empty {} object is skipped, never fabricated); NumberOfUnits is free text kept as a string; dates are strings; every scalar is null-never-empty-string. Fixed host www.saferproducts.gov (SSRF-guarded); dates are ^\\d{4}-\\d{2}-\\d{2}$ and recallNumber is letters/digits/hyphen only.",
6483
+ description: "Look up U.S. CPSC consumer-product RECALLS — recall title, hazard description, remedy, affected products, manufacturers, retailers, injuries, and country of manufacture (CPSC SaferProducts /RestWebServices/Recall; www.saferproducts.gov). KEYLESS. Siblings: nhtsa_recalls (vehicles), openfda_enforcement. Inputs (ALL optional): `dateStart`/`dateEnd` (YYYY-MM-DD recall date range), `productName` (substring), `manufacturer` (substring), `recallNumber`. Returns { recalls:[{ recallNumber, recallDate, title, description, url, products:[names], numberOfUnits, manufacturers:[names], retailers:[names], hazards:[descriptions], remedies:[descriptions], injuries:[names], manufacturerCountries:[names] }] } + honest _meta. HONESTY: the CPSC response is a bare array with NO count field and NO pagination — it returns the COMPLETE matching set, so totalAvailable = number of returned recalls and complete:true (never a fabricated total). ★With NO filter given, results are bounded to a DEFAULT ~90-day recent window (RecallDateStart, disclosed in _meta.notes) rather than a silent whole-dataset fetch. Empty result → HONEST EMPTY (returned:0), NOT an error; 4xx → invalid_input; 5xx/timeout → THROWS; 200 non-JSON or non-array → schema_drift. Nested arrays are flattened to name/description strings; NumberOfUnits kept as a string; every scalar is null-never-empty-string. Fixed host (SSRF-guarded); dates are ^\\d{4}-\\d{2}-\\d{2}$ and recallNumber is letters/digits/hyphen only.",
6509
6484
  inputSchema: CpscRecallsInput,
6510
6485
  handler: (input) => cpsc.recalls(input),
6511
6486
  }),
@@ -6527,8 +6502,7 @@ export const TOOLS: ToolDef[] = [
6527
6502
  // DataValue is a comma-formatted string; suppression codes ((NA)/(D)/(NM)/(L)/*) → null.
6528
6503
  defineTool({
6529
6504
  name: "bea_regional_data",
6530
- description:
6531
- "Regional (county / state / MSA) economic data — GDP by industry and personal income — from the US Bureau of Economic Analysis (BEA) Regional Economic Accounts (apps.bea.gov/api/data, dataset 'Regional'). ★REQUIRES a free BEA_API_KEY: the BEA Data API has NO keyless tier, so without the key this tool THROWS an honest config error (get one at https://apps.bea.gov/API/signup/; call api_key_status to see every source's key requirement). Input: `tableName` (required, e.g. 'CAGDP2' county GDP by industry, 'SAGDP2N' state GDP, 'CAINC1'/'SAINC1' personal income), `geoFips` (required — 'STATE' for all states, a county FIPS like '06075', or an MSA code), `lineCode` (required — an integer industry line like '1', or 'ALL'), optional `year` ('LAST5' default, a 4-digit year, or 'ALL'), `frequency` ('A' annual default, or 'Q'). Returns { rows:[{ geoFips, geoName, timePeriod, lineCode, dataValue, unitOfMeasure, unitMult, noteRef }], notes:[{ noteRef, noteText }] } + honest _meta. ★HONESTY (the crux): a missing/invalid key — or ANY bad parameter — returns HTTP 200 carrying an Error object (NOT an HTTP error status); this is detected and surfaced as invalid_input carrying BEA's APIErrorDescription — NEVER a fake empty. dataValue is parsed from BEA's comma-formatted string ('1,234,567'→1234567); BEA suppression/not-available codes ((NA)/(D)/(NM)/(L)/*) map to null — NEVER 0 (a genuine 0 stays 0). unitMult (a power-of-10 multiplier) and unitOfMeasure are reported ALONGSIDE the raw dataValue — the value is NOT multiplied in (apply unitMult yourself). BEA returns the COMPLETE set for the filter (no pagination) ⇒ totalAvailable = the row count, complete:true; a genuine empty Data:[] ⇒ honest empty (returned:0); a 5xx ⇒ THROWS; a 200 non-JSON ⇒ schema_drift. The key rides ONLY in the UserID= query param — never logged or echoed.",
6505
+ description: "Regional (county / state / MSA) GDP by industry and personal income from the BEA Regional Economic Accounts (apps.bea.gov/api/data, dataset 'Regional'). ★REQUIRES a free BEA_API_KEY — NO keyless tier; without the key this tool THROWS an honest config error (get one at https://apps.bea.gov/API/signup/; call api_key_status to check). Input: `tableName` (required, e.g. 'CAGDP2' county GDP, 'SAGDP2N' state GDP, 'CAINC1'/'SAINC1' personal income), `geoFips` (required — 'STATE', county FIPS like '06075', or MSA code), `lineCode` (required — integer industry line or 'ALL'), optional `year` ('LAST5' default, 4-digit year, or 'ALL'), `frequency` ('A'/'Q'). Returns { rows:[{ geoFips, geoName, timePeriod, lineCode, dataValue, unitOfMeasure, unitMult, noteRef }], notes:[{ noteRef, noteText }] } + honest _meta. ★HONESTY: a missing/invalid key OR ANY bad parameter returns HTTP 200 carrying an Error object — detected and surfaced as invalid_input carrying BEA's APIErrorDescription, NEVER a fake empty. dataValue parsed from BEA's comma-formatted string ('1,234,567' → 1234567). BEA suppression codes (NA)/(D)/(NM)/(L)/* → null (NEVER 0; genuine 0 stays 0). unitMult and unitOfMeasure reported ALONGSIDE raw dataValue — NOT pre-multiplied in. BEA returns the COMPLETE filter result (no pagination) → complete:true. Genuine empty Data:[] → honest empty; 5xx → THROWS; 200 non-JSON → schema_drift. Key rides ONLY in the UserID= query param.",
6532
6506
  inputSchema: BeaRegionalDataInput,
6533
6507
  handler: (input) => bea.regionalData(input),
6534
6508
  }),
@@ -6541,8 +6515,7 @@ export const TOOLS: ToolDef[] = [
6541
6515
  // never-0; standardRate/isOconus are STRING booleans coerced to real booleans.
6542
6516
  defineTool({
6543
6517
  name: "gsa_perdiem_rates",
6544
- description:
6545
- "Look up GSA Federal Travel PER-DIEM rates — the max lodging + Meals & Incidental Expenses (M&IE) reimbursement ceilings for official U.S. government travel (api.gsa.gov /travel/perdiem/v2, keyed — DATA_GOV_API_KEY or the shared DEMO_KEY). Input: EITHER `city` (e.g. 'Washington') + `state` (2-letter, e.g. 'DC') OR `zip` (5-digit) — supplying BOTH, or NEITHER, ⇒ invalid_input with 0 fetch; optional `year` (default: the current U.S. federal fiscal year). Returns { rates:[{ city, county, state, zip, year, isOconus, standardRate, mealsUsd, monthlyLodgingUsd:[{ month (1-12), monthName, lodgingUsd }] }] } + honest _meta. HONESTY: lodgingUsd (the API's monthly `value`) is the MAX nightly lodging ceiling for that month — it VARIES SEASONALLY (hence a per-month array), and mealsUsd is the daily M&IE ceiling; both are integer US dollars, null-when-withheld (NEVER 0 — a genuine 0 is preserved). standardRate/isOconus are booleans coerced from the API's string 'true'/'false' (an unrecognized value ⇒ null, never a fabricated false); the months array is preserved AS-IS (never padded to 12). The API returns the COMPLETE rate set (no pagination) ⇒ totalAvailable = the row count, complete:true. A genuine no-match (rates:[]/rate:[]) ⇒ honest empty (returned:0); the API's `errors` field non-null ⇒ invalid_input carrying the message (never a fake empty); a 429 (DEMO_KEY ~10 req/hr, hit quickly) ⇒ rate_limited THROWS; a 5xx/timeout ⇒ upstream_unavailable THROWS; a 200 non-JSON ⇒ schema_drift. DEMO_KEY ~10 req/hr shared ceiling — set DATA_GOV_API_KEY (free at api.data.gov/signup) for 1000/hr. The key rides ONLY in the X-Api-Key header (never the URL/_meta).",
6518
+ description: "Look up GSA Federal Travel PER-DIEM rates — the max lodging + Meals & Incidental Expenses (M&IE) reimbursement ceilings for official U.S. government travel (api.gsa.gov /travel/perdiem/v2, keyed — DATA_GOV_API_KEY or the shared DEMO_KEY). Input: EITHER `city` + `state` (2-letter) OR `zip` (5-digit) — supplying BOTH or NEITHER → invalid_input with 0 fetch; optional `year` (default: current federal fiscal year). Returns { rates:[{ city, county, state, zip, year, isOconus, standardRate, mealsUsd, monthlyLodgingUsd:[{ month (1-12), monthName, lodgingUsd }] }] } + honest _meta. HONESTY: lodgingUsd is the MAX nightly lodging ceiling for that month — VARIES SEASONALLY (hence a per-month array); mealsUsd is the daily M&IE ceiling; both are integer US dollars, null-when-withheld (NEVER 0 — genuine 0 preserved). standardRate/isOconus are booleans coerced from the API's string 'true'/'false' (unrecognized → null, never fabricated false); months array preserved AS-IS (never padded to 12). API returns COMPLETE rate set (no pagination) → totalAvailable = row count, complete:true. Genuine no-match → honest empty; `errors` field non-null → invalid_input; 429 (DEMO_KEY ~10 req/hr) → rate_limited THROWS; set DATA_GOV_API_KEY (free, api.data.gov/signup) for 1000/hr. 5xx/timeout → upstream_unavailable THROWS; 200 non-JSON → schema_drift. Key rides ONLY in the X-Api-Key header.",
6546
6519
  inputSchema: GsaPerdiemRatesInput,
6547
6520
  handler: (input) => gsaPerdiem.perdiemRates(input),
6548
6521
  }),
@@ -6562,8 +6535,7 @@ export const TOOLS: ToolDef[] = [
6562
6535
  }),
6563
6536
  defineTool({
6564
6537
  name: "dol_get_dataset",
6565
- description:
6566
- "Fetch records from ONE US DOL dataset (apiprod.dol.gov /v4/get/{agency}/{endpoint}/json). ★REQUIRES a free DOL_API_KEY: the DOL DATA endpoint has NO keyless tier, so without the key this tool THROWS an honest config error (get one at https://dataportal.dol.gov/registration; the dataset CATALOG — dol_list_datasets — and agency list stay keyless). Input: `agency` (required — the `agencyAbbr` from dol_list_datasets, e.g. 'WHD', 'OSHA', 'ILAB'; rides the PATH, ^[A-Za-z0-9_]+$), `table` (required — the dataset's `apiUrl` endpoint from dol_list_datasets, e.g. 'Child_Labor_Report__2016_to_2022'; rides the PATH, ^[A-Za-z0-9_]+$), optional `limit` (default 10, max 100), `offset`, `filterField`+`filterValue` (a paired equality filter → a DOL filter_object), `fields` (best-effort column selection). Returns { records:[…verbatim dataset rows…] } + honest _meta. HONESTY: records are surfaced VERBATIM (the data-record envelope is key-gated and unverified, so field names/values are preserved as-is — a genuine 0 stays 0, a missing field stays null; the tool never coerces or fabricates). totalAvailable is a real count field ONLY when the response carries one, else null (an honest unknown — `returned` is NEVER passed off as the total); offset pagination (a full page ⇒ hasMore, page forward to confirm). A missing/invalid key (401/403) ⇒ invalid_input carrying the DOL_API_KEY guidance (never empty); a 400 ⇒ invalid_input; a genuine empty ⇒ honest empty (returned:0); a 429 ⇒ rate_limited THROWS (Retry-After honored); a 5xx/timeout ⇒ upstream_unavailable THROWS; a 200 non-JSON / no row array ⇒ schema_drift. The key rides ONLY in the X-API-KEY request header — never the URL / _meta / a log.",
6538
+ description: "Fetch records from ONE US DOL dataset (apiprod.dol.gov /v4/get/{agency}/{endpoint}/json). ★REQUIRES a free DOL_API_KEY: the DOL DATA endpoint has NO keyless tier — without the key this tool THROWS an honest config error (get one at https://dataportal.dol.gov/registration; dol_list_datasets stays keyless). Input: `agency` (required — the `agencyAbbr` from dol_list_datasets, e.g. 'WHD'/'OSHA'/'ILAB'; rides the PATH, ^[A-Za-z0-9_]+$), `table` (required — the dataset's `apiUrl` endpoint from dol_list_datasets; rides the PATH, ^[A-Za-z0-9_]+$), optional `limit` (def 10, max 100), `offset`, `filterField`+`filterValue` (paired equality filter), `fields` (best-effort column selection). Returns { records:[…verbatim dataset rows…] } + honest _meta. HONESTY: records are surfaced VERBATIM (field names/values preserved as-is — genuine 0 stays 0, missing field stays null; never coerced or fabricated). totalAvailable is a real count ONLY when the response carries one, else null (honest unknown — `returned` is NEVER passed off as the total). A full page → hasMore; page forward to confirm. Missing/invalid key (401/403) → invalid_input carrying DOL_API_KEY guidance (never empty); 400 → invalid_input; genuine empty → honest empty; 429 → rate_limited THROWS (Retry-After honored); 5xx/timeout → upstream_unavailable THROWS; 200 non-JSON / no row array → schema_drift. Key rides ONLY in the X-API-KEY request header — never URL/_meta.",
6567
6539
  inputSchema: DolGetDatasetInput,
6568
6540
  handler: (input) => dol.getDataset(input),
6569
6541
  }),
@@ -6576,8 +6548,7 @@ export const TOOLS: ToolDef[] = [
6576
6548
  // page-based pagination. income/expenses are null-or-decimal-string ⇒ null-never-0.
6577
6549
  defineTool({
6578
6550
  name: "lda_search_filings",
6579
- description:
6580
- "Search US Senate LDA (Lobbying Disclosure Act) filings — who is paid HOW MUCH to lobby WHICH federal agency on WHICH issue (lda.senate.gov/api/v1/filings, KEYLESS — anonymous access works; an optional free LDA_API_KEY only raises the rate limit). All inputs optional: `registrantName` (the lobbying firm/in-house filer), `clientName` (who it's for), `lobbyistName`, `filingYear` (4-digit), `filingType` (short code, e.g. 'Q1'/'RR'/'YE'), `agency` (NOTE: /filings/ has NO server-side agency filter — the LDA API silently ignores it, so it is reported in _meta.filtersDropped and NOT applied; government entities are nested per activity in lobbyingActivities[].governmentEntities), `issue` (specific lobbying issues text), `page` (1-based, default 1), `pageSize` (1..25, default 25). Returns { filings:[{ filingUuid, filingType, filingYear, filingPeriod, incomeUsd, expensesUsd, registrant, client, lobbyingActivities:[{ issueCode, description, governmentEntities:[names] }], documentUrl, postedDate, terminationDate }] } + honest _meta. HONESTY: totalAvailable is the API's REAL total match count (the corpus is ~1.95M filings) — NOT the rows on this page; pagination is page-based (pass the next page number when hasMore). incomeUsd/expensesUsd are parsed from the null-or-decimal-string income/expenses — null (not reported) ⇒ null, NEVER 0 (a genuine 0 stays 0); a filing reports EITHER income OR expenses, so the other is typically null. Missing lobbying_activities/government_entities ⇒ empty arrays (never fabricated). A genuine no-match (results:[]) ⇒ honest empty (returned:0); a 400 (bad filter) ⇒ invalid_input surfacing the API's message; a 429 ⇒ rate_limited THROWS (Retry-After honored, never routed around); a 5xx/timeout ⇒ upstream_unavailable THROWS; a 200 non-JSON / non-array results / non-number count ⇒ schema_drift. The optional key rides ONLY in the Authorization: Token header (never the URL/_meta).",
6551
+ description: "Search US Senate LDA (Lobbying Disclosure Act) filings — who is paid how much to lobby which federal agency on which issue (lda.senate.gov/api/v1/filings, KEYLESS — anonymous access works; optional free LDA_API_KEY only raises the rate limit). Filters (all optional): `registrantName` (the lobbying firm/in-house filer), `clientName`, `lobbyistName`, `filingYear` (4-digit), `filingType` (e.g. 'Q1'/'RR'/'YE'), `agency` (NOTE: /filings/ has NO server-side agency filter — the LDA API silently ignores it, so it is reported in _meta.filtersDropped and NOT applied; government entities are nested per activity in lobbyingActivities[].governmentEntities), `issue`, `page` (1-based), `pageSize` (1..25). Returns { filings:[{ filingUuid, filingType, filingYear, filingPeriod, incomeUsd, expensesUsd, registrant, client, lobbyingActivities:[{issueCode, description, governmentEntities:[names]}], documentUrl, postedDate, terminationDate }] } + honest _meta. HONESTY: totalAvailable is the API's REAL total match count (corpus ~1.95M filings) — NOT the rows on this page; pagination is page-based. incomeUsd/expensesUsd parsed from null-or-decimal-string — null (not reported) → null, NEVER 0 (genuine 0 stays 0); a filing reports EITHER income OR expenses, so the other is typically null. Missing lobbying_activities/government_entities → empty arrays. Genuine no-match → honest empty; 400 → invalid_input; 429 → rate_limited THROWS (Retry-After honored); 5xx/timeout THROWS; 200 non-JSON/non-array results/non-number count → schema_drift. Token rides ONLY in the Authorization: Token header.",
6581
6552
  inputSchema: LdaSearchFilingsInput,
6582
6553
  handler: (input) => lda.searchFilings(input),
6583
6554
  }),
@@ -6591,8 +6562,7 @@ export const TOOLS: ToolDef[] = [
6591
6562
  // (nextCursor extracted from `next`, host re-asserted). type=o FIXED.
6592
6563
  defineTool({
6593
6564
  name: "courtlistener_search_opinions",
6594
- description:
6595
- "Search US FEDERAL COURT OPINIONS (case law / litigation) via CourtListener (www.courtlistener.com/api/rest/v4/search, type=o). ★PROVENANCE: the DATA is US federal court PUBLIC RECORDS, but the API is CourtListener, run by the Free Law Project (a NON-PROFIT) — this is NOT a .gov API; CourtListener republishes these records KEYLESS because the .gov primary source (PACER) is PAYWALLED. KEYLESS (anonymous access works; an optional free COURTLISTENER_API_TOKEN only raises the rate limit; get one at https://www.courtlistener.com/help/api/rest/; call api_key_status to see every source's key requirement). All inputs optional: `query` (full-text → q), `court` (a court id, ^[a-z0-9]+$ — e.g. 'uscfc' US Court of Federal Claims for contract claims/bid protests, 'cafc' Federal Circuit for contract/patent appeals, 'scotus'), `dateFiledAfter`/`dateFiledBefore` (ISO ^\\d{4}-\\d{2}-\\d{2}$ → filed_after/filed_before), `natureOfSuit` (folded into the q query — no verified dedicated filter, disclosed in notes), `cursor` (opaque continuation — pass back _meta.nextCursor), `order` (→ order_by, default 'dateFiled desc'). Returns { opinions:[{ caseName, court, courtId, dateFiled, docketNumber, natureOfSuit, status, judge, citation, absoluteUrl }] } + honest _meta. HONESTY: totalAvailable is the API's REAL `count` (the total match count for the filter) — NOT the rows on this page; pagination is an OPAQUE CURSOR (offset/nextOffset are null/meaningless — pass _meta.nextCursor back as `cursor`; nextCursor:null/hasMore:false = last page). CourtListener v4 stops counting on deep cursor pages (count:null) ⇒ totalAvailable:null is DISCLOSED, never faked as results.length. dateFiled is a date STRING; citation may be an array/object ⇒ flattened to a safe string/string[] (never fabricated); judge/natureOfSuit/docketNumber are null when absent (never ''); absoluteUrl is the full https://www.courtlistener.com link. A genuine no-match (results:[]) ⇒ honest empty (returned:0); a 400 (bad param) ⇒ invalid_input surfacing the API's message; a 429 (unauth throttle) ⇒ rate_limited THROWS (Retry-After honored, never routed around); a 5xx/timeout ⇒ upstream_unavailable THROWS; a 200 non-JSON / non-array results / a count that is neither a number nor null ⇒ schema_drift; an off-host `next` is REFUSED (SSRF). The optional token rides ONLY in the Authorization: Token header (never the URL/_meta).",
6565
+ description: "Search US federal court opinions via CourtListener (www.courtlistener.com/api/rest/v4/search, type=o). ★PROVENANCE: DATA is US federal court PUBLIC RECORDS; the API is CourtListener (Free Law Project, NON-PROFIT) — NOT a .gov API; the .gov primary source (PACER) is PAYWALLED. KEYLESS (optional free COURTLISTENER_API_TOKEN only raises the rate limit). Filters (all optional): `query` (full-text → q), `court` (^[a-z0-9]+$ — e.g. 'uscfc' US Court of Federal Claims, 'cafc' Federal Circuit, 'scotus'), `dateFiledAfter`/`dateFiledBefore` (ISO YYYY-MM-DD), `natureOfSuit` (folded into q — no verified dedicated filter, disclosed in notes), `cursor` (opaque continuation), `order` (default 'dateFiled desc'). Returns { opinions:[{ caseName, court, courtId, dateFiled, docketNumber, natureOfSuit, status, judge, citation, absoluteUrl }] } + honest _meta. HONESTY: totalAvailable is the API's REAL `count` (total match count) — NOT rows on this page. Pagination is OPAQUE CURSOR (pass _meta.nextCursor back as `cursor`; nextCursor:null/hasMore:false = last page). CourtListener v4 stops counting on deep cursor pages (count:null) → totalAvailable:null DISCLOSED, never faked as results.length. dateFiled is a date STRING; citation → flattened to string/string[]; judge/natureOfSuit/docketNumber null when absent; absoluteUrl is the full CL link. Genuine no-match → honest empty; 400 → invalid_input; 429 → rate_limited (Retry-After honored); 5xx/timeout THROWS; 200 non-JSON/count not number or null → schema_drift; off-host `next` REFUSED (SSRF). Token rides ONLY in the Authorization: Token header.",
6596
6566
  inputSchema: CourtlistenerSearchOpinionsInput,
6597
6567
  handler: (input) => courtlistener.searchOpinions(input),
6598
6568
  }),
@@ -6613,8 +6583,7 @@ export const TOOLS: ToolDef[] = [
6613
6583
  }),
6614
6584
  defineTool({
6615
6585
  name: "nonprofit_financials",
6616
- description:
6617
- "Fetch ONE US tax-exempt nonprofit's IRS Form 990 profile + FINANCIALS by EIN via ProPublica Nonprofit Explorer (projects.propublica.org/nonprofits/api/v2/organizations/{ein}.json). ★PROVENANCE: the DATA is IRS Form 990 filings (federal tax-exempt public records) but the API is ProPublica Nonprofit Explorer, run by ProPublica (a NON-PROFIT newsroom) — NOT a .gov API; ProPublica republishes these records KEYLESS because the IRS has no clean query API. KEYLESS (no key). Input: `ein` (required — the Employer Identification Number, 1..9 digits ^\\d{1,9}$, e.g. '530196605' American National Red Cross; rides the URL path). Returns { organization:{ ein, name, address, city, state, zip, nteeCode, subsectionCode, rulingDate, statusCode }, filings:[{ taxYear, formType, revenueUsd, expensesUsd, assetsUsd, liabilitiesUsd, pdfUrl }] } + honest _meta. HONESTY: the four Form 990 figures (revenueUsd/expensesUsd/assetsUsd/liabilitiesUsd, from totrevenue/totfuncexpns/totassetsend/totliabend) ride null-never-0 coercion — a genuine reported 0 stays 0, an absent figure ⇒ null (NEVER 0-faked); ein/codes are strings; rulingDate is a date string. totalAvailable = filings.length (the COMPLETE Form 990 filing set from the one detail document — no pagination). An unknown EIN (HTTP 404) ⇒ not_found (NEVER a fabricated empty org); a 4xx ⇒ invalid_input; a 429 ⇒ rate_limited THROWS; a 5xx/timeout ⇒ upstream_unavailable THROWS; a 200 non-JSON / non-object organization / non-array filings_with_data ⇒ schema_drift. Data is IRS Form 990 data via ProPublica Nonprofit Explorer, disclosed in _meta.source and a note.",
6586
+ description: "Fetch ONE US tax-exempt nonprofit's IRS Form 990 profile + FINANCIALS by EIN via ProPublica Nonprofit Explorer (projects.propublica.org/nonprofits/api/v2/organizations/). ★PROVENANCE: the DATA is IRS Form 990 filings (federal tax-exempt public records) but the API is ProPublica Nonprofit Explorer (NON-PROFIT newsroom) — NOT a .gov API; ProPublica republishes these records KEYLESS (no key). Input: `ein` (required — the Employer Identification Number, 1..9 digits, e.g. '530196605' for American National Red Cross; rides the URL path). Returns { organization:{ ein, name, address, city, state, zip, nteeCode, subsectionCode, rulingDate, statusCode }, filings:[{ taxYear, formType, revenueUsd, expensesUsd, assetsUsd, liabilitiesUsd, pdfUrl }] } + honest _meta. HONESTY: the four Form 990 figures (revenueUsd/expensesUsd/assetsUsd/liabilitiesUsd) ride null-never-0 coercion — genuine reported 0 stays 0, absent → null (NEVER 0-faked); ein/codes are strings; rulingDate is a date string. totalAvailable = filings.length (COMPLETE filing set — no pagination). Unknown EIN (HTTP 404) → not_found (NEVER fabricated empty org); 4xx → invalid_input; 429 → rate_limited THROWS; 5xx/timeout → upstream_unavailable THROWS; 200 non-JSON / non-object org / non-array filings → schema_drift. Data source disclosed in _meta.source.",
6618
6587
  inputSchema: NonprofitFinancialsInput,
6619
6588
  handler: (input) => nonprofit.financials(input),
6620
6589
  }),
@@ -6668,20 +6637,73 @@ async function main() {
6668
6637
  },
6669
6638
  });
6670
6639
 
6640
+ // ── Toolset resolution (MCP_SAM_GOV_TOOLSETS) ────────────────────
6641
+ const toolsetEnv = process.env.MCP_SAM_GOV_TOOLSETS;
6642
+ const allToolNames = TOOLS.map((t) => t.name);
6643
+ const tsResult = resolveToolsets(toolsetEnv, allToolNames);
6644
+
6645
+ // Warn on unknown names.
6646
+ for (const u of tsResult.unknown) {
6647
+ console.error(
6648
+ `[mcp-sam-gov] WARNING: unknown toolset name '${u}' in MCP_SAM_GOV_TOOLSETS — valid names: ${ALL_TOOLSET_NAMES.join(", ")}. Ignoring.`,
6649
+ );
6650
+ }
6651
+ if (tsResult.fellBack) {
6652
+ console.error(
6653
+ `[mcp-sam-gov] WARNING: no valid toolset names found in MCP_SAM_GOV_TOOLSETS='${toolsetEnv}' — falling back to all tools.`,
6654
+ );
6655
+ }
6656
+
6657
+ const loadedToolNames = tsResult.loaded;
6658
+ const isAllTools = tsResult.sets.length === 1 && tsResult.sets[0] === "all";
6659
+
6660
+ // Build instructions: add a one-liner about loaded/available toolsets when
6661
+ // not using the default (all). The default instructions stay byte-identical.
6662
+ // B2: unknown toolset names are reported here (not only stderr) because
6663
+ // stderr is invisible in Claude Desktop.
6664
+ const BASE_INSTRUCTIONS =
6665
+ "This server wraps US government open data (keyless-first). If a tool result looks wrong, a tool stays broken, or the user wants a capability this server lacks, help improve it: call the `feedback` tool — or use the `report` URL present on schema_drift / upstream_unavailable errors — to get a PREFILLED GitHub issue link, and offer it to the user to open and submit. Nothing is posted automatically; the user submits. Never include secrets or personal data in a report (the repo is public).";
6666
+
6667
+ let serverInstructions: string;
6668
+ if (isAllTools && tsResult.unknown.length === 0) {
6669
+ // Default: byte-identical to main.
6670
+ serverInstructions = BASE_INSTRUCTIONS;
6671
+ } else {
6672
+ const parts: string[] = [];
6673
+ if (!isAllTools) {
6674
+ // List loaded sets.
6675
+ parts.push(`Loaded toolsets: ${tsResult.sets.join(", ")}.`);
6676
+ // List other available sets with a 2–4 word hint each (NB5).
6677
+ const otherSets = ALL_TOOLSET_NAMES
6678
+ .filter((s) => !tsResult.sets.includes(s))
6679
+ .map((s) => `${s} (${TOOLSET_HINTS[s]})`);
6680
+ if (otherSets.length > 0) {
6681
+ parts.push(`Other available sets (set MCP_SAM_GOV_TOOLSETS to enable): ${otherSets.join("; ")}.`);
6682
+ }
6683
+ }
6684
+ // B2: report unknown names in instructions so they are visible in Claude Desktop.
6685
+ if (tsResult.unknown.length > 0) {
6686
+ parts.push(`Unknown toolset name(s) ignored: ${tsResult.unknown.join(", ")} — valid names: ${ALL_TOOLSET_NAMES.join(", ")}.`);
6687
+ }
6688
+ if (tsResult.fellBack) {
6689
+ parts.push("No valid toolset names found; fell back to all tools.");
6690
+ }
6691
+ serverInstructions = `${BASE_INSTRUCTIONS} ${parts.join(" ")}`;
6692
+ }
6693
+
6671
6694
  const server = new Server(
6672
6695
  { name: SERVER_NAME, version: SERVER_VERSION },
6673
6696
  {
6674
6697
  capabilities: { tools: {} },
6675
6698
  // Surfaced to the agent at initialize. Tells it how to route real-usage
6676
6699
  // friction back to the project WITHOUT the server ever posting anything.
6677
- instructions:
6678
- "This server wraps US government open data (keyless-first). If a tool result looks wrong, a tool stays broken, or the user wants a capability this server lacks, help improve it: call the `feedback` tool — or use the `report` URL present on schema_drift / upstream_unavailable errors — to get a PREFILLED GitHub issue link, and offer it to the user to open and submit. Nothing is posted automatically; the user submits. Never include secrets or personal data in a report (the repo is public).",
6700
+ instructions: serverInstructions,
6679
6701
  },
6680
6702
  );
6681
6703
 
6682
6704
  server.setRequestHandler(ListToolsRequestSchema, async () => {
6683
6705
  return {
6684
- tools: TOOLS.map((t) => ({
6706
+ tools: filterToolsFor(TOOLS, loadedToolNames).map((t) => ({
6685
6707
  name: t.name,
6686
6708
  description: t.description,
6687
6709
  inputSchema: zodToJsonSchema(t.inputSchema),
@@ -6692,6 +6714,22 @@ async function main() {
6692
6714
 
6693
6715
  server.setRequestHandler(CallToolRequestSchema, async (req) => {
6694
6716
  const { name, arguments: args } = req.params;
6717
+ // Check if the tool exists but is not loaded in the current toolset profile.
6718
+ const knownEntry = TOOLS.find((t) => t.name === name);
6719
+ if (knownEntry && !loadedToolNames.has(name)) {
6720
+ // B3: suggest the union of current sets + the needed set, so the user's
6721
+ // existing profile is not silently dropped.
6722
+ const envelope = toolNotLoadedEnvelope(name, tsResult.sets);
6723
+ return {
6724
+ content: [
6725
+ {
6726
+ type: "text" as const,
6727
+ text: JSON.stringify(envelope, null, 2),
6728
+ },
6729
+ ],
6730
+ isError: true,
6731
+ };
6732
+ }
6695
6733
  try {
6696
6734
  const raw = await runTool(name, args ?? {}, sam);
6697
6735
  // A handler may return either its raw domain object OR a MetaBundle
@@ -6736,9 +6774,11 @@ async function main() {
6736
6774
 
6737
6775
  const transport = new StdioServerTransport();
6738
6776
  await server.connect(transport);
6739
- console.error(
6740
- `[mcp-sam-gov] v${SERVER_VERSION} listening on stdio (${TOOLS.length} tools).`,
6741
- );
6777
+ const loadedCount = loadedToolNames.size;
6778
+ const profileNote = isAllTools
6779
+ ? `${loadedCount} tools`
6780
+ : `${loadedCount}/${TOOLS.length} tools, toolsets: ${tsResult.sets.join(",")}`;
6781
+ console.error(`[mcp-sam-gov] v${SERVER_VERSION} listening on stdio (${profileNote}).`);
6742
6782
  // Fire-and-forget, opt-out, fail-silent update notice (STDERR only, never stdout).
6743
6783
  // Deliberately NOT awaited: it must never delay or affect the server (update-check.ts).
6744
6784
  void checkForUpdate(SERVER_VERSION);