@cliwant/mcp-sam-gov 1.13.2 → 1.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.ja.md +8 -1
- package/README.ko.md +8 -1
- package/README.md +53 -2
- package/dist/arcgis-feature.d.ts.map +1 -1
- package/dist/arcgis-feature.js +8 -0
- package/dist/arcgis-feature.js.map +1 -1
- package/dist/bls.d.ts.map +1 -1
- package/dist/bls.js +4 -3
- package/dist/bls.js.map +1 -1
- package/dist/bonfire.d.ts +1 -1
- package/dist/bonfire.d.ts.map +1 -1
- package/dist/bonfire.js +15 -11
- package/dist/bonfire.js.map +1 -1
- package/dist/data-map.d.ts +43 -0
- package/dist/data-map.d.ts.map +1 -0
- package/dist/data-map.js +289 -0
- package/dist/data-map.js.map +1 -0
- package/dist/errors.d.ts +6 -1
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js.map +1 -1
- package/dist/server.d.ts.map +1 -1
- package/dist/server.js +159 -58
- package/dist/server.js.map +1 -1
- package/dist/socrata.d.ts +1 -1
- package/dist/socrata.d.ts.map +1 -1
- package/dist/socrata.js +46 -1
- package/dist/socrata.js.map +1 -1
- package/dist/toolsets.d.ts +111 -0
- package/dist/toolsets.d.ts.map +1 -0
- package/dist/toolsets.js +390 -0
- package/dist/toolsets.js.map +1 -0
- package/package.json +1 -1
- package/src/arcgis-feature.ts +8 -0
- package/src/bls.ts +4 -3
- package/src/bonfire.ts +15 -11
- package/src/data-map.ts +329 -0
- package/src/errors.ts +6 -1
- package/src/server.ts +180 -99
- package/src/socrata.ts +47 -1
- package/src/toolsets.ts +435 -0
package/dist/server.js
CHANGED
|
@@ -18,7 +18,7 @@
|
|
|
18
18
|
*/
|
|
19
19
|
import { Server } from "@modelcontextprotocol/sdk/server/index.js";
|
|
20
20
|
import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
|
|
21
|
-
import { CallToolRequestSchema, ListToolsRequestSchema, } from "@modelcontextprotocol/sdk/types.js";
|
|
21
|
+
import { CallToolRequestSchema, ListResourcesRequestSchema, ListToolsRequestSchema, ReadResourceRequestSchema, } from "@modelcontextprotocol/sdk/types.js";
|
|
22
22
|
import { z } from "zod";
|
|
23
23
|
import { SamGovClient, daysUntilResponse, applyResponseDeadlineWindow, } from "./sam-gov/index.js";
|
|
24
24
|
import * as usas from "./usaspending.js";
|
|
@@ -86,13 +86,15 @@ import * as keys from "./keys.js";
|
|
|
86
86
|
import { toToolError, ToolErrorCarrier, errorFromResponse } from "./errors.js";
|
|
87
87
|
import * as feedback from "./feedback.js";
|
|
88
88
|
import { checkForUpdate } from "./update-check.js";
|
|
89
|
+
import { renderDataMapMarkdown } from "./data-map.js";
|
|
89
90
|
import { buildMeta, isMetaBundle, withMeta, } from "./meta.js";
|
|
90
91
|
import { pathToFileURL, fileURLToPath } from "node:url";
|
|
91
92
|
import { realpathSync } from "node:fs";
|
|
93
|
+
import { resolveToolsets, ALL_TOOLSET_NAMES, TOOLSET_HINTS, filterToolsFor, toolNotLoadedEnvelope, } from "./toolsets.js";
|
|
92
94
|
const SERVER_NAME = "mcp-sam-gov";
|
|
93
95
|
// Kept in lockstep with package.json / manifest.json / server.json.
|
|
94
96
|
// Keep in sync with package.json "version" (asserted at release; see CHANGELOG).
|
|
95
|
-
const SERVER_VERSION = "1.
|
|
97
|
+
const SERVER_VERSION = "1.15.0";
|
|
96
98
|
// ─── Tool input schemas (Zod) ────────────────────────────────────
|
|
97
99
|
const SamSearchInput = z.object({
|
|
98
100
|
query: z.string().optional().describe("Free-text title query"),
|
|
@@ -1567,7 +1569,8 @@ const EdgarCompanyConceptInput = z.object({
|
|
|
1567
1569
|
// validated strings (a bad column ⇒ upstream 400 ⇒ invalid_input, surfaced).
|
|
1568
1570
|
const SocrataDomainEnum = z
|
|
1569
1571
|
.enum(socrata.SOCRATA_DOMAINS)
|
|
1570
|
-
.describe("Which allowlisted Socrata portal to query (
|
|
1572
|
+
.describe("Which allowlisted Socrata portal to query (the SSRF host allowlist — no free host). " +
|
|
1573
|
+
"Jurisdiction of non-obvious hosts: cthru.data.socrata.com=MASSACHUSETTS statewide (CTHRU); atlanta.data.socrata.com=Atlanta GA; controllerdata.lacity.org+data.lacity.org=Los Angeles; www.dallasopendata.com=Dallas TX; data.brla.gov=Baton Rouge LA; data.kcmo.org=Kansas City MO; data.cstx.gov=College Station TX; data.weho.org=West Hollywood CA; opendata.usac.org+datahub.usac.org=federal USAC E-rate. data.colorado.gov's procurement data is CITY OF DENVER, not CO state.");
|
|
1571
1574
|
const SocrataQueryInput = z.object({
|
|
1572
1575
|
domain: SocrataDomainEnum,
|
|
1573
1576
|
datasetId: z
|
|
@@ -1620,7 +1623,8 @@ const SocrataDiscoverDatasetsInput = z.object({
|
|
|
1620
1623
|
.string()
|
|
1621
1624
|
.min(1)
|
|
1622
1625
|
.describe("Keyword(s) to find datasets, e.g. 'procurement', 'vendor payments', 'checkbook'."),
|
|
1623
|
-
domain: SocrataDomainEnum.optional().describe("Optional: scope discovery to ONE
|
|
1626
|
+
domain: SocrataDomainEnum.optional().describe("Optional: scope discovery to ONE portal; omit to search all. The catalog does not index every host (USAC returns 0); those stay queryable via socrata_query with a known 4x4. " +
|
|
1627
|
+
"Jurisdiction of non-obvious hosts: cthru.data.socrata.com=MASSACHUSETTS statewide (CTHRU); atlanta.data.socrata.com=Atlanta GA; controllerdata.lacity.org+data.lacity.org=Los Angeles; www.dallasopendata.com=Dallas TX; data.brla.gov=Baton Rouge LA; data.kcmo.org=Kansas City MO; data.cstx.gov=College Station TX; data.weho.org=West Hollywood CA; opendata.usac.org+datahub.usac.org=federal USAC E-rate. data.colorado.gov's procurement data is CITY OF DENVER, not CO state."),
|
|
1624
1628
|
limit: z
|
|
1625
1629
|
.number()
|
|
1626
1630
|
.int()
|
|
@@ -2624,7 +2628,7 @@ const TableauViewCsvInput = z.object({
|
|
|
2624
2628
|
const ArcgisFeatureQueryInput = z.object({
|
|
2625
2629
|
service: z
|
|
2626
2630
|
.enum(arcgisFeature.ARCGIS_SERVICES.map((s) => s.key))
|
|
2627
|
-
.describe("
|
|
2631
|
+
.describe("Service key (SSRF allowlist; 29 services). DC OCP PASS (solicitations/contracts/purchase_orders/payments). US local govs: Asheville NC, Bellevue WA, Miami-Dade FL×2, Suffolk County NY, Mat-Su AK, Las Vegas NV×2, Baltimore MD, Naperville IL, Worcester MA, Topeka KS (FY2015–23), Hennepin County MN (CIP pipeline), Charlotte-Mecklenburg NC (CIP pipeline); TX/AK/IA/OK DOT bid/award registers. ND DOT flex-funding to local agencies (nddot_flex×4 — NOT vendor contracts)."),
|
|
2628
2632
|
where: z
|
|
2629
2633
|
.string()
|
|
2630
2634
|
.min(1)
|
|
@@ -2962,7 +2966,7 @@ const HtsLookupInput = z.object({
|
|
|
2962
2966
|
// SECOND POST-batch getJson-port consumer (after NIH). SSRF surface = a compile-
|
|
2963
2967
|
// time-CONSTANT host+path; seriesids ride in the module-built POST body. `series`
|
|
2964
2968
|
// is a FROZEN 9-key curated enum (the SSRF value guard + the units-label source);
|
|
2965
|
-
// `seriesId` is the raw passthrough, charclass-validated ^[A-Z0-9]{1,
|
|
2969
|
+
// `seriesId` is the raw passthrough, charclass-validated ^[A-Z0-9]{1,25}$. Years
|
|
2966
2970
|
// are bounded ints (1900..currentYear+1); the span is clamped to the tier cap
|
|
2967
2971
|
// (v1 ~10y) BEFORE the fetch + disclosed. An OPTIONAL free BLS_API_KEY rides ONLY
|
|
2968
2972
|
// in the POST body (v2, ~500/day) — never a URL/header/label/_meta/log.
|
|
@@ -2973,10 +2977,10 @@ const BlsTimeseriesInput = z.object({
|
|
|
2973
2977
|
.optional()
|
|
2974
2978
|
.describe("One or more CURATED series enum keys (typo-proof; each carries a meaning + units label): cpi_u_all (CPI-U all items NSA, index), cpi_u_core (CPI-U core NSA, index), ppi_final_demand (PPI final demand NSA, index), eci_total_comp (ECI total comp — ★12-MO % CHANGE, not an index), eci_wages (ECI wages — ★12-MO % CHANGE), unemployment_rate (SA, percent), labor_force_participation (SA, percent), employment_total_nonfarm (SA, thousands of persons), avg_hourly_earnings (SA, dollars/hour). NSA CPI-U is the escalation/EPA-clause reference. At least one of series/seriesId is required; both may be combined."),
|
|
2975
2979
|
seriesId: z
|
|
2976
|
-
.array(z.string().regex(/^[A-Z0-9]{1,
|
|
2980
|
+
.array(z.string().regex(/^[A-Z0-9]{1,25}$/))
|
|
2977
2981
|
.max(bls.BLS_SERIES_KEYS.length + 50)
|
|
2978
2982
|
.optional()
|
|
2979
|
-
.describe("One or more RAW BLS series IDs (power-user passthrough for the un-curatable space — OEWS area×occupation, local-area unemployment LAUCN…, SA/regional CPI variants). Charclass ^[A-Z0-9]{1,
|
|
2983
|
+
.describe("One or more RAW BLS series IDs (power-user passthrough for the un-curatable space — OEWS area×occupation, local-area unemployment LAUCN…, SA/regional CPI variants). Charclass ^[A-Z0-9]{1,25}$ (uppercase alnum; punctuation/whitespace/lowercase rejected — SSRF + 'verify the ID' honesty). OEWS IDs are 25 chars (e.g. OEUN000000000000015125201). A raw ID has units:null (consult BLS). A nonexistent/typo'd ID returns BLS success + empty data (the ambiguity is disclosed, not asserted as 'no data'). At least one of series/seriesId is required."),
|
|
2980
2984
|
startYear: z
|
|
2981
2985
|
.number()
|
|
2982
2986
|
.int()
|
|
@@ -3827,7 +3831,7 @@ const LdaSearchFilingsInput = z.object({
|
|
|
3827
3831
|
.string()
|
|
3828
3832
|
.min(1)
|
|
3829
3833
|
.optional()
|
|
3830
|
-
.describe("NOTE:
|
|
3834
|
+
.describe("NOTE: /filings/ has NO server-side government-entity filter — the LDA API silently ignores this field (reported in _meta.filtersDropped, never as a narrowed total). Government entities are nested per activity in lobbyingActivities[].governmentEntities; narrow by registrantName/clientName/issue and inspect those nested entities. Retained for discoverability."),
|
|
3831
3835
|
issue: z
|
|
3832
3836
|
.string()
|
|
3833
3837
|
.min(1)
|
|
@@ -4779,7 +4783,7 @@ export const TOOLS = [
|
|
|
4779
4783
|
// kev.listed to null (never false), and a kevOnly filter during an outage THROWS.
|
|
4780
4784
|
defineTool({
|
|
4781
4785
|
name: "cve_lookup",
|
|
4782
|
-
description: "Look up NIST NVD CVE records (keyless; services.nvd.nist.gov CVE API 2.0) — exact by `cveId`
|
|
4786
|
+
description: "Look up NIST NVD CVE records (keyless; services.nvd.nist.gov CVE API 2.0) — exact by `cveId` OR search by `keyword`/`cpeName`/`cvssV3Severity`/date range — each row JOINED with its CISA KEV status. Returns { results:[{ cveId, vulnStatus, rejected, published, lastModified, description, cvssMetrics:[{version,source,type,baseScore,baseSeverity,vectorString,exploitabilityScore,impactScore}], primaryCvss:{version,baseScore,baseSeverity,type}|null, cwes, references, kev }] } + honest _meta. Optional `kevOnly`, `resultsPerPage` (≤2000, def 50), `startIndex`. CVSS HONESTY: every ^cvssMetric key (V2/V30/V31/V40) is its own element — versions never conflated; V2 baseSeverity reads from metric level; primaryCvss is highest-version, type:Primary preferred but FALLS BACK to highest Secondary (real CNA score never dropped), null ONLY when no CVSS exists — base scores null-never-0. KEV HONESTY: kev is {listed:true,dateAdded,dueDate,ransomware,requiredAction,catalogVersion} | {listed:false,note} | {listed:null,status:'unavailable'}; not-listed ≠ safe (absence is NOT a clearance); if KEV catalog cannot load, kev.listed degrades to NULL (never false) with fieldsUnavailable:['kev']; a kevOnly filter during KEV outage THROWS. PAGINATION from NVD EXACT totalResults (never page length). Genuine totalResults:0 → honest found:false; 403/429 → rate_limited THROWS with NVD_API_KEY tier disclosure; 404/5xx/timeout/off-host THROW. Optional free NVD_API_KEY (env) lifts the rate — sent ONLY in the apiKey header.",
|
|
4783
4787
|
inputSchema: CveLookupInput,
|
|
4784
4788
|
handler: (input) => nvd.cveLookup(input),
|
|
4785
4789
|
}),
|
|
@@ -4791,14 +4795,14 @@ export const TOOLS = [
|
|
|
4791
4795
|
}),
|
|
4792
4796
|
defineTool({
|
|
4793
4797
|
name: "nist_800_53_controls",
|
|
4794
|
-
description: "Look up NIST SP 800-53 Rev 5 security & privacy CONTROLS (keyless) — the requirement backbone for FedRAMP / CMMC / RMF compliance work.
|
|
4798
|
+
description: "Look up NIST SP 800-53 Rev 5 security & privacy CONTROLS (keyless) — the requirement backbone for FedRAMP / CMMC / RMF compliance work. Complements cve_lookup + cisa_kev_lookup. Retrieve by `controlId` (exact, e.g. 'AC-2', 'SC-7', 'AC-2(1)'), `family` (2-letter 'AC'/'SC'/'IA' or name substring), and/or `keyword` (case-insensitive substring over title + statement); `limit`/`offset` pagination. Each row: { id, family, title, status ('withdrawn'|null), statement (requirement prose; NULL for a WITHDRAWN control, never ''), guidance (discussion), incorporatedInto:[ids that superseded a withdrawn control], enhancements:[{id,title}] }. HONESTY: source is NIST's OFFICIAL OSCAL catalog at github.com/usnistgov/oscal-content (authoritative first-party data served from GitHub, not a .gov API host — provenance disclosed in _meta); the exact OSCAL version + last-modified are surfaced in _meta (catalog fetched live from the MOVING 'main' branch, so control text can shift between point releases — cite the version); a WITHDRAWN control has statement:null and is NOT an active requirement (see incorporatedInto for what replaced it); filtering is CLIENT-SIDE and totalAvailable is the EXACT match count; applicability depends on the system's FIPS-199 impact baseline (Low/Moderate/High), which the catalog does not encode; a download failure or implausibly-truncated catalog (< 15 families) THROWS (never fake-empty).",
|
|
4795
4799
|
inputSchema: NistControlsInput,
|
|
4796
4800
|
handler: (input) => nistControls.searchControls(input),
|
|
4797
4801
|
}),
|
|
4798
4802
|
// ━━━ NPPES NPI Registry — Healthcare-Provider Vetting (1) ━━━ ADR-0036
|
|
4799
4803
|
defineTool({
|
|
4800
4804
|
name: "nppes_lookup_provider",
|
|
4801
|
-
description: "Keyless CMS/HHS NPPES NPI Registry
|
|
4805
|
+
description: "Keyless CMS/HHS NPPES NPI Registry — every US healthcare provider (NPI-1 individual + NPI-2 organization). EXACT-NPI mode (when `number` supplied): NPI is CMS-Luhn-validated — typo'd NPI is invalid_input, NEVER a fake 'does not exist'; wire carries `number`+version ALONE — co-supplied filters are DROPPED from wire and checked CLIENT-SIDE (filterMatch:{field:bool} + filtersDropped) because NPPES AND-combines number+filters and a mismatch would falsely zero a real active provider. SEARCH mode: required-one of {first_name, last_name, organization_name, taxonomy_description, city, postal_code} (state + enumeration_type are REFINERS ONLY — rejected alone); trailing '*' wildcard needs ≥2 leading literal chars. Returns EXACT-mode { found, provider:{number, enumerationType, active, basic, taxonomies, addresses, practiceLocations, identifiers, otherNames, endpoints, createdEpoch, lastUpdatedEpoch}, filterMatch? } OR SEARCH { providers:[…] } + honest _meta. HONESTY: active = basic.status==='A'; epochs are ms numeric STRINGS → number|null; addresses[] and practiceLocations[] kept SEPARATE (a provider may appear in practiceLocations ONLY); NPPES exposes NO match total — full page → totalAvailable is a LOWER BOUND (totalIsLowerBound) + reach cap (limit ≤ 200, skip ≤ 1,000). Genuine {result_count:0} → honest found:false; {Errors:[…]} 200 body THROWS; 4xx/5xx/timeout THROW; count mismatch → schema_drift. ★NOT a fitness/exclusion/licensure/sanctions determination — cross-check SAM + OFAC; NPI-1 records may surface personal/home addresses + phone/fax verbatim.",
|
|
4802
4806
|
inputSchema: NppesLookupInput,
|
|
4803
4807
|
handler: (input) => nppes.lookupProvider(input),
|
|
4804
4808
|
}),
|
|
@@ -4811,20 +4815,20 @@ export const TOOLS = [
|
|
|
4811
4815
|
}),
|
|
4812
4816
|
defineTool({
|
|
4813
4817
|
name: "cms_query_dataset",
|
|
4814
|
-
description: "Query a CMS Open Payments DKAN datastore distribution by datasetId + index (keyless; openpaymentsdata.cms.gov)
|
|
4818
|
+
description: "Query a CMS Open Payments DKAN datastore distribution by datasetId + index (keyless; openpaymentsdata.cms.gov). Returns { datasetId, index, results (mode), fields:[{name,type,mysqlType,description}], rows:[…verbatim…] } + honest _meta. ★HONESTY: `count` is the EXACT grand total → totalAvailable=count + real offset pagination (NOT a page-length lower bound). `conditions` are server-side self-policing — BAD column → HTTP 400 → invalid_input; filtersDropped is ALWAYS empty (no silent-drop path). limit ≤ 500 is the HARD API cap (higher → invalid_input, no silent clamp). Every column is text, amounts arrive as STRINGS verbatim (null-never-0). ★results:false = COUNT/SCHEMA-discovery mode: no rows, pagination disabled (no livelock), EXACT count + column schema returned. Genuine {count:0} → honest empty; 400/404/HTML/5xx/timeout/missing schema/non-array → THROW. ★SSRF: datasetId (36-char UUID) + index are validated before URL interpolation. ★PII: Open Payments is PUBLIC transparency-BY-LAW data — bounded to targeted vetting (offset ≤ 2000 reach cap), NO enrichment, NO covered_recipient_npi→NPPES auto-join. NOT a conflict-of-interest / fitness / exclusion determination — cross-check SAM + OFAC + OIG-LEIE. The caveat + reach-cap disclosure ride EVERY response.",
|
|
4815
4819
|
inputSchema: CmsQueryDatasetInput,
|
|
4816
4820
|
handler: (input) => cms.queryDataset(input),
|
|
4817
4821
|
}),
|
|
4818
4822
|
// ━━━ FAC Federal Audit Clearinghouse — Single Audit audit-risk vetting (2) ━━━ ADR-0038
|
|
4819
4823
|
defineTool({
|
|
4820
4824
|
name: "fac_search_audits",
|
|
4821
|
-
description: "Search entity Single Audit summaries from the Federal Audit Clearinghouse (keyless via
|
|
4825
|
+
description: "Search entity Single Audit summaries from the Federal Audit Clearinghouse (keyless via api.data.gov DEMO_KEY; api.fac.gov PostgREST `general` table) — the SUBCONTRACTOR / teaming AUDIT-RISK vetting entry point (2 CFR 200 Subpart F / Single Audit Act; every entity expending ≥$750K/yr in federal awards). Filters (all optional, AND-combined): `auditeeUei` (12-char SAM UEI — PRIMARY join key to SAM/USAspending/EDGAR), `auditeeState` (2-letter), `auditYear` (int), `totalExpendedMin`/`totalExpendedMax` (USD). `limit` (≤100, def 25), `offset`. Returns { audits:[{ report_id, auditee_uei, audit_year, auditee_name, auditee_ein, auditee_state, auditee_city, total_amount_expended, fac_accepted_date }] } + honest _meta. Feed report_id (or UEI) to fac_get_findings for the audit-RISK flags. ★PII: a HARDCODED select-allowlist surfaces ONLY entity + audit-summary fields and DELIBERATELY EXCLUDES personal-contact columns — NO caller `select`/column param. HONESTY: totalAvailable is the EXACT Content-Range total (a response header under Prefer:count=exact; '*'/absent/non-numeric denominator → totalAvailable:null + page-fullness hedge, NEVER 0); total_amount_expended is null-never-0; a bad column → PostgREST 400 → invalid_input (filtersDropped ALWAYS empty); genuine [] → honest empty; 400/403/5xx/timeout/HTML/non-array THROW. NOT a debarment/exclusion/fitness determination — cross-check SAM + OFAC. Keyless-first via DEMO_KEY (~10 req/hr shared; set DATA_GOV_API_KEY for production — never logged).",
|
|
4822
4826
|
inputSchema: FacSearchAuditsInput,
|
|
4823
4827
|
handler: (input) => fac.searchAudits(input),
|
|
4824
4828
|
}),
|
|
4825
4829
|
defineTool({
|
|
4826
4830
|
name: "fac_get_findings",
|
|
4827
|
-
description: "Drill into
|
|
4831
|
+
description: "Drill into audit-RISK findings for an entity from the Federal Audit Clearinghouse (keyless via api.data.gov DEMO_KEY; api.fac.gov PostgREST `findings` table) — the risk-detail step after fac_search_audits. At least ONE of `auditeeUei` (12-char UEI) or `reportId` is REQUIRED (empty query refused); optional `auditYear`, `limit` (≤100, def 50), `offset`. Returns { findings:[{ report_id, auditee_uei, audit_year, award_reference, reference_number, is_material_weakness, is_modified_opinion, is_questioned_costs, is_repeat_finding, is_significant_deficiency, is_other_findings, is_other_matters, type_requirement, prior_finding_ref_numbers, riskFlags:{materialWeakness, modifiedOpinion, questionedCosts, repeatFinding, significantDeficiency, otherFindings, otherMatters} }] } + honest _meta. ★RISK-FLAG HONESTY: is_* flags surfaced VERBATIM (\"Y\"/\"N\") PLUS typed riskFlags tri-state (\"Y\"→true / \"N\"→false / blank/absent → null=UNKNOWN) — null NEVER rendered as false (the false-CLEAR class). ★EMPTY ≠ CLEAN: empty findings does NOT confirm a clean audit — the entity may not have filed a Single Audit (below the $750K threshold), may predate FAC coverage, or UEI wrong; a disclosure note fires on any empty result; on empty, confirm an ACCEPTED audit via fac_search_audits. ★PII: HARDCODED select-allowlist, entity + audit-risk fields only. totalAvailable is EXACT Content-Range total ('*'/absent → null + hedge, never 0); 400/403/5xx/timeout/HTML/non-array THROW. NOT a debarment/determination — cross-check SAM + OFAC. DEMO_KEY ~10 req/hr — set DATA_GOV_API_KEY.",
|
|
4828
4832
|
inputSchema: FacGetFindingsInput,
|
|
4829
4833
|
handler: (input) => fac.getFindings(input),
|
|
4830
4834
|
}),
|
|
@@ -4893,26 +4897,26 @@ export const TOOLS = [
|
|
|
4893
4897
|
}),
|
|
4894
4898
|
defineTool({
|
|
4895
4899
|
name: "edgar_filing_index",
|
|
4896
|
-
description: "Bulk cross-filer SEC filing index for a quarter (keyless
|
|
4900
|
+
description: "Bulk sweep — per-filer edgar tools need a CIK. Bulk cross-filer SEC filing index for a quarter (keyless; www.sec.gov EDGAR full-index master.idx). Reads the WHOLE quarter's index (~370K rows: every filer's every filing — CIK|Company|Form|Date|Filename), full-scans it, and returns offset-paginated filings matching CLIENT-SIDE filters with the EXACT total. Input: `year` (≥1993, ≤current year), `quarter` (1..4); optional `formType` (exact, e.g. '8-K'), `cik` (numeric, leading-zero-safe), `companyContains` (LITERAL case-insensitive substring), `dateFrom`/`dateTo` (ISO YYYY-MM-DD), `limit` (≤1000, def 100), `offset`. Returns { year, quarter, indexFile, returned, totalAvailable, filings:[{ cik, cikPadded, companyName, formType, dateFiled, filename, filingUrl }] }. HONESTY: totalAvailable is the EXACT match count over the full quarter scan — never a page length, never a byte-capped subset (SEC ignores HTTP Range). A 0-match result is a genuine EXACT ZERO (complete:true), NOT a truncation. A bounds-valid but unpublished quarter returns HTTP 403 and is surfaced as an AMBIGUOUS error (quarter-not-published OR the 10 req/s rate-block) — never a bare rate-limit and never a fake-empty. A non-index or all-malformed body → schema_drift. A future year / bad quarter → invalid_input pre-fetch. The CURRENT quarter grows daily (totalAvailable is exact as-of-snapshot). filingUrl is a resolvable archive URL. NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — no authoritative CIK↔UEI join.",
|
|
4897
4901
|
inputSchema: EdgarFilingIndexInput,
|
|
4898
4902
|
handler: (input) => edgar.filingIndex(input),
|
|
4899
4903
|
}),
|
|
4900
4904
|
defineTool({
|
|
4901
4905
|
name: "edgar_daily_filing_index",
|
|
4902
|
-
description: "Per-
|
|
4906
|
+
description: "Per-day sibling of edgar_filing_index. Per-day cross-filer SEC filing index (keyless; www.sec.gov EDGAR daily-index master.YYYYMMDD.idx) — reads ONE calendar day's index (~8K rows), full-scans it, and returns offset-paginated filings matching CLIENT-SIDE filters with the EXACT total. Answers the monitoring/alerting question ('every 8-K filed on 2024-01-03'). Input: `date` (required ISO YYYY-MM-DD, ≥1994-01-01, not future); optional `formType` (exact), `cik` (numeric), `companyContains` (LITERAL case-insensitive), `limit` (≤1000, def 100), `offset`. Returns { found, date, year, quarter, indexFile, returned, totalAvailable, filings:[{ cik, cikPadded, companyName, formType, dateFiled, filename, filingUrl }] }. HONESTY: totalAvailable is EXACT match count — never a page length. The daily-index pervasive-403 model is disambiguated via the quarter's index.json existence oracle, RECENCY-AWARE: a day NEWER than the newest published index → found:false, complete:FALSE, retryable not-yet-disseminated note (NEVER a confident empty); an unlisted day INSIDE the covered range (real weekend/holiday) → found:false, complete:true; a LISTED day whose .idx 403s → honest rate_limited; oracle inconclusive → ambiguous upstream_unavailable. A non-real/future date → invalid_input pre-fetch; non-index/all-malformed body → schema_drift. dateFiled normalized from compact YYYYMMDD to ISO. NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — no authoritative CIK↔UEI join.",
|
|
4903
4907
|
inputSchema: EdgarDailyFilingIndexInput,
|
|
4904
4908
|
handler: (input) => edgar.dailyFilingIndex(input),
|
|
4905
4909
|
}),
|
|
4906
4910
|
defineTool({
|
|
4907
4911
|
name: "edgar_company_concept",
|
|
4908
|
-
description: "One filer × one XBRL concept ×
|
|
4912
|
+
description: "One filer × one XBRL concept × complete reported time-series (keyless; data.sec.gov companyconcept), including amendment/restatement history. Sits between edgar_company_facts (many concepts, one filer) and edgar_xbrl_frames (one concept, all filers). start=null for INSTANT concepts. Input: `cikOrTicker`, `concept` (EXACT alnum XBRL tag, e.g. 'Assets'), optional `taxonomy` (us-gaap|dei|ifrs-full, def us-gaap), `unit` (CLIENT-SIDE filter), `form`/`fy` (client-side), `canonicalOnly` (def false), `limit`/`offset`. Returns { found, cik, entityName, taxonomy, concept, label, description, unitsAvailable:[{unit,count}], rows:[{unit, start, end, val, accn, fy, fp, form, filed, frame, canonical}] }. HONESTY M1: period identity is the (start,end) PAIR — the SAME `end` with a DIFFERENT `start` is a different-duration fact (3-month vs 12-month), NOT a revision; a revision is multiple rows sharing the same (start,end) with differing accn/filed/val. DEFAULT returns ALL rows including restatement history + per-row `canonical`; `canonicalOnly:true` dedupes to one canonical row per (unit,start,end), fully disclosed, never a silent drop. Every row is unit-tagged; unitsAvailable discloses ALL units even under a unit filter; val is null-never-0. A bad CIK/taxonomy/concept → 404 → found:false (NEVER fabricated val:0); 5xx/timeout/non-JSON/shape-drift THROWS; `unit` not present → honest empty + available-units note (unit is CLIENT-SIDE, not a path segment). cik/taxonomy/concept are validated path segments (no injection). NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — no authoritative CIK↔UEI join.",
|
|
4909
4913
|
inputSchema: EdgarCompanyConceptInput,
|
|
4910
4914
|
handler: (input) => edgar.companyConcept(input),
|
|
4911
4915
|
}),
|
|
4912
4916
|
// ━━━ Socrata / SODA — keyless SLED + E-rate open data (2) ━━━ ADR-0004
|
|
4913
4917
|
defineTool({
|
|
4914
4918
|
name: "socrata_query",
|
|
4915
|
-
description: "Query rows from an allowlisted Socrata/SODA open-data portal (keyless; ~a dozen US state portals + USAC E-rate on one identical API — state spend/checkbook/contract/vendor-payment datasets). Input `domain` (curated allowlist enum — the SSRF host guard), `datasetId` (4x4, from socrata_discover_datasets), optional SoQL `select`/`where`/`order`/`q`, `limit` (≤1000, def 100), `offset`, `withTotal` (def true). HONESTY: SODA's row response has no total, so a count(*) companion supplies an exact totalAvailable; if it fails the rows still return with totalAvailable:null + a note (hasMore is then inferred from page-fill, never a false complete). Genuine-empty ⇒ complete:true/total:0; an outage/400/404 THROWS (never a fake empty). Value fields are strings.",
|
|
4919
|
+
description: "Query rows from an allowlisted Socrata/SODA open-data portal (keyless; ~a dozen US state portals + USAC E-rate on one identical API — state spend/checkbook/contract/vendor-payment datasets). State procurement mirrors: NY ehig-g5x3, NJ ubnu-tqu7, WA s8d5-pj78, MA cthru.data.socrata.com pegc-naaa (~49M payment rows). Full map: read resource samgov://data-map/state-local. Input `domain` (curated allowlist enum — the SSRF host guard), `datasetId` (4x4, from socrata_discover_datasets), optional SoQL `select`/`where`/`order`/`q`, `limit` (≤1000, def 100), `offset`, `withTotal` (def true). HONESTY: SODA's row response has no total, so a count(*) companion supplies an exact totalAvailable; if it fails the rows still return with totalAvailable:null + a note (hasMore is then inferred from page-fill, never a false complete). Genuine-empty ⇒ complete:true/total:0; an outage/400/404 THROWS (never a fake empty). Value fields are strings.",
|
|
4916
4920
|
inputSchema: SocrataQueryInput,
|
|
4917
4921
|
handler: (input) => socrata.query(input),
|
|
4918
4922
|
}),
|
|
@@ -4925,7 +4929,7 @@ export const TOOLS = [
|
|
|
4925
4929
|
// ━━━ CKAN datastore_search — keyless SLED open data (2) ━━━ ADR-0006
|
|
4926
4930
|
defineTool({
|
|
4927
4931
|
name: "ckan_query",
|
|
4928
|
-
description: "Query rows from an allowlisted CKAN datastore resource (keyless;
|
|
4932
|
+
description: "Query rows from an allowlisted CKAN datastore resource (keyless; state/city spend/checkbook/procurement/vendor tables on the CKAN Action API). VA eVA PO lines: host=data.virginia.gov resourceId=3c7f1bde-35b0-4fbf-b89c-978a19124d53. Full map: read resource samgov://data-map/state-local. Input `host` (curated allowlist enum — the SSRF host guard: data.ca.gov, data.virginia.gov, data.boston.gov), `resourceId` (36-char lowercase UUID, from ckan_discover_datasets), optional `q` (full-text), `filters` (constrained object {field:value} we JSON.stringify), `sort`, `limit` (≤1000, def 100), `offset`. HONESTY: CKAN's envelope carries a real result.total — the DEFAULT is an EXACT total (exact totalAvailable + hasMore); the rare estimated total (total_was_estimated:true) is disclosed via totalIsEstimated + a note and does NOT drive pagination (it can be above OR below the truth). Genuine-empty ⇒ complete:true/total:0; an outage/404/409 or success:false THROWS (never a fake empty). Values are typed per result.fields[].type.",
|
|
4929
4933
|
inputSchema: CkanQueryInput,
|
|
4930
4934
|
handler: (input) => ckan.query(input),
|
|
4931
4935
|
}),
|
|
@@ -4950,26 +4954,26 @@ export const TOOLS = [
|
|
|
4950
4954
|
}),
|
|
4951
4955
|
defineTool({
|
|
4952
4956
|
name: "fdic_bank_failures",
|
|
4953
|
-
description: "Historical FDIC-insured bank failures & assistance transactions (keyless
|
|
4957
|
+
description: "Historical FDIC-insured bank failures & assistance transactions (keyless; api.fdic.gov/banks/failures). CERT links a failure back to fdic_search_institutions / fdic_institution_financials. Filters: `state` (2-letter → PSTALP — NOTE: /failures state field is PSTALP, NOT STALP), `failYear` (→FAILYR), `cert` (→CERT, the STABLE entity key). `limit` (≤1000), `offset` (≤100000), `sortBy` (FAILDATE/COST/QBFASSET/QBFDEP/NAME/FAILYR, def FAILDATE), `sortOrder` (def DESC). Returns { failures:[{ name, cert, failDate, failYear, city, state, resolutionType, resolutionFund, estimatedLossUSD, depositsUSD, assetsUSD, id }] }. NO name/city filter — FDIC /failures `search` param is IGNORED (returns the whole dataset); to find a specific bank's failure, resolve its CERT via fdic_search_institutions. HONESTY: totalAvailable is EXACT meta.total (stable across offset). failDate normalized from FDIC's M/D/YYYY to ISO YYYY-MM-DD (unrecognized → surfaced raw + disclosed, never nulled). COST/QBFDEP/QBFASSET are $thousands normalized to whole USD ×1000 (null-never-0 — genuine 0 = a no-loss assisted transaction stays 0; NEGATIVE COST = a net DIF recovery/gain, not a loss; absent → null). ONLY honest empty is meta.total:0/data:[] → complete:true/total:0; any other envelope (400 errors[]/404/non-JSON/missing meta or data) THROWS (never fake-empty). The point-in-time snapshot build time is disclosed. NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
|
|
4954
4958
|
inputSchema: FdicBankFailuresInput,
|
|
4955
4959
|
handler: (input) => fdic.bankFailures(input),
|
|
4956
4960
|
}),
|
|
4957
4961
|
defineTool({
|
|
4958
4962
|
name: "fdic_institution_history",
|
|
4959
|
-
description: "Institution-level STRUCTURAL-CHANGE event log for FDIC-insured banks (keyless
|
|
4963
|
+
description: "Institution-level STRUCTURAL-CHANGE event log for FDIC-insured banks (keyless; api.fdic.gov/banks/history) — mergers, absorptions, consolidations, failures, name/location/charter/regulator changes, branch open/close, trust-power grants & FRS-membership. CERT-linked merger lineage: each row carries acquiring/outgoing/surviving institution CERT + name, linking to fdic_search_institutions / fdic_bank_failures. Filters (all optional, AND-combined): `cert` (→CERT, PRIMARY lookup), `changeCode` (→CHANGECODE; e.g. 223=merger, 211=failure, 721=branch closing), `effYear` (→EFFYEAR), `state` (2-letter → PSTALP — NOTE: /history uses PSTALP, NOT STALP). `limit` (≤1000), `offset` (≤100000), `sortBy` (EFFDATE/PROCDATE/CHANGECODE/TRANSNUM), `sortOrder`. Returns { history:[{ cert, instName, state, changeCode, changeDescription, effectiveDate, processDate, effYear, transNum, acquirerCert, acquirerName, outgoingCert, outgoingName, survivingCert, survivingName, id }] }. NO name/city filter — FDIC /history `search` returns 0 for INSTNAME (a false-empty); resolve CERT via fdic_search_institutions first. HONESTY: totalAvailable is EXACT meta.total; changeDescription is FDIC's CHANGECODE_DESC verbatim (changeCode is authoritative — never hand-mapped); effectiveDate/processDate normalized to ISO YYYY-MM-DD (unrecognized → surfaced raw); acquirer/outgoing/surviving CERTs null on non-merger events (null-never-0); ONLY honest empty is meta.total:0/data:[] → complete:true/total:0; any other envelope THROWS (never fake-empty). NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
|
|
4960
4964
|
inputSchema: FdicInstitutionHistoryInput,
|
|
4961
4965
|
handler: (input) => fdic.institutionHistory(input),
|
|
4962
4966
|
}),
|
|
4963
4967
|
defineTool({
|
|
4964
4968
|
name: "fdic_industry_summary",
|
|
4965
|
-
description: "FDIC
|
|
4969
|
+
description: "FDIC banking-sector ANNUAL AGGREGATES — total assets, deposits, net income, equity & net interest income + institution/office/branch/employee counts for the whole US OR one state, split by charter class (keyless; api.fdic.gov/banks/summary). Filters (all optional): `year` (→YEAR), `state` (→STALP — NOTE: /summary uses STALP, NOT PSTALP; accepts TX/CA/DC/GU/PR or ROLL-UP codes USA/US/OT/PI), `charterClass` (CB=commercial, SI=savings; omit for both). `limit` (≤1000), `offset` (≤100000), `sortBy` (YEAR/ASSET/DEP/NETINC/BANKS), `sortOrder`. Returns { summary:[{ year, charterClass, charterClassCode, geography, stateCode, stateFips, scope, isRollup, institutionCount, officeCount, branchCount, employeeCount, totalAssetsUSD, totalDepositsUSD, netIncomeUSD, totalEquityUSD, netInterestIncomeUSD, id }] }. ★ROLL-UP HONESTY: STALP ∈ {USA,US,OT,PI} are GEOGRAPHIC AGGREGATES (isRollup:true). NEVER sum a roll-up row with jurisdiction rows or across scopes — USA is the one national figure; a roll-up is NOT a state. ★netInterestIncomeUSD is net interest INCOME ($ sum), NOT the margin ratio; NO ratio fields (ROA/ROE); derive from netIncomeUSD/totalAssetsUSD/totalEquityUSD. NO name/city filter — FDIC /summary `search` is ignored; drill via fdic_search_institutions. HONESTY: totalAvailable is EXACT meta.total; money ($thousands → whole USD ×1000, null-never-0; genuine 0 stays 0; absent → null); counts pass through unscaled; non-int year rejected pre-fetch; ONLY honest empty is meta.total:0/data:[] → complete:true/total:0; any other envelope THROWS. NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
|
|
4966
4970
|
inputSchema: FdicIndustrySummaryInput,
|
|
4967
4971
|
handler: (input) => fdic.industrySummary(input),
|
|
4968
4972
|
}),
|
|
4969
4973
|
// ━━━ FDIC BankFind Suite — WITHIN-SOURCE DEPTH: counterparty risk ratios + branch deposits (2) ━━━ ADR-0040
|
|
4970
4974
|
defineTool({
|
|
4971
4975
|
name: "fdic_risk_ratios",
|
|
4972
|
-
description: "FDIC
|
|
4976
|
+
description: "FDIC risk ratios for ONE institution by certificate number (keyless; api.fdic.gov/banks/financials) — profitability (ROA/pretax ROA/ROE), net interest margin, efficiency ratio, asset quality (net charge-offs to loans), capital adequacy (leverage, tier-1 risk-based, total risk-based ratios) + tier-1 capital level. Input: `cert` (REQUIRED FDIC certificate, from fdic_search_institutions), `reportDate` (optional YYYYMMDD quarter-end → REPDTE; omit for full quarterly time-series), `limit` (≤1000), `offset`, `sortBy` (REPDTE/ROA/ROE/RBCRWAJ/EEFFR), `sortOrder`. Returns { cert, ratios:[{ cert, reportDate, cblrFramework, returnOnAssetsPct, preTaxReturnOnAssetsPct, returnOnEquityPct, netInterestMarginPct, efficiencyRatioPct, netChargeOffsToLoansPct, leverageRatioPct, tier1RiskBasedCapitalRatioPct, totalRiskBasedCapitalRatioPct, tier1CapitalUSD, id }] }. ★UNITS: every *Pct field is FDIC-published PERCENTAGE verbatim (no scaling); tier1CapitalUSD is $thousands × 1000. ★NULL-NEVER-0: not-reported ratio is null (never 0). ★CBLR banks (cblrFramework:true) do NOT report risk-based capital ratios — FDIC returns literal 0 for totalRiskBased only; this tool maps 0→null for BOTH tier1RiskBased and totalRiskBased; null is a framework artifact, not a 0% red flag — read alongside leverageRatioPct. No ratio is recomputed; each is FDIC's published Call-Report figure verbatim. HONESTY: totalAvailable is EXACT meta.total; ONLY honest empty is meta.total:0/data:[] → complete:true/total:0; any other envelope THROWS. NOTE: regulatory metrics, NOT a soundness rating. FDIC keys on CERT, not SAM UEI/DUNS.",
|
|
4973
4977
|
inputSchema: FdicRiskRatiosInput,
|
|
4974
4978
|
handler: (input) => fdic.riskRatios(input),
|
|
4975
4979
|
}),
|
|
@@ -4982,7 +4986,7 @@ export const TOOLS = [
|
|
|
4982
4986
|
// ━━━ USITC Harmonized Tariff Schedule — keyless import-tariff / duty-rate lookup (1) ━━━ ADR-0039
|
|
4983
4987
|
defineTool({
|
|
4984
4988
|
name: "hts_lookup",
|
|
4985
|
-
description: "Look up US import-tariff classification + duty rates from the USITC Harmonized Tariff Schedule (keyless; hts.usitc.gov/reststop/search)
|
|
4989
|
+
description: "Look up US import-tariff classification + duty rates from the USITC Harmonized Tariff Schedule (keyless; hts.usitc.gov/reststop/search). A single `query` serves BOTH modes: KEYWORD (e.g. 'laptop') OR HTS number (e.g. '8471.30'). Returns { query, lines:[{ htsno, statisticalSuffix, indent, description, units, columnOneGeneral, specialPreferential, columnTwo, additionalDuties, footnotes, quotaQuantity, effectivePeriod, status, isChapter99 }] } + honest _meta. ★DUTY-RATE HONESTY: columnOneGeneral, specialPreferential, and columnTwo are AUTHORITATIVE VERBATIM TEXT — 'Free', '35%', '0.47¢/kg', compound/range, or null — NEVER coerced to a number (0/NaN fabricates a false 'duty-free'); empty Special ('') → null = NO special rate (NEVER read as Free). ★HIERARCHY: rate stated at a shallower level and inherits downward; read to the nearest ancestor with a non-empty rate; blank deepest ≠ 'no duty'. ★ADDITIONAL DUTIES: `additionalDuties` is frequently null even when Section 301/232 duties apply — the real additional duty rides Chapter-99 rows (isChapter99:true) and STACKS on the base rate. ★COMPLETENESS: endpoint returns the FULL match array; offset IGNORED → totalAvailable is the EXACT served array length; paging CLIENT-SIDE. query must be ≥3 non-whitespace chars (1-2 → invalid_input). limit (≤200, def 50), offset. No-match → honest empty; 404/5xx/timeout/non-array/HTML → schema_drift THROW; transient 400 → upstream_unavailable. NOT a binding CBP ruling or landed-cost quote — duty owed depends on country of origin + trade program + Ch-99 stacking; confirm via CBP (CROSS/eRulings).",
|
|
4986
4990
|
inputSchema: HtsLookupInput,
|
|
4987
4991
|
handler: (input) => usitc.htsLookup(input),
|
|
4988
4992
|
}),
|
|
@@ -4997,7 +5001,7 @@ export const TOOLS = [
|
|
|
4997
5001
|
// rides ONLY in the POST body (v2) — never a URL/header/label/_meta/log.
|
|
4998
5002
|
defineTool({
|
|
4999
5003
|
name: "bls_timeseries",
|
|
5000
|
-
description: "Fetch
|
|
5004
|
+
description: "Fetch one or more BLS time-series over a year range (keyless v1 default; optional free BLS_API_KEY lifts to v2; api.bls.gov POST). At least one of `series` (curated enum key) or `seriesId` (raw ^[A-Z0-9]{1,25}$; covers 25-char OEWS IDs) is required; both may be combined. Optional `startYear`/`endYear` (defaults to active tier's span cap). Returns { series:[{ seriesId, key, meaning, units, observations:[{year, period, periodName, value:number|null, valueUnavailable:bool, footnotes:[{code,text}], latest:bool}], observationCount, coveredRange:{from,to} }] } + honest _meta. HONESTY: BLS '-' unavailable marker → value:null (NEVER 0); valueUnavailable:true on the observation + footnote reason lifted into _meta.notes so the gap is DISCLOSED, never silent. Each series carries its own units label — an ECI '…A' series is a 12-month PERCENT CHANGE, NOT an index level; CPI/PPI are index levels; CES nonfarm employment is thousands of persons. Do NOT compare values across series without reading each units label. status !== 'REQUEST_SUCCEEDED' THROWS: REQUEST_NOT_PROCESSED (v1 daily limit) → rate_limited retryable; REQUEST_FAILED → upstream_unavailable. Series count refused over active tier cap (v1: 25 series/~10yr; v2: 50 series/~20yr) — overflow is NEVER silently dropped. Span is CLAMPED to tier cap before the fetch and disclosed. totalAvailable is null (batch fetch has no upstream total). A typo'd seriesId returns an empty series with 'Invalid Series' upstream message — NOT a real available series. BLS_API_KEY rides ONLY in the POST body, never URL/label/_meta/log.",
|
|
5001
5005
|
inputSchema: BlsTimeseriesInput,
|
|
5002
5006
|
handler: (input) => bls.timeseries(input),
|
|
5003
5007
|
}),
|
|
@@ -5005,12 +5009,12 @@ export const TOOLS = [
|
|
|
5005
5009
|
// The LEVEL layer next to bls_timeseries's ESCALATION layer: mean/median annual &
|
|
5006
5010
|
// hourly wages + employment by SOC occupation × geography — the highest-value B2G
|
|
5007
5011
|
// BLS slice (labor-rate benchmarking) that bls_timeseries structurally cannot reach
|
|
5008
|
-
// (OEWS IDs are 25 chars
|
|
5009
|
-
// ID INTERNALLY from validated structured inputs; REUSES the same POST/JSON transport
|
|
5012
|
+
// (OEWS IDs are 25 chars; bls_timeseries raw-seriesId cap is now widened to 25). BUILDS
|
|
5013
|
+
// the 25-char series ID INTERNALLY from validated structured inputs; REUSES the same POST/JSON transport
|
|
5010
5014
|
// + parseBlsBody status-throw + mapObservation ("-"→null-never-0) + tier/key seam.
|
|
5011
5015
|
defineTool({
|
|
5012
5016
|
name: "bls_oews_wages",
|
|
5013
|
-
description: "
|
|
5017
|
+
description: "BLS OEWS occupational wage/employment benchmarking by area × occupation × datatype (keyless; api.bls.gov). Builds validated 25-char series IDs from structured inputs and batches them into one POST — NO year input (OEWS serves only the latest annual release). Inputs: `occupation` (curated enum, e.g. 'software_developer') or `soc` (raw 6-digit SOC, ^[0-9]{6}$ NO hyphen — at least ONE required); `area` (default 'national', 2-letter USPS state, or 5-digit CBSA metro code); `datatype` (default 'annual_mean'; annual_mean/annual_median/hourly_mean/hourly_median/employment). Returns { results:[{ area:{type,code,label}, occupation:{soc,key,label}, measure:{key,code,units}, value:number|null, valueUnavailable:bool, referenceYear, referencePeriod, footnotes, seriesId }] } + honest _meta. ★H1: OEWS is an ANNUAL point-in-time snapshot (reference May <year>); BLS API serves ONLY the most recent release, may lag ~1 year. NOT monthly/current-quarter. ★H2: a built-ID with no published value (occupation not surveyed or suppressed in that area) → value:null (NOT a tool error). ★H3: measure.units is set from datatype — annual_mean/annual_median=dollars/year; hourly_mean/hourly_median=dollars/hour; employment=count. NEVER mislabeled. ★H4: the API returns real numerics (no '#' top-code). area×occupation×datatype is capped at the tier's series cap (v1 25 / v2 50) and refused over-cap with the count named — never silently truncated; REQUEST_NOT_PROCESSED → rate_limited THROWS. Active tier (v1 keyless ~25/day or v2 BLS_API_KEY ~500/day) and series-cap limits disclosed.",
|
|
5014
5018
|
inputSchema: BlsOewsWagesInput,
|
|
5015
5019
|
handler: (input) => bls.oewsWages(input),
|
|
5016
5020
|
}),
|
|
@@ -5027,7 +5031,7 @@ export const TOOLS = [
|
|
|
5027
5031
|
// → honest empty. NEW gate key "bls_qcew"; NO BLS_API_KEY on this keyless path.
|
|
5028
5032
|
defineTool({
|
|
5029
5033
|
name: "bls_qcew",
|
|
5030
|
-
description: "BLS QCEW (Quarterly Census of Employment & Wages) — county×NAICS
|
|
5034
|
+
description: "BLS QCEW (Quarterly Census of Employment & Wages) — county×NAICS market-size / wages / location-quotient (keyless; data.bls.gov/cew Open Data Access CSV, un-rate-limited). Inputs: `mode` (REQUIRED {area,industry}); `area` (area_fips ^[0-9A-Za-z]{1,6}$ — REQUIRED for mode=area); `industry` (NAICS ^[0-9]{1,6}$ DIGIT-ONLY — REQUIRED for mode=industry; hyphenated 31-33 404s, use digit aggregate); `year` (REQUIRED), `quarter` (REQUIRED 1|2|3|4); client-side `ownership`/`aggregationLevel`/`sizeCode`; `limit`/`offset`. Returns { found, mode, rows:[{ area_fips, own_code, industry_code, agglvl_code, size_code, base:{disclosed, disclosureCode, qtrly_estabs, month1/2/3_emplvl, total_qtrly_wages, taxable_qtrly_wages, qtrly_contributions, avg_wkly_wage}, locationQuotient:{disclosed, disclosureCode, lq_…}, overTheYear:{disclosed, disclosureCode, oty_…} }] }. ★DISCLOSURE HONESTY: each row has three disclosure codes (base/lq/oty). QCEW encodes SUPPRESSED values as literal 0 — under 'N': confidential emplvl/wage/avg-wkly → null (WITHHELD), estab count + oty-estab change stay DISCLOSED; under '-': WHOLE block → null; under blank: genuine reported/NEGATIVE 0 SURVIVES. NEVER blanket 0→null; null carries disclosed:false + raw disclosureCode; suppression note fires on any suppressed row. HONESTY: totalAvailable is EXACT filtered row count (fetch-once; QCEW does not paginate); per-tuple HTTP 404 → honest empty; 5xx/timeout THROW; 200 non-CSV/renamed header/wrong field-count → schema_drift THROW. Do-NOT-sum-across-agglvl/ownership note rides every response.",
|
|
5031
5035
|
inputSchema: BlsQcewInput,
|
|
5032
5036
|
handler: (input) => bls.qcew(input),
|
|
5033
5037
|
}),
|
|
@@ -5083,7 +5087,7 @@ export const TOOLS = [
|
|
|
5083
5087
|
// (silent no-op). The 15,000-record retrieval window is disclosed, not hidden.
|
|
5084
5088
|
defineTool({
|
|
5085
5089
|
name: "nih_reporter_search_projects",
|
|
5086
|
-
description: "Search awarded NIH RePORTER research-
|
|
5090
|
+
description: "Search awarded NIH RePORTER research-grant projects (keyless; api.reporter.nih.gov v2, POST/JSON), joinable to SAM/USAspending via primary_uei. LIVE-CONFIRMED-narrowing criteria ONLY, AND-combined: orgStates (UPPERCASE 2-letter USPS — lowercase/unknown code silently returns zeros), orgNames (≤512 chars each, ≤20 names), fiscalYears (int array, 1985..currentYear+1, ≤20), limit (1..500), offset (0..14,999). Returns { projects:[{ projectNum, projectTitle, fiscalYear, awardAmount, organization:{name, state, primaryUei, primaryDuns, ueis, duns}, principalInvestigators, contactPiName, fundingIc }] } + honest _meta. HONESTY: records are RESEARCH GRANTS, NOT procurement contracts — primaryUei joins SAM/USAspending but the award nature differs (disclosed in every _meta.notes). totalAvailable = EXACT meta.total (NEVER the page size, NEVER a lower bound). NIH caps keyless retrieval at the first 15,000 records (offset 0..14,999); offset ≥ 15,000 → invalid_input; past the window the count stays exact while records are UNREACHABLE (disclosed in a note). An unscoped query returns the first page + exact total + narrow-your-criteria note. agencyIcCodes is NOT a filter (NIH silently drops it — would be a false 'applied'). awardAmount is number|null (genuine $0 is 0, absent is null). Genuine total:0 → complete:true/total:0; outage/5xx THROWS; 400 (bad offset/limit/type) → invalid_input; 200 not {meta,results} or non-numeric meta.total → schema_drift.",
|
|
5087
5091
|
inputSchema: NihSearchProjectsInput,
|
|
5088
5092
|
handler: (input) => nih.searchProjects(input),
|
|
5089
5093
|
}),
|
|
@@ -5099,7 +5103,7 @@ export const TOOLS = [
|
|
|
5099
5103
|
// HTTP 200 loud-fails (never a fake empty); grant≠contract in every response.
|
|
5100
5104
|
defineTool({
|
|
5101
5105
|
name: "nsf_search_awards",
|
|
5102
|
-
description: "Search awarded NSF research-
|
|
5106
|
+
description: "Search awarded NSF research-grant awards (keyless; api.nsf.gov/services/v1/awards.json), joinable to SAM/USAspending via ueiNumber/parentUeiNumber. Filters: keyword (MULTI-WORD is OR-tokenized — 'machine learning' = machine OR learning, disclosed), awardeeStateCode (UPPERCASE 2-letter USPS — non-state typo silently returns 0), awardeeName, ueiNumber (12-char UEI — EXACT SAM/USAspending join), parentUeiNumber, pdPIName, dateStart/dateEnd (strict mm/dd/yyyy — wrong format silently mis-parsed), limit (1..100), offset (0..9999). Returns { awards:[{ id, title, agency, cfdaNumber, transType, awardee:{name, city, stateCode, ueiNumber, parentUeiNumber}, principalInvestigator, coPrincipalInvestigators, programOfficer, amounts:{fundsObligatedAmt, estimatedTotalAmt, fundsObligatedByYear}, dates, program, activeAward, historicalAward }] } (abstract EXCLUDED — use nsf_get_award) + honest _meta. HONESTY: NSF Awards are RESEARCH GRANTS, NOT procurement contracts (ueiNumber joins SAM/USAspending but the award nature differs — disclosed every response). totalAvailable = EXACT metadata.totalCount below 10,000; SATURATES at 10,000 (ES track_total_hits cap → totalIsLowerBound:true + note; first 10,000 only retrievable). NSF caps retrieval at offset+rpp ≤ 10,000 (offset ≥ 10,000 → invalid_input). fundsObligatedAmt/estimatedTotalAmt: STRINGS → number|null (genuine $0 is 0, absent is null). Genuine totalCount:0 → complete:true/total:0; serviceNotification at HTTP 200 → THROWS; outage/5xx THROWS; 200 not {response:{award,metadata}} or non-numeric totalCount → schema_drift.",
|
|
5103
5107
|
inputSchema: NsfSearchAwardsInput,
|
|
5104
5108
|
handler: (input) => nsf.searchAwards(input),
|
|
5105
5109
|
}),
|
|
@@ -5123,7 +5127,7 @@ export const TOOLS = [
|
|
|
5123
5127
|
// AND-tokenized (disclosed); trial≠federal-award caveat in every response.
|
|
5124
5128
|
defineTool({
|
|
5125
5129
|
name: "clinicaltrials_search_studies",
|
|
5126
|
-
description: "Search federally-registered clinical-research studies with LEAD-SPONSOR / COLLABORATOR /
|
|
5130
|
+
description: "Search federally-registered clinical-research studies with LEAD-SPONSOR / COLLABORATOR / FUNDING-SOURCE enrichment (keyless; clinicaltrials.gov/api/v2/studies). Filters: query.term (broad free-text), sponsor (→query.spons, fuzzy NAME search), condition (→query.cond), location (→query.locn), overallStatus (frozen 14-value enum), funderType (frozen 4-value enum nih/fed/industry/other), pageSize (1..1000), pageToken (OPAQUE cursor). Returns { studies:[{ nctId, briefTitle, orgStudyId, organization:{name,class}, leadSponsor:{name,class}, collaborators:[{name,class}], fundingClass, overallStatus, startDate, studyType, phases, conditions }] } (briefSummary EXCLUDED — use clinicaltrials_get_study) + honest _meta. HONESTY: countTotal=true ALWAYS sent → totalAvailable = EXACT uncapped total (NEVER studies.length; missing/non-number totalCount → schema_drift). Pagination is OPAQUE cursor (nextCursor = nextPageToken back as pageToken; terminal = token absent; bad token → HTTP 400 THROWS). funderType re-validated in handler — UNLISTED value silently returns totalCount:0 at HTTP 200 (fake-empty trap) → invalid_input pre-fetch. funderType is an OVERLAPPING facet (counts MUST NOT be summed). MULTI-WORD query.term/sponsor/condition is AND-conjunctive (all tokens must co-occur — disclosed). Registered trial is NOT a federal award; leadSponsor.name is FREE TEXT (not a UEI) → NOMINAL name match only — disclosed every response. Genuine totalCount:0 → complete:true/total:0; bad overallStatus/pageToken → HTTP 400/404 THROWS; outage/5xx THROWS. Feed nctId to clinicaltrials_get_study.",
|
|
5127
5131
|
inputSchema: ClinicaltrialsSearchStudiesInput,
|
|
5128
5132
|
handler: (input) => clinicaltrials.searchStudies(input),
|
|
5129
5133
|
}),
|
|
@@ -5144,7 +5148,7 @@ export const TOOLS = [
|
|
|
5144
5148
|
// every response. NO free-text ⇒ no tokenization.
|
|
5145
5149
|
defineTool({
|
|
5146
5150
|
name: "clinicaltrials_facet_counts",
|
|
5147
|
-
description: "Aggregate
|
|
5151
|
+
description: "Aggregate EXACT per-value study counts over the WHOLE ClinicalTrials.gov registry for 1..11 whitelisted ENUM fields (keyless; clinicaltrials.gov/api/v2/stats/field/values). Input `fields` (deduped): OverallStatus, StudyType, Phase, LeadSponsorClass (NIH/FED/OTHER_GOV/INDUSTRY/OTHER/NETWORK/INDIV/UNKNOWN/AMBIG — richer than the 4-value funderType in the search tool), Sex, DesignAllocation, DesignPrimaryPurpose, DesignInterventionModel, DesignMasking, DesignObservationalModel, DesignTimePerspective. Returns { facets:[{ field, fieldPath, valueType, uniqueValuesCount, missingStudiesCount, returned, truncated, overlapping, values:[{value, studiesCount}] }] } + honest _meta. HONESTY: each studiesCount/uniqueValuesCount is EXACT (typeof NUMBER — non-number → schema_drift); non-ENUM shape for a whitelisted field → schema_drift. _meta.totalAvailable/returned count DISTINCT FIELD VALUES, NOT studies — see facets[].values[].studiesCount / clinicaltrials_search_studies for study counts. Counts cover the ENTIRE registry and are NOT filterable — /stats/field/values rejects query.*/filter.*/pageSize (HTTP 400). returned<uniqueValuesCount → truncated (hard cap 250). Phase is ARRAY-valued (overlapping:true, MUST NOT sum counts); scalar fields partition the registry minus missingStudiesCount. High missingStudiesCount → buckets cover a MINORITY of the registry. MANDATORY CAVEAT: facet counts are distributions over trial REGISTRATIONS, NOT federal awards; LeadSponsorClass is the funding-SOURCE class, not a UEI-keyed award join. Unlisted field → invalid_input pre-fetch; 404/400/5xx → THROWS.",
|
|
5148
5152
|
inputSchema: ClinicaltrialsFacetCountsInput,
|
|
5149
5153
|
handler: (input) => clinicaltrials.facetCounts(input),
|
|
5150
5154
|
}),
|
|
@@ -5235,7 +5239,7 @@ export const TOOLS = [
|
|
|
5235
5239
|
// is a planned, separately-guarded addition).
|
|
5236
5240
|
defineTool({
|
|
5237
5241
|
name: "arcgis_hub_discover_datasets",
|
|
5238
|
-
description: "Discover ArcGIS Hub datasets by keyword — the SLED/GIS open-data layer that Socrata and CKAN do NOT cover (keyless; hub.arcgis.com/api/v3/datasets).
|
|
5242
|
+
description: "Discover ArcGIS Hub datasets by keyword — the SLED/GIS open-data layer that Socrata and CKAN do NOT cover (keyless; hub.arcgis.com/api/v3/datasets). Input `query` (→q, REQUIRED, ≥2 non-whitespace chars — broad whole-Hub scan refused), `openDataOnly` (default TRUE → filter[openData]=true, the B2G-relevant designated-open-data subset; false broadens to all shared items), `limit` (1..100, def 20 → page[size]), `offset` (0-based → page[start]=offset+1). Returns { query, openDataOnly, datasets:[{ id, name, description, owner, orgName, source, region, type, sector, keywords, downloadable, hasApi, created, modified, landingPage, itemId }] } + honest _meta. ★PROVENANCE (the crux): ArcGIS Hub is a GLOBAL, OPEN publishing platform — results include NON-US and NON-GOVERNMENTAL publishers. This is a DISCOVERY aid, NOT a curated official-source allowlist (unlike socrata_query): the per-row owner/orgName/source/region are surfaced VERBATIM so you can VET the publisher, and the global-platform caveat rides EVERY response. DISCOVERY ONLY — metadata + links; to read rows, arcgis_feature_query covers only its curated allowlist; other datasets must be followed on their own endpoint. HONESTY: totalAvailable = EXACT Hub match count (meta.total, NEVER data.length); pagination is 0-based offset; scalars null-never-empty, booleans null-preserving; genuine no-match → complete:true/returned:0; 429 → rate_limited / 5xx/timeout → upstream_unavailable THROWS; 200 non-JSON / non-array → schema_drift.",
|
|
5239
5243
|
inputSchema: ArcgisHubDiscoverInput,
|
|
5240
5244
|
handler: (input) => arcgisHub.discoverDatasets(input),
|
|
5241
5245
|
}),
|
|
@@ -5259,12 +5263,12 @@ export const TOOLS = [
|
|
|
5259
5263
|
}),
|
|
5260
5264
|
// ━━━ Bonfire (Euna) — keyless per-org open-opportunity RSS (SLED bids) ━━━
|
|
5261
5265
|
// SLED bid campaign. Thousands of US state/local govs on Bonfire expose a keyless
|
|
5262
|
-
// RSS of open opportunities. Ships a curated
|
|
5266
|
+
// RSS of open opportunities. Ships a curated 195-org live-verified seed directory
|
|
5263
5267
|
// (Bonfire's authoritative org API is auth-gated → out of bounds). Fixed-suffix
|
|
5264
5268
|
// SSRF (.bonfirehub.com). RSS = the complete open set (totalAvailable honest).
|
|
5265
5269
|
defineTool({
|
|
5266
5270
|
name: "bonfire_list_organizations",
|
|
5267
|
-
description: "List US governments on the Bonfire (Euna) eProcurement platform — the directory for bonfire_search_opportunities (keyless). Bonfire hosts thousands of US state/local governments' open-bid portals, each with a keyless RSS feed. Filter the curated seed by `state` (2-letter) / `query` (case-insensitive name substring); `limit`(1..200)/`offset`. Output: { organizations:[{ org, name, state }] }. Feed a result's `org` to bonfire_search_opportunities. ★HONESTY: this is a CURATED, live-verified SEED of
|
|
5271
|
+
description: "List US governments on the Bonfire (Euna) eProcurement platform — the directory for bonfire_search_opportunities (keyless). Bonfire hosts thousands of US state/local governments' open-bid portals, each with a keyless RSS feed. Filter the curated seed by `state` (2-letter) / `query` (case-insensitive name substring); `limit`(1..200)/`offset`. Output: { organizations:[{ org, name, state }] }. Feed a result's `org` to bonfire_search_opportunities. ★HONESTY: this is a CURATED, live-verified SEED of 195 US orgs — Bonfire has NO keyless org-list API (its authoritative directory is auth-gated, out of bounds), and Euna markets up to ~900 US orgs, so the seed is PARTIAL (disclosed in _meta); probe `{slug}.bonfirehub.com/opportunities/rss` to extend. totalAvailable = the exact filtered seed count.",
|
|
5268
5272
|
inputSchema: BonfireListOrganizationsInput,
|
|
5269
5273
|
handler: (input) => bonfire.listOrganizations(input),
|
|
5270
5274
|
}),
|
|
@@ -5341,7 +5345,7 @@ export const TOOLS = [
|
|
|
5341
5345
|
// vintage enum is the (benchmark,vintage) UNION ([M2]); GEOIDs stay strings.
|
|
5342
5346
|
defineTool({
|
|
5343
5347
|
name: "census_geocode_address",
|
|
5344
|
-
description: "Resolve a one-line US address →
|
|
5348
|
+
description: "Resolve a one-line US address → matched address(es) + the Census GEOGRAPHIES for set-aside / place-of-performance analysis (US Census Geocoder, keyless; geocoding.geo.census.gov/geocoder/geographies/onelineaddress). Input: `address` (≤500 chars), optional `benchmark` (default Public_AR_Current), `vintage` (default Current_Current). Returns { matches:[{ matchedAddress, coordinates:{x,y}, tigerLineId, addressComponents, geographies:{ state, county, congressionalDistrict, censusTract, censusBlock, place, cbsaOrCsa, stateLegislativeUpper, stateLegislativeLower } }], matchCount, vintageResolved }. Each geography = { layerKey (raw vintage-versioned key), geoid (STRING — leading zeros survive, e.g. '0102'), name }. HONESTY: genuine empty (addressMatches:[]) → matchCount:0/complete:true (NOT an error; verify spelling + add city/state/ZIP). MULTIPLE matches are ALL surfaced (each with its own geographies) + a note. A historical vintage can return >1 layer per type with DISTINCT GEOIDs → BOTH surfaced (chosen + alternates[]) + a mandatory note (NEVER silently dropped). The resolved benchmark/vintage is echoed + a 'Current is a MOVING vintage' note. Invalid/missing benchmark/vintage → HTTP 400 THROWS (never fake-empty); outage/5xx THROWS. MANDATORY CAVEAT every response: these are a NOMINAL input, NOT an authoritative HUBZone / Opportunity-Zone / set-aside determination — feed censusTract.geoid / county.geoid to SBA's HUBZone map / Treasury's OZ-tract list.",
|
|
5345
5349
|
inputSchema: CensusGeocodeAddressInput,
|
|
5346
5350
|
handler: (input) => census.geocodeAddress(input),
|
|
5347
5351
|
}),
|
|
@@ -5360,7 +5364,7 @@ export const TOOLS = [
|
|
|
5360
5364
|
// a negative number / never 0). The 2D-array body is parsed by header name.
|
|
5361
5365
|
defineTool({
|
|
5362
5366
|
name: "census_business_patterns",
|
|
5363
|
-
description: "Market sizing by NAICS × geography — establishments, employment, and annual payroll from the US Census County Business Patterns (CBP) API (api.census.gov/data/{year}/cbp). ★REQUIRES a free CENSUS_API_KEY: the Census Data API has NO keyless tier, so without the key this tool THROWS an honest config error (get one at https://api.census.gov/data/key_signup.html; call api_key_status to
|
|
5367
|
+
description: "Market sizing by NAICS × geography — establishments, employment, and annual payroll from the US Census County Business Patterns (CBP) API (api.census.gov/data/{year}/cbp). ★REQUIRES a free CENSUS_API_KEY: the Census Data API has NO keyless tier, so without the key this tool THROWS an honest config error (get one at https://api.census.gov/data/key_signup.html; call api_key_status to check). Input: optional `naics` (2–6 digit NAICS-2017, e.g. '5415'; omit to aggregate all sectors), `geography` (us|state|county, default us; county REQUIRES `state`), `state` (2-digit FIPS, e.g. '06'), `year` (default '2023'), optional `limit` (client-side top-N). Returns { rows:[{ name, geoId, naicsCode, naicsLabel, establishments, employees, annualPayrollUsd, state }] } + honest _meta. HONESTY: establishments/employees are integer counts and annualPayrollUsd is annual US dollars (×1000 from source's $1,000-unit PAYANN); large-negative suppression sentinels map to null — NEVER a negative number and NEVER 0 (genuine 0 stays 0; CBP primarily uses noise-infusion + suppression flags, surfaced as reported); geoId/naicsCode/state are STRINGS (leading zeros survive). CBP returns the COMPLETE geography set (no pagination) → totalAvailable = row count, complete:true. Missing/invalid key → invalid_input (302 to Missing-Key page); header-only body → honest empty; 5xx → THROWS; 200 non-JSON → schema_drift. Key rides ONLY in the &key= query param.",
|
|
5364
5368
|
inputSchema: CensusBusinessPatternsInput,
|
|
5365
5369
|
handler: (input) => censusEconomic.businessPatterns(input),
|
|
5366
5370
|
}),
|
|
@@ -5384,7 +5388,7 @@ export const TOOLS = [
|
|
|
5384
5388
|
// SPECIFIC ANNUAL VINTAGE (surfaced in a _meta note; update yearly).
|
|
5385
5389
|
defineTool({
|
|
5386
5390
|
name: "cms_medicare_provider_services",
|
|
5387
|
-
description: "
|
|
5391
|
+
description: "Medicare Part-B provider utilization — HCPCS services rendered, beneficiaries served, and submitted / Medicare-allowed / Medicare-paid amounts (CMS 'Medicare Physician & Other Practitioners — by Provider and Service', keyless; data.cms.gov data-API). Input: `npi` (10-digit) OR `state` (2-letter) — at least ONE is REQUIRED (the table is 9.78M rows; an all-empty query is refused; providerType/hcpcsCode alone are NOT enough to scope). Optional `providerType` (exact CMS specialty, e.g. 'Family Practice'), `hcpcsCode` (e.g. '97110'), `size` (1–100, def 25), `offset`. Returns { services:[{ npi, providerName, credentials, providerType, city, state, zip, hcpcsCode, hcpcsDescription, totalBeneficiaries, totalServices, avgSubmittedCharge, avgMedicareAllowed, avgMedicarePayment }] } + honest _meta. ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (…/data-viewer/stats → found_rows); if that count fails, totalAvailable is null + a disclosing note (never length-faked). hasMore = offset+returned < total. Aggregate/payment values: numeric-string → number|null (genuine 0 stays 0, absent → null, never 0-faked); NPI/HCPCS/names are null-never-empty-string. Genuine no-match → honest empty; 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array/non-JSON → schema_drift. PUBLIC PROVIDER-LEVEL AGGREGATE figures (no patient identifiers) for ONE annual vintage (dataset year disclosed in _meta) — utilization snapshot, NOT a fraud/quality/fitness determination.",
|
|
5388
5392
|
inputSchema: CmsMedicareProviderServicesInput,
|
|
5389
5393
|
handler: (input) => cmsUtilization.providerServices(input),
|
|
5390
5394
|
}),
|
|
@@ -5397,7 +5401,7 @@ export const TOOLS = [
|
|
|
5397
5401
|
// AND-combined server-side. REQUIRE state OR facilityName (never scanned unscoped).
|
|
5398
5402
|
defineTool({
|
|
5399
5403
|
name: "cms_hospital_compare",
|
|
5400
|
-
description: "Look up Medicare-certified hospitals by
|
|
5404
|
+
description: "Look up Medicare-certified hospitals by state and/or facility-name fragment — location, type, ownership, emergency-services flag, and CMS star rating (CMS Hospital Compare 'Hospital General Information', keyless; data.cms.gov provider-data datastore-query API, ~5,432 hospitals). Input: `state` (2-letter, EXACT) OR `facilityName` (case-insensitive substring) — at least ONE is REQUIRED (all-empty query refused; hospitalType alone is NOT enough to scope); optional `hospitalType` (substring, e.g. 'Acute', 'Critical Access'), `size` (1–100, default 25), `offset`. Returns { hospitals:[{ facilityId, facilityName, address, city, state, zip, county, phone, hospitalType, ownership, emergencyServices, overallRating }] } + honest _meta. ★HONESTY: totalAvailable is the response's EXACT top-level `count` for the filter set, NEVER the returned-rows length. overallRating is CMS's 1–5 star rating; 'Not Available'/blank/non-numeric → null (NEVER 0). emergencyServices normalizes 'Yes'→true / 'No'→false / else null. IDs/names/addresses are null-never-empty-string. Genuine no-match → honest empty; 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array or missing count/results → schema_drift. Filters applied SERVER-SIDE (AND-combined). Summary star rating, NOT a clinical-quality or fitness determination.",
|
|
5401
5405
|
inputSchema: CmsHospitalCompareInput,
|
|
5402
5406
|
handler: (input) => cmsHospital.hospitalCompare(input),
|
|
5403
5407
|
}),
|
|
@@ -5410,7 +5414,7 @@ export const TOOLS = [
|
|
|
5410
5414
|
// per dataset → coalesced (null if none — never empty-string, never fabricated).
|
|
5411
5415
|
defineTool({
|
|
5412
5416
|
name: "cms_facility_directory",
|
|
5413
|
-
description: "
|
|
5417
|
+
description: "Medicare/Medicaid-certified healthcare facilities by type — nursing homes, home health agencies, hospices, or dialysis facilities — with name, address, city, state, zip, and ownership (CMS provider-data, keyless; data.cms.gov datastore-query API, four datasets). Input: `facilityType` (REQUIRED enum — 'nursing_home' ~14,695 / 'home_health' ~12,460 / 'hospice' ~6,852 / 'dialysis' ~7,490; selects the dataset id via a constant map, the value NEVER enters the URL path). Optional `state` (2-letter, EXACT), `facilityName` (case-insensitive substring), `size` (1–100, def 25), `offset`. Returns { facilities:[{ name, address, city, state, zip, facilityType, ownership }] } + honest _meta. ★HONESTY: totalAvailable is the response's EXACT top-level `count` for the filter set — NEVER the returned-rows length; hasMore = offset+returned < count. name/address/ownership column names DIFFER across the four datasets → each is COALESCED over per-dataset candidates (name: provider_name/facility_name/legal_business_name; address: address/provider_address/address_line_1; ownership: ownership_type/type_of_ownership/profit_or_nonprofit) — a field absent in the chosen dataset is null (NEVER an empty string, NEVER fabricated). facilityType is echoed on each row. Filters applied SERVER-SIDE (AND-combined) — nothing silently dropped. Genuine no-match → honest empty; invalid facilityType → invalid_input (enum-blocked); 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array or missing count/results → schema_drift. NOT a clinical-quality or fitness determination.",
|
|
5414
5418
|
inputSchema: CmsFacilityDirectoryInput,
|
|
5415
5419
|
handler: (input) => cmsFacility.facilityDirectory(input),
|
|
5416
5420
|
}),
|
|
@@ -5422,7 +5426,7 @@ export const TOOLS = [
|
|
|
5422
5426
|
// scanned unscoped). The dataset UUID is a SPECIFIC ANNUAL VINTAGE (update yearly).
|
|
5423
5427
|
defineTool({
|
|
5424
5428
|
name: "cms_dmepos_suppliers",
|
|
5425
|
-
description: "Look up Medicare DMEPOS (Durable Medical Equipment
|
|
5429
|
+
description: "Look up Medicare DMEPOS (Durable Medical Equipment) SUPPLIERS — supplier identity plus aggregate Medicare figures: HCPCS codes billed, beneficiaries served, claims, services, submitted / Medicare-allowed / Medicare-paid amounts (CMS 'Medicare DMEPOS — by Supplier', keyless; data.cms.gov data-API). The supply-side complement to cms_medicare_provider_services. Input: `npi` (10-digit) OR `state` (2-letter) — at least ONE is REQUIRED (all-empty query refused); optional `size` (1–100, default 25), `offset`. Returns { suppliers:[{ npi, supplierName, credentials, entityType, city, state, zip, totalHcpcsCodes, totalBeneficiaries, totalClaims, totalServices, submittedCharges, medicareAllowed, medicarePayment }] } + honest _meta. ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (…/data-viewer/stats → found_rows), NEVER the returned-rows length; if that count fails, totalAvailable is null + a disclosing note (never length-faked). Aggregate/payment values are numeric-string → number|null (genuine 0 stays 0, absent → null); NPI/entityType/names are null-never-empty-string; supplierName coalesces Last_Name_Org + First_Name. Genuine no-match → honest empty; 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array/non-JSON → schema_drift. PUBLIC SUPPLIER-LEVEL AGGREGATE figures (no patient identifiers) for ONE annual vintage — NOT a fraud/quality/fitness determination.",
|
|
5426
5430
|
inputSchema: CmsDmeposSuppliersInput,
|
|
5427
5431
|
handler: (input) => cmsSupplier.dmeposSuppliers(input),
|
|
5428
5432
|
}),
|
|
@@ -5434,7 +5438,7 @@ export const TOOLS = [
|
|
|
5434
5438
|
// P1 pattern; filter VALUES ride via URLSearchParams (bracket key + value encoded).
|
|
5435
5439
|
defineTool({
|
|
5436
5440
|
name: "cms_revoked_providers",
|
|
5437
|
-
description: "Search CMS's PUBLIC 'Revoked Medicare Providers & Suppliers' list — the legally-published register of Medicare enrollment revocations, with the revoked provider
|
|
5441
|
+
description: "Search CMS's PUBLIC 'Revoked Medicare Providers & Suppliers' list — the legally-published register of Medicare enrollment revocations, with the revoked provider's identity, provider type, revocation reason, effective date, and re-enrollment-bar expiration (CMS 'Revoked Providers and Suppliers', keyless; data.cms.gov data-API, ~7,059 rows). A vetting lane in the same class as OFAC / SAM-exclusions lists. Input (ALL optional — the ~7K-row list is safe to page unfiltered): `npi` (10-digit), `state` (2-letter, EXACT), `lastName` (→ LAST_NAME, exact), `size` (1–100, default 25), `offset`. Returns { revocations:[{ enrollmentId, npi, name, state, providerType, revocationReason, revocationEffectiveDate, reenrollmentBarExpiration }] } + honest _meta (which notes this is CMS's public revocation list — a due-diligence signal, NOT a current-eligibility, guilt, or fitness determination). ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (found_rows), NEVER the returned-rows length (null + note if count fails). name coalesces ORG_NAME else FIRST_NAME + LAST_NAME; NPI/reasons/dates null-never-empty. Genuine no-match → honest empty; 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array/non-JSON → schema_drift. KEYLESS.",
|
|
5438
5442
|
inputSchema: CmsRevokedProvidersInput,
|
|
5439
5443
|
handler: (input) => cmsSupplier.revokedProviders(input),
|
|
5440
5444
|
}),
|
|
@@ -5464,7 +5468,7 @@ export const TOOLS = [
|
|
|
5464
5468
|
// empty, never a throw. The optional key rides &api_key= ONLY.
|
|
5465
5469
|
defineTool({
|
|
5466
5470
|
name: "openfda_enforcement",
|
|
5467
|
-
description: "Search openFDA recall/enforcement records — drug/device/food product recalls with the recalling firm, product, reason, FDA classification (Class I/II/III), status, and geography (
|
|
5471
|
+
description: "Search openFDA recall/enforcement records — drug/device/food product recalls with the recalling firm, product, reason, FDA classification (Class I/II/III), status, and geography (api.fda.gov/{category}/enforcement.json). KEYLESS — an OPTIONAL free OPENFDA_API_KEY only raises the rate limit; keyless works at ~1000 requests/day and NEVER throws for a missing key. Input: `category` (drug|device|food, default drug), structured filters — `firm` (→recalling_firm), `product` (→product_description), `reason` (→reason_for_recall), `classification` (Class I|II|III), `status` (e.g. Ongoing/Terminated), `state` (2-letter) — safely assembled + escaped into openFDA search= Lucene string (NO raw passthrough), plus `limit` (1..100, def 25) and `skip`. Returns { recalls:[{ recallingFirm, productDescription, reasonForRecall, classification, status, state, city, recallInitiationDate, recallNumber, voluntaryMandated, distributionPattern }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination via hasMore/nextOffset — never results.length). Every scalar (recall_initiation_date is a YYYYMMDD string) is null-never-empty-string. ★A no-match query returns openFDA HTTP 404 NOT_FOUND → HONEST EMPTY (returned:0, totalAvailable:0), NOT an error; 400 → invalid_input surfacing openFDA's message; 5xx → THROWS; 200 non-JSON → schema_drift. Optional key rides ONLY the &api_key= query param.",
|
|
5468
5472
|
inputSchema: OpenfdaEnforcementInput,
|
|
5469
5473
|
handler: (input) => openfda.enforcement(input),
|
|
5470
5474
|
}),
|
|
@@ -5478,13 +5482,13 @@ export const TOOLS = [
|
|
|
5478
5482
|
// empty, never a throw. The optional key rides &api_key= ONLY.
|
|
5479
5483
|
defineTool({
|
|
5480
5484
|
name: "openfda_device_clearances",
|
|
5481
|
-
description: "Search openFDA 510(k) DEVICE CLEARANCES —
|
|
5485
|
+
description: "Search openFDA 510(k) DEVICE CLEARANCES — FDA premarket-notification clearances for medical devices, with the applicant/manufacturer, device name, clearance number (K-number), decision (date + description), clearance type, product code, advisory committee, and geography (openFDA /device/510k.json; api.fda.gov). KEYLESS (optional free OPENFDA_API_KEY only raises the rate limit — keyless works at ~1000 requests/day; NEVER throws for a missing key). Input: STRUCTURED filters — `applicant`, `deviceName`, `productCode`, `clearanceType` (e.g. Traditional/Special/Abbreviated), `kNumber` (e.g. 'K123456'), `state` (2-letter) — safely escaped into the openFDA search= Lucene string (NO raw passthrough), plus `limit` (1..100, default 25) and `skip` (offset ≥0). Returns { clearances:[{ applicant, deviceName, kNumber, decisionDate, decisionDescription, clearanceType, productCode, advisoryCommittee, state }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination via hasMore/nextOffset — never results.length); every scalar is null-never-empty-string; decision_date is a YYYY-MM-DD string. ★A no-match query returns openFDA HTTP 404 → HONEST EMPTY (returned:0/total:0), NOT an error; 400 → invalid_input surfacing openFDA's message; 5xx → THROWS; 200 non-JSON → schema_drift. Optional key rides ONLY in &api_key= param.",
|
|
5482
5486
|
inputSchema: OpenfdaDeviceClearancesInput,
|
|
5483
5487
|
handler: (input) => openfdaDevice.deviceClearances(input),
|
|
5484
5488
|
}),
|
|
5485
5489
|
defineTool({
|
|
5486
5490
|
name: "openfda_drug_approvals",
|
|
5487
|
-
description: "Search openFDA Drugs@FDA DRUG APPROVALS — FDA-approved drug applications (NDA/ANDA/BLA) with the sponsor, application number,
|
|
5491
|
+
description: "Search openFDA Drugs@FDA DRUG APPROVALS — FDA-approved drug applications (NDA/ANDA/BLA) with the sponsor, application number, approved products (brand + generic/active-ingredient name, dosage form, route, marketing status), and submission/approval history (openFDA /drug/drugsfda.json; api.fda.gov). KEYLESS (optional free OPENFDA_API_KEY only raises the rate limit — ~1000 req/day keyless; NEVER throws for a missing key). Input: STRUCTURED filters — `sponsorName`, `brandName`, `activeIngredient`, `applicationNumber` — safely escaped into the openFDA search= Lucene string (NO raw passthrough), plus `limit` (1..100, default 25) and `skip` (offset ≥0). Returns { applications:[{ applicationNumber, sponsorName, products:[{ brandName, genericIngredients:[{name,strength}], dosageForm, route, marketingStatus }], submissions:[{ submissionType, submissionNumber, submissionStatus, submissionStatusDate, submissionClass }] }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination — never results.length); every scalar is null-never-empty-string; a 'Discontinued' marketingStatus is NOT an approval revocation (disclosed in _meta). ★A no-match query returns openFDA HTTP 404 → HONEST EMPTY (returned:0/total:0), NOT an error; 400 → invalid_input; 5xx → THROWS; 200 non-JSON → schema_drift. Optional key rides ONLY in &api_key= param.",
|
|
5488
5492
|
inputSchema: OpenfdaDrugApprovalsInput,
|
|
5489
5493
|
handler: (input) => openfdaDrugsfda.drugApprovals(input),
|
|
5490
5494
|
}),
|
|
@@ -5515,7 +5519,7 @@ export const TOOLS = [
|
|
|
5515
5519
|
// (disclosed) rather than silently fetch the entire dataset.
|
|
5516
5520
|
defineTool({
|
|
5517
5521
|
name: "cpsc_recalls",
|
|
5518
|
-
description: "Look up U.S. CPSC consumer-product RECALLS —
|
|
5522
|
+
description: "Look up U.S. CPSC consumer-product RECALLS — recall title, hazard description, remedy, affected products, manufacturers, retailers, injuries, and country of manufacture (CPSC SaferProducts /RestWebServices/Recall; www.saferproducts.gov). KEYLESS. Siblings: nhtsa_recalls (vehicles), openfda_enforcement. Inputs (ALL optional): `dateStart`/`dateEnd` (YYYY-MM-DD recall date range), `productName` (substring), `manufacturer` (substring), `recallNumber`. Returns { recalls:[{ recallNumber, recallDate, title, description, url, products:[names], numberOfUnits, manufacturers:[names], retailers:[names], hazards:[descriptions], remedies:[descriptions], injuries:[names], manufacturerCountries:[names] }] } + honest _meta. HONESTY: the CPSC response is a bare array with NO count field and NO pagination — it returns the COMPLETE matching set, so totalAvailable = number of returned recalls and complete:true (never a fabricated total). ★With NO filter given, results are bounded to a DEFAULT ~90-day recent window (RecallDateStart, disclosed in _meta.notes) rather than a silent whole-dataset fetch. Empty result → HONEST EMPTY (returned:0), NOT an error; 4xx → invalid_input; 5xx/timeout → THROWS; 200 non-JSON or non-array → schema_drift. Nested arrays are flattened to name/description strings; NumberOfUnits kept as a string; every scalar is null-never-empty-string. Fixed host (SSRF-guarded); dates are ^\\d{4}-\\d{2}-\\d{2}$ and recallNumber is letters/digits/hyphen only.",
|
|
5519
5523
|
inputSchema: CpscRecallsInput,
|
|
5520
5524
|
handler: (input) => cpsc.recalls(input),
|
|
5521
5525
|
}),
|
|
@@ -5536,7 +5540,7 @@ export const TOOLS = [
|
|
|
5536
5540
|
// DataValue is a comma-formatted string; suppression codes ((NA)/(D)/(NM)/(L)/*) → null.
|
|
5537
5541
|
defineTool({
|
|
5538
5542
|
name: "bea_regional_data",
|
|
5539
|
-
description: "Regional (county / state / MSA)
|
|
5543
|
+
description: "Regional (county / state / MSA) GDP by industry and personal income from the BEA Regional Economic Accounts (apps.bea.gov/api/data, dataset 'Regional'). ★REQUIRES a free BEA_API_KEY — NO keyless tier; without the key this tool THROWS an honest config error (get one at https://apps.bea.gov/API/signup/; call api_key_status to check). Input: `tableName` (required, e.g. 'CAGDP2' county GDP, 'SAGDP2N' state GDP, 'CAINC1'/'SAINC1' personal income), `geoFips` (required — 'STATE', county FIPS like '06075', or MSA code), `lineCode` (required — integer industry line or 'ALL'), optional `year` ('LAST5' default, 4-digit year, or 'ALL'), `frequency` ('A'/'Q'). Returns { rows:[{ geoFips, geoName, timePeriod, lineCode, dataValue, unitOfMeasure, unitMult, noteRef }], notes:[{ noteRef, noteText }] } + honest _meta. ★HONESTY: a missing/invalid key OR ANY bad parameter returns HTTP 200 carrying an Error object — detected and surfaced as invalid_input carrying BEA's APIErrorDescription, NEVER a fake empty. dataValue parsed from BEA's comma-formatted string ('1,234,567' → 1234567). BEA suppression codes (NA)/(D)/(NM)/(L)/* → null (NEVER 0; genuine 0 stays 0). unitMult and unitOfMeasure reported ALONGSIDE raw dataValue — NOT pre-multiplied in. BEA returns the COMPLETE filter result (no pagination) → complete:true. Genuine empty Data:[] → honest empty; 5xx → THROWS; 200 non-JSON → schema_drift. Key rides ONLY in the UserID= query param.",
|
|
5540
5544
|
inputSchema: BeaRegionalDataInput,
|
|
5541
5545
|
handler: (input) => bea.regionalData(input),
|
|
5542
5546
|
}),
|
|
@@ -5549,7 +5553,7 @@ export const TOOLS = [
|
|
|
5549
5553
|
// never-0; standardRate/isOconus are STRING booleans coerced to real booleans.
|
|
5550
5554
|
defineTool({
|
|
5551
5555
|
name: "gsa_perdiem_rates",
|
|
5552
|
-
description: "Look up GSA Federal Travel PER-DIEM rates — the max lodging + Meals & Incidental Expenses (M&IE) reimbursement ceilings for official U.S. government travel (api.gsa.gov /travel/perdiem/v2, keyed — DATA_GOV_API_KEY or the shared DEMO_KEY). Input: EITHER `city`
|
|
5556
|
+
description: "Look up GSA Federal Travel PER-DIEM rates — the max lodging + Meals & Incidental Expenses (M&IE) reimbursement ceilings for official U.S. government travel (api.gsa.gov /travel/perdiem/v2, keyed — DATA_GOV_API_KEY or the shared DEMO_KEY). Input: EITHER `city` + `state` (2-letter) OR `zip` (5-digit) — supplying BOTH or NEITHER → invalid_input with 0 fetch; optional `year` (default: current federal fiscal year). Returns { rates:[{ city, county, state, zip, year, isOconus, standardRate, mealsUsd, monthlyLodgingUsd:[{ month (1-12), monthName, lodgingUsd }] }] } + honest _meta. HONESTY: lodgingUsd is the MAX nightly lodging ceiling for that month — VARIES SEASONALLY (hence a per-month array); mealsUsd is the daily M&IE ceiling; both are integer US dollars, null-when-withheld (NEVER 0 — genuine 0 preserved). standardRate/isOconus are booleans coerced from the API's string 'true'/'false' (unrecognized → null, never fabricated false); months array preserved AS-IS (never padded to 12). API returns COMPLETE rate set (no pagination) → totalAvailable = row count, complete:true. Genuine no-match → honest empty; `errors` field non-null → invalid_input; 429 (DEMO_KEY ~10 req/hr) → rate_limited THROWS; set DATA_GOV_API_KEY (free, api.data.gov/signup) for 1000/hr. 5xx/timeout → upstream_unavailable THROWS; 200 non-JSON → schema_drift. Key rides ONLY in the X-Api-Key header.",
|
|
5553
5557
|
inputSchema: GsaPerdiemRatesInput,
|
|
5554
5558
|
handler: (input) => gsaPerdiem.perdiemRates(input),
|
|
5555
5559
|
}),
|
|
@@ -5568,7 +5572,7 @@ export const TOOLS = [
|
|
|
5568
5572
|
}),
|
|
5569
5573
|
defineTool({
|
|
5570
5574
|
name: "dol_get_dataset",
|
|
5571
|
-
description: "Fetch records from ONE US DOL dataset (apiprod.dol.gov /v4/get/{agency}/{endpoint}/json). ★REQUIRES a free DOL_API_KEY: the DOL DATA endpoint has NO keyless tier
|
|
5575
|
+
description: "Fetch records from ONE US DOL dataset (apiprod.dol.gov /v4/get/{agency}/{endpoint}/json). ★REQUIRES a free DOL_API_KEY: the DOL DATA endpoint has NO keyless tier — without the key this tool THROWS an honest config error (get one at https://dataportal.dol.gov/registration; dol_list_datasets stays keyless). Input: `agency` (required — the `agencyAbbr` from dol_list_datasets, e.g. 'WHD'/'OSHA'/'ILAB'; rides the PATH, ^[A-Za-z0-9_]+$), `table` (required — the dataset's `apiUrl` endpoint from dol_list_datasets; rides the PATH, ^[A-Za-z0-9_]+$), optional `limit` (def 10, max 100), `offset`, `filterField`+`filterValue` (paired equality filter), `fields` (best-effort column selection). Returns { records:[…verbatim dataset rows…] } + honest _meta. HONESTY: records are surfaced VERBATIM (field names/values preserved as-is — genuine 0 stays 0, missing field stays null; never coerced or fabricated). totalAvailable is a real count ONLY when the response carries one, else null (honest unknown — `returned` is NEVER passed off as the total). A full page → hasMore; page forward to confirm. Missing/invalid key (401/403) → invalid_input carrying DOL_API_KEY guidance (never empty); 400 → invalid_input; genuine empty → honest empty; 429 → rate_limited THROWS (Retry-After honored); 5xx/timeout → upstream_unavailable THROWS; 200 non-JSON / no row array → schema_drift. Key rides ONLY in the X-API-KEY request header — never URL/_meta.",
|
|
5572
5576
|
inputSchema: DolGetDatasetInput,
|
|
5573
5577
|
handler: (input) => dol.getDataset(input),
|
|
5574
5578
|
}),
|
|
@@ -5581,7 +5585,7 @@ export const TOOLS = [
|
|
|
5581
5585
|
// page-based pagination. income/expenses are null-or-decimal-string ⇒ null-never-0.
|
|
5582
5586
|
defineTool({
|
|
5583
5587
|
name: "lda_search_filings",
|
|
5584
|
-
description: "Search US Senate LDA (Lobbying Disclosure Act) filings — who is paid
|
|
5588
|
+
description: "Search US Senate LDA (Lobbying Disclosure Act) filings — who is paid how much to lobby which federal agency on which issue (lda.senate.gov/api/v1/filings, KEYLESS — anonymous access works; optional free LDA_API_KEY only raises the rate limit). Filters (all optional): `registrantName` (the lobbying firm/in-house filer), `clientName`, `lobbyistName`, `filingYear` (4-digit), `filingType` (e.g. 'Q1'/'RR'/'YE'), `agency` (NOTE: /filings/ has NO server-side agency filter — the LDA API silently ignores it, so it is reported in _meta.filtersDropped and NOT applied; government entities are nested per activity in lobbyingActivities[].governmentEntities), `issue`, `page` (1-based), `pageSize` (1..25). Returns { filings:[{ filingUuid, filingType, filingYear, filingPeriod, incomeUsd, expensesUsd, registrant, client, lobbyingActivities:[{issueCode, description, governmentEntities:[names]}], documentUrl, postedDate, terminationDate }] } + honest _meta. HONESTY: totalAvailable is the API's REAL total match count (corpus ~1.95M filings) — NOT the rows on this page; pagination is page-based. incomeUsd/expensesUsd parsed from null-or-decimal-string — null (not reported) → null, NEVER 0 (genuine 0 stays 0); a filing reports EITHER income OR expenses, so the other is typically null. Missing lobbying_activities/government_entities → empty arrays. Genuine no-match → honest empty; 400 → invalid_input; 429 → rate_limited THROWS (Retry-After honored); 5xx/timeout THROWS; 200 non-JSON/non-array results/non-number count → schema_drift. Token rides ONLY in the Authorization: Token header.",
|
|
5585
5589
|
inputSchema: LdaSearchFilingsInput,
|
|
5586
5590
|
handler: (input) => lda.searchFilings(input),
|
|
5587
5591
|
}),
|
|
@@ -5595,7 +5599,7 @@ export const TOOLS = [
|
|
|
5595
5599
|
// (nextCursor extracted from `next`, host re-asserted). type=o FIXED.
|
|
5596
5600
|
defineTool({
|
|
5597
5601
|
name: "courtlistener_search_opinions",
|
|
5598
|
-
description: "Search US
|
|
5602
|
+
description: "Search US federal court opinions via CourtListener (www.courtlistener.com/api/rest/v4/search, type=o). ★PROVENANCE: DATA is US federal court PUBLIC RECORDS; the API is CourtListener (Free Law Project, NON-PROFIT) — NOT a .gov API; the .gov primary source (PACER) is PAYWALLED. KEYLESS (optional free COURTLISTENER_API_TOKEN only raises the rate limit). Filters (all optional): `query` (full-text → q), `court` (^[a-z0-9]+$ — e.g. 'uscfc' US Court of Federal Claims, 'cafc' Federal Circuit, 'scotus'), `dateFiledAfter`/`dateFiledBefore` (ISO YYYY-MM-DD), `natureOfSuit` (folded into q — no verified dedicated filter, disclosed in notes), `cursor` (opaque continuation), `order` (default 'dateFiled desc'). Returns { opinions:[{ caseName, court, courtId, dateFiled, docketNumber, natureOfSuit, status, judge, citation, absoluteUrl }] } + honest _meta. HONESTY: totalAvailable is the API's REAL `count` (total match count) — NOT rows on this page. Pagination is OPAQUE CURSOR (pass _meta.nextCursor back as `cursor`; nextCursor:null/hasMore:false = last page). CourtListener v4 stops counting on deep cursor pages (count:null) → totalAvailable:null DISCLOSED, never faked as results.length. dateFiled is a date STRING; citation → flattened to string/string[]; judge/natureOfSuit/docketNumber null when absent; absoluteUrl is the full CL link. Genuine no-match → honest empty; 400 → invalid_input; 429 → rate_limited (Retry-After honored); 5xx/timeout THROWS; 200 non-JSON/count not number or null → schema_drift; off-host `next` REFUSED (SSRF). Token rides ONLY in the Authorization: Token header.",
|
|
5599
5603
|
inputSchema: CourtlistenerSearchOpinionsInput,
|
|
5600
5604
|
handler: (input) => courtlistener.searchOpinions(input),
|
|
5601
5605
|
}),
|
|
@@ -5615,7 +5619,7 @@ export const TOOLS = [
|
|
|
5615
5619
|
}),
|
|
5616
5620
|
defineTool({
|
|
5617
5621
|
name: "nonprofit_financials",
|
|
5618
|
-
description: "Fetch ONE US tax-exempt nonprofit's IRS Form 990 profile + FINANCIALS by EIN via ProPublica Nonprofit Explorer (projects.propublica.org/nonprofits/api/v2/organizations/
|
|
5622
|
+
description: "Fetch ONE US tax-exempt nonprofit's IRS Form 990 profile + FINANCIALS by EIN via ProPublica Nonprofit Explorer (projects.propublica.org/nonprofits/api/v2/organizations/). ★PROVENANCE: the DATA is IRS Form 990 filings (federal tax-exempt public records) but the API is ProPublica Nonprofit Explorer (NON-PROFIT newsroom) — NOT a .gov API; ProPublica republishes these records KEYLESS (no key). Input: `ein` (required — the Employer Identification Number, 1..9 digits, e.g. '530196605' for American National Red Cross; rides the URL path). Returns { organization:{ ein, name, address, city, state, zip, nteeCode, subsectionCode, rulingDate, statusCode }, filings:[{ taxYear, formType, revenueUsd, expensesUsd, assetsUsd, liabilitiesUsd, pdfUrl }] } + honest _meta. HONESTY: the four Form 990 figures (revenueUsd/expensesUsd/assetsUsd/liabilitiesUsd) ride null-never-0 coercion — genuine reported 0 stays 0, absent → null (NEVER 0-faked); ein/codes are strings; rulingDate is a date string. totalAvailable = filings.length (COMPLETE filing set — no pagination). Unknown EIN (HTTP 404) → not_found (NEVER fabricated empty org); 4xx → invalid_input; 429 → rate_limited THROWS; 5xx/timeout → upstream_unavailable THROWS; 200 non-JSON / non-object org / non-array filings → schema_drift. Data source disclosed in _meta.source.",
|
|
5619
5623
|
inputSchema: NonprofitFinancialsInput,
|
|
5620
5624
|
handler: (input) => nonprofit.financials(input),
|
|
5621
5625
|
}),
|
|
@@ -5661,15 +5665,60 @@ async function main() {
|
|
|
5661
5665
|
},
|
|
5662
5666
|
},
|
|
5663
5667
|
});
|
|
5668
|
+
// ── Toolset resolution (MCP_SAM_GOV_TOOLSETS) ────────────────────
|
|
5669
|
+
const toolsetEnv = process.env.MCP_SAM_GOV_TOOLSETS;
|
|
5670
|
+
const allToolNames = TOOLS.map((t) => t.name);
|
|
5671
|
+
const tsResult = resolveToolsets(toolsetEnv, allToolNames);
|
|
5672
|
+
// Warn on unknown names.
|
|
5673
|
+
for (const u of tsResult.unknown) {
|
|
5674
|
+
console.error(`[mcp-sam-gov] WARNING: unknown toolset name '${u}' in MCP_SAM_GOV_TOOLSETS — valid names: ${ALL_TOOLSET_NAMES.join(", ")}. Ignoring.`);
|
|
5675
|
+
}
|
|
5676
|
+
if (tsResult.fellBack) {
|
|
5677
|
+
console.error(`[mcp-sam-gov] WARNING: no valid toolset names found in MCP_SAM_GOV_TOOLSETS='${toolsetEnv}' — falling back to all tools.`);
|
|
5678
|
+
}
|
|
5679
|
+
const loadedToolNames = tsResult.loaded;
|
|
5680
|
+
const isAllTools = tsResult.sets.length === 1 && tsResult.sets[0] === "all";
|
|
5681
|
+
// Build instructions: add a one-liner about loaded/available toolsets when
|
|
5682
|
+
// not using the default (all). The default instructions stay byte-identical.
|
|
5683
|
+
// B2: unknown toolset names are reported here (not only stderr) because
|
|
5684
|
+
// stderr is invisible in Claude Desktop.
|
|
5685
|
+
const BASE_INSTRUCTIONS = "This server wraps US government open data (keyless-first). State/local dataset IDs (Socrata, CKAN, etc.) are listed in the MCP resource samgov://data-map/state-local. If a tool result looks wrong, a tool stays broken, or the user wants a capability this server lacks, help improve it: call the `feedback` tool — or use the `report` URL present on schema_drift / upstream_unavailable errors — to get a PREFILLED GitHub issue link, and offer it to the user to open and submit. Nothing is posted automatically; the user submits. Never include secrets or personal data in a report (the repo is public).";
|
|
5686
|
+
let serverInstructions;
|
|
5687
|
+
if (isAllTools && tsResult.unknown.length === 0) {
|
|
5688
|
+
// Default: byte-identical to main.
|
|
5689
|
+
serverInstructions = BASE_INSTRUCTIONS;
|
|
5690
|
+
}
|
|
5691
|
+
else {
|
|
5692
|
+
const parts = [];
|
|
5693
|
+
if (!isAllTools) {
|
|
5694
|
+
// List loaded sets.
|
|
5695
|
+
parts.push(`Loaded toolsets: ${tsResult.sets.join(", ")}.`);
|
|
5696
|
+
// List other available sets with a 2–4 word hint each (NB5).
|
|
5697
|
+
const otherSets = ALL_TOOLSET_NAMES
|
|
5698
|
+
.filter((s) => !tsResult.sets.includes(s))
|
|
5699
|
+
.map((s) => `${s} (${TOOLSET_HINTS[s]})`);
|
|
5700
|
+
if (otherSets.length > 0) {
|
|
5701
|
+
parts.push(`Other available sets (set MCP_SAM_GOV_TOOLSETS to enable): ${otherSets.join("; ")}.`);
|
|
5702
|
+
}
|
|
5703
|
+
}
|
|
5704
|
+
// B2: report unknown names in instructions so they are visible in Claude Desktop.
|
|
5705
|
+
if (tsResult.unknown.length > 0) {
|
|
5706
|
+
parts.push(`Unknown toolset name(s) ignored: ${tsResult.unknown.join(", ")} — valid names: ${ALL_TOOLSET_NAMES.join(", ")}.`);
|
|
5707
|
+
}
|
|
5708
|
+
if (tsResult.fellBack) {
|
|
5709
|
+
parts.push("No valid toolset names found; fell back to all tools.");
|
|
5710
|
+
}
|
|
5711
|
+
serverInstructions = `${BASE_INSTRUCTIONS} ${parts.join(" ")}`;
|
|
5712
|
+
}
|
|
5664
5713
|
const server = new Server({ name: SERVER_NAME, version: SERVER_VERSION }, {
|
|
5665
|
-
capabilities: { tools: {} },
|
|
5714
|
+
capabilities: { tools: {}, resources: {} },
|
|
5666
5715
|
// Surfaced to the agent at initialize. Tells it how to route real-usage
|
|
5667
5716
|
// friction back to the project WITHOUT the server ever posting anything.
|
|
5668
|
-
instructions:
|
|
5717
|
+
instructions: serverInstructions,
|
|
5669
5718
|
});
|
|
5670
5719
|
server.setRequestHandler(ListToolsRequestSchema, async () => {
|
|
5671
5720
|
return {
|
|
5672
|
-
tools: TOOLS.map((t) => ({
|
|
5721
|
+
tools: filterToolsFor(TOOLS, loadedToolNames).map((t) => ({
|
|
5673
5722
|
name: t.name,
|
|
5674
5723
|
description: t.description,
|
|
5675
5724
|
inputSchema: zodToJsonSchema(t.inputSchema),
|
|
@@ -5677,8 +5726,56 @@ async function main() {
|
|
|
5677
5726
|
})),
|
|
5678
5727
|
};
|
|
5679
5728
|
});
|
|
5729
|
+
// ── MCP Resources: state & local data map ──────────────────────────────
|
|
5730
|
+
const DATA_MAP_URI = "samgov://data-map/state-local";
|
|
5731
|
+
const DATA_MAP_NAME = "State & local data map";
|
|
5732
|
+
const DATA_MAP_DESCRIPTION = "Jurisdiction → verified tool call → row count for every allowlisted state/local Socrata, CKAN, Tableau and Open Checkbook dataset.";
|
|
5733
|
+
server.setRequestHandler(ListResourcesRequestSchema, async () => {
|
|
5734
|
+
return {
|
|
5735
|
+
resources: [
|
|
5736
|
+
{
|
|
5737
|
+
uri: DATA_MAP_URI,
|
|
5738
|
+
name: DATA_MAP_NAME,
|
|
5739
|
+
description: DATA_MAP_DESCRIPTION,
|
|
5740
|
+
mimeType: "text/markdown",
|
|
5741
|
+
},
|
|
5742
|
+
],
|
|
5743
|
+
};
|
|
5744
|
+
});
|
|
5745
|
+
server.setRequestHandler(ReadResourceRequestSchema, async (req) => {
|
|
5746
|
+
const { uri } = req.params;
|
|
5747
|
+
if (uri !== DATA_MAP_URI) {
|
|
5748
|
+
throw new Error(`Unknown resource: ${uri}`);
|
|
5749
|
+
}
|
|
5750
|
+
return {
|
|
5751
|
+
contents: [
|
|
5752
|
+
{
|
|
5753
|
+
uri: DATA_MAP_URI,
|
|
5754
|
+
mimeType: "text/markdown",
|
|
5755
|
+
text: renderDataMapMarkdown(),
|
|
5756
|
+
},
|
|
5757
|
+
],
|
|
5758
|
+
};
|
|
5759
|
+
});
|
|
5760
|
+
// ── end MCP Resources ──────────────────────────────────────────────────
|
|
5680
5761
|
server.setRequestHandler(CallToolRequestSchema, async (req) => {
|
|
5681
5762
|
const { name, arguments: args } = req.params;
|
|
5763
|
+
// Check if the tool exists but is not loaded in the current toolset profile.
|
|
5764
|
+
const knownEntry = TOOLS.find((t) => t.name === name);
|
|
5765
|
+
if (knownEntry && !loadedToolNames.has(name)) {
|
|
5766
|
+
// B3: suggest the union of current sets + the needed set, so the user's
|
|
5767
|
+
// existing profile is not silently dropped.
|
|
5768
|
+
const envelope = toolNotLoadedEnvelope(name, tsResult.sets);
|
|
5769
|
+
return {
|
|
5770
|
+
content: [
|
|
5771
|
+
{
|
|
5772
|
+
type: "text",
|
|
5773
|
+
text: JSON.stringify(envelope, null, 2),
|
|
5774
|
+
},
|
|
5775
|
+
],
|
|
5776
|
+
isError: true,
|
|
5777
|
+
};
|
|
5778
|
+
}
|
|
5682
5779
|
try {
|
|
5683
5780
|
const raw = await runTool(name, args ?? {}, sam);
|
|
5684
5781
|
// A handler may return either its raw domain object OR a MetaBundle
|
|
@@ -5723,7 +5820,11 @@ async function main() {
|
|
|
5723
5820
|
});
|
|
5724
5821
|
const transport = new StdioServerTransport();
|
|
5725
5822
|
await server.connect(transport);
|
|
5726
|
-
|
|
5823
|
+
const loadedCount = loadedToolNames.size;
|
|
5824
|
+
const profileNote = isAllTools
|
|
5825
|
+
? `${loadedCount} tools`
|
|
5826
|
+
: `${loadedCount}/${TOOLS.length} tools, toolsets: ${tsResult.sets.join(",")}`;
|
|
5827
|
+
console.error(`[mcp-sam-gov] v${SERVER_VERSION} listening on stdio (${profileNote}).`);
|
|
5727
5828
|
// Fire-and-forget, opt-out, fail-silent update notice (STDERR only, never stdout).
|
|
5728
5829
|
// Deliberately NOT awaited: it must never delay or affect the server (update-check.ts).
|
|
5729
5830
|
void checkForUpdate(SERVER_VERSION);
|