@cliwant/mcp-sam-gov 1.13.2 → 1.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.ja.md +8 -1
- package/README.ko.md +8 -1
- package/README.md +53 -2
- package/dist/arcgis-feature.d.ts.map +1 -1
- package/dist/arcgis-feature.js +8 -0
- package/dist/arcgis-feature.js.map +1 -1
- package/dist/bls.d.ts.map +1 -1
- package/dist/bls.js +4 -3
- package/dist/bls.js.map +1 -1
- package/dist/bonfire.d.ts +1 -1
- package/dist/bonfire.d.ts.map +1 -1
- package/dist/bonfire.js +15 -11
- package/dist/bonfire.js.map +1 -1
- package/dist/data-map.d.ts +43 -0
- package/dist/data-map.d.ts.map +1 -0
- package/dist/data-map.js +289 -0
- package/dist/data-map.js.map +1 -0
- package/dist/errors.d.ts +6 -1
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js.map +1 -1
- package/dist/server.d.ts.map +1 -1
- package/dist/server.js +159 -58
- package/dist/server.js.map +1 -1
- package/dist/socrata.d.ts +1 -1
- package/dist/socrata.d.ts.map +1 -1
- package/dist/socrata.js +46 -1
- package/dist/socrata.js.map +1 -1
- package/dist/toolsets.d.ts +111 -0
- package/dist/toolsets.d.ts.map +1 -0
- package/dist/toolsets.js +390 -0
- package/dist/toolsets.js.map +1 -0
- package/package.json +1 -1
- package/src/arcgis-feature.ts +8 -0
- package/src/bls.ts +4 -3
- package/src/bonfire.ts +15 -11
- package/src/data-map.ts +329 -0
- package/src/errors.ts +6 -1
- package/src/server.ts +180 -99
- package/src/socrata.ts +47 -1
- package/src/toolsets.ts +435 -0
package/src/server.ts
CHANGED
|
@@ -21,7 +21,9 @@ import { Server } from "@modelcontextprotocol/sdk/server/index.js";
|
|
|
21
21
|
import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
|
|
22
22
|
import {
|
|
23
23
|
CallToolRequestSchema,
|
|
24
|
+
ListResourcesRequestSchema,
|
|
24
25
|
ListToolsRequestSchema,
|
|
26
|
+
ReadResourceRequestSchema,
|
|
25
27
|
} from "@modelcontextprotocol/sdk/types.js";
|
|
26
28
|
import { z } from "zod";
|
|
27
29
|
import {
|
|
@@ -96,6 +98,7 @@ import * as keys from "./keys.js";
|
|
|
96
98
|
import { toToolError, ToolErrorCarrier, errorFromResponse } from "./errors.js";
|
|
97
99
|
import * as feedback from "./feedback.js";
|
|
98
100
|
import { checkForUpdate } from "./update-check.js";
|
|
101
|
+
import { renderDataMapMarkdown } from "./data-map.js";
|
|
99
102
|
import {
|
|
100
103
|
buildMeta,
|
|
101
104
|
isMetaBundle,
|
|
@@ -104,11 +107,19 @@ import {
|
|
|
104
107
|
} from "./meta.js";
|
|
105
108
|
import { pathToFileURL, fileURLToPath } from "node:url";
|
|
106
109
|
import { realpathSync } from "node:fs";
|
|
110
|
+
import {
|
|
111
|
+
resolveToolsets,
|
|
112
|
+
TOOL_TOOLSET_MAP,
|
|
113
|
+
ALL_TOOLSET_NAMES,
|
|
114
|
+
TOOLSET_HINTS,
|
|
115
|
+
filterToolsFor,
|
|
116
|
+
toolNotLoadedEnvelope,
|
|
117
|
+
} from "./toolsets.js";
|
|
107
118
|
|
|
108
119
|
const SERVER_NAME = "mcp-sam-gov";
|
|
109
120
|
// Kept in lockstep with package.json / manifest.json / server.json.
|
|
110
121
|
// Keep in sync with package.json "version" (asserted at release; see CHANGELOG).
|
|
111
|
-
const SERVER_VERSION = "1.
|
|
122
|
+
const SERVER_VERSION = "1.15.0";
|
|
112
123
|
|
|
113
124
|
// ─── Tool input schemas (Zod) ────────────────────────────────────
|
|
114
125
|
|
|
@@ -1932,7 +1943,8 @@ const EdgarCompanyConceptInput = z.object({
|
|
|
1932
1943
|
const SocrataDomainEnum = z
|
|
1933
1944
|
.enum(socrata.SOCRATA_DOMAINS)
|
|
1934
1945
|
.describe(
|
|
1935
|
-
"Which allowlisted Socrata portal to query (
|
|
1946
|
+
"Which allowlisted Socrata portal to query (the SSRF host allowlist — no free host). " +
|
|
1947
|
+
"Jurisdiction of non-obvious hosts: cthru.data.socrata.com=MASSACHUSETTS statewide (CTHRU); atlanta.data.socrata.com=Atlanta GA; controllerdata.lacity.org+data.lacity.org=Los Angeles; www.dallasopendata.com=Dallas TX; data.brla.gov=Baton Rouge LA; data.kcmo.org=Kansas City MO; data.cstx.gov=College Station TX; data.weho.org=West Hollywood CA; opendata.usac.org+datahub.usac.org=federal USAC E-rate. data.colorado.gov's procurement data is CITY OF DENVER, not CO state.",
|
|
1936
1948
|
);
|
|
1937
1949
|
|
|
1938
1950
|
const SocrataQueryInput = z.object({
|
|
@@ -1993,7 +2005,8 @@ const SocrataDiscoverDatasetsInput = z.object({
|
|
|
1993
2005
|
.min(1)
|
|
1994
2006
|
.describe("Keyword(s) to find datasets, e.g. 'procurement', 'vendor payments', 'checkbook'."),
|
|
1995
2007
|
domain: SocrataDomainEnum.optional().describe(
|
|
1996
|
-
"Optional: scope discovery to ONE
|
|
2008
|
+
"Optional: scope discovery to ONE portal; omit to search all. The catalog does not index every host (USAC returns 0); those stay queryable via socrata_query with a known 4x4. " +
|
|
2009
|
+
"Jurisdiction of non-obvious hosts: cthru.data.socrata.com=MASSACHUSETTS statewide (CTHRU); atlanta.data.socrata.com=Atlanta GA; controllerdata.lacity.org+data.lacity.org=Los Angeles; www.dallasopendata.com=Dallas TX; data.brla.gov=Baton Rouge LA; data.kcmo.org=Kansas City MO; data.cstx.gov=College Station TX; data.weho.org=West Hollywood CA; opendata.usac.org+datahub.usac.org=federal USAC E-rate. data.colorado.gov's procurement data is CITY OF DENVER, not CO state.",
|
|
1997
2010
|
),
|
|
1998
2011
|
limit: z
|
|
1999
2012
|
.number()
|
|
@@ -3098,7 +3111,7 @@ const TableauViewCsvInput = z.object({
|
|
|
3098
3111
|
const ArcgisFeatureQueryInput = z.object({
|
|
3099
3112
|
service: z
|
|
3100
3113
|
.enum(arcgisFeature.ARCGIS_SERVICES.map((s) => s.key) as [string, ...string[]])
|
|
3101
|
-
.describe("
|
|
3114
|
+
.describe("Service key (SSRF allowlist; 29 services). DC OCP PASS (solicitations/contracts/purchase_orders/payments). US local govs: Asheville NC, Bellevue WA, Miami-Dade FL×2, Suffolk County NY, Mat-Su AK, Las Vegas NV×2, Baltimore MD, Naperville IL, Worcester MA, Topeka KS (FY2015–23), Hennepin County MN (CIP pipeline), Charlotte-Mecklenburg NC (CIP pipeline); TX/AK/IA/OK DOT bid/award registers. ND DOT flex-funding to local agencies (nddot_flex×4 — NOT vendor contracts)."),
|
|
3102
3115
|
where: z
|
|
3103
3116
|
.string()
|
|
3104
3117
|
.min(1)
|
|
@@ -3532,7 +3545,7 @@ const HtsLookupInput = z.object({
|
|
|
3532
3545
|
// SECOND POST-batch getJson-port consumer (after NIH). SSRF surface = a compile-
|
|
3533
3546
|
// time-CONSTANT host+path; seriesids ride in the module-built POST body. `series`
|
|
3534
3547
|
// is a FROZEN 9-key curated enum (the SSRF value guard + the units-label source);
|
|
3535
|
-
// `seriesId` is the raw passthrough, charclass-validated ^[A-Z0-9]{1,
|
|
3548
|
+
// `seriesId` is the raw passthrough, charclass-validated ^[A-Z0-9]{1,25}$. Years
|
|
3536
3549
|
// are bounded ints (1900..currentYear+1); the span is clamped to the tier cap
|
|
3537
3550
|
// (v1 ~10y) BEFORE the fetch + disclosed. An OPTIONAL free BLS_API_KEY rides ONLY
|
|
3538
3551
|
// in the POST body (v2, ~500/day) — never a URL/header/label/_meta/log.
|
|
@@ -3545,11 +3558,11 @@ const BlsTimeseriesInput = z.object({
|
|
|
3545
3558
|
"One or more CURATED series enum keys (typo-proof; each carries a meaning + units label): cpi_u_all (CPI-U all items NSA, index), cpi_u_core (CPI-U core NSA, index), ppi_final_demand (PPI final demand NSA, index), eci_total_comp (ECI total comp — ★12-MO % CHANGE, not an index), eci_wages (ECI wages — ★12-MO % CHANGE), unemployment_rate (SA, percent), labor_force_participation (SA, percent), employment_total_nonfarm (SA, thousands of persons), avg_hourly_earnings (SA, dollars/hour). NSA CPI-U is the escalation/EPA-clause reference. At least one of series/seriesId is required; both may be combined.",
|
|
3546
3559
|
),
|
|
3547
3560
|
seriesId: z
|
|
3548
|
-
.array(z.string().regex(/^[A-Z0-9]{1,
|
|
3561
|
+
.array(z.string().regex(/^[A-Z0-9]{1,25}$/))
|
|
3549
3562
|
.max(bls.BLS_SERIES_KEYS.length + 50)
|
|
3550
3563
|
.optional()
|
|
3551
3564
|
.describe(
|
|
3552
|
-
"One or more RAW BLS series IDs (power-user passthrough for the un-curatable space — OEWS area×occupation, local-area unemployment LAUCN…, SA/regional CPI variants). Charclass ^[A-Z0-9]{1,
|
|
3565
|
+
"One or more RAW BLS series IDs (power-user passthrough for the un-curatable space — OEWS area×occupation, local-area unemployment LAUCN…, SA/regional CPI variants). Charclass ^[A-Z0-9]{1,25}$ (uppercase alnum; punctuation/whitespace/lowercase rejected — SSRF + 'verify the ID' honesty). OEWS IDs are 25 chars (e.g. OEUN000000000000015125201). A raw ID has units:null (consult BLS). A nonexistent/typo'd ID returns BLS success + empty data (the ambiguity is disclosed, not asserted as 'no data'). At least one of series/seriesId is required.",
|
|
3553
3566
|
),
|
|
3554
3567
|
startYear: z
|
|
3555
3568
|
.number()
|
|
@@ -4573,7 +4586,7 @@ const LdaSearchFilingsInput = z.object({
|
|
|
4573
4586
|
.string()
|
|
4574
4587
|
.min(1)
|
|
4575
4588
|
.optional()
|
|
4576
|
-
.describe("NOTE:
|
|
4589
|
+
.describe("NOTE: /filings/ has NO server-side government-entity filter — the LDA API silently ignores this field (reported in _meta.filtersDropped, never as a narrowed total). Government entities are nested per activity in lobbyingActivities[].governmentEntities; narrow by registrantName/clientName/issue and inspect those nested entities. Retained for discoverability."),
|
|
4577
4590
|
issue: z
|
|
4578
4591
|
.string()
|
|
4579
4592
|
.min(1)
|
|
@@ -5684,8 +5697,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
5684
5697
|
// kev.listed to null (never false), and a kevOnly filter during an outage THROWS.
|
|
5685
5698
|
defineTool({
|
|
5686
5699
|
name: "cve_lookup",
|
|
5687
|
-
description:
|
|
5688
|
-
"Look up NIST NVD CVE records (keyless; services.nvd.nist.gov CVE API 2.0) — exact by `cveId` (CVE-YYYY-NNNN) OR search by `keyword`/`cpeName`/`cvssV3Severity`/a publication or last-modified date range — each row JOINED with its CISA KEV (Known Exploited Vulnerabilities) status. THE B2G unlock for FedRAMP/CMMC/SBOM IT-compliance: CVSS severity AND whether CISA mandates remediation by a date, in one row. Returns { results:[{ cveId, vulnStatus, rejected, published, lastModified, description, cvssMetrics:[{version,source,type,baseScore,baseSeverity,vectorString,exploitabilityScore,impactScore}], primaryCvss:{version,baseScore,baseSeverity,type}|null, cwes, references, kev }] } + honest _meta. Optional `kevOnly` (KEV-listed rows only), `resultsPerPage` (≤2000, def 50), `startIndex`. CVSS HONESTY: every metrics key matching ^cvssMetric (V2/V30/V31/V40) is surfaced as its own cvssMetrics[] element — versions are NEVER conflated and ssvcV203/non-CVSS keys are excluded; V2 baseSeverity reads from the metric level; primaryCvss is the highest-version metric, preferring type:'Primary' but FALLING BACK to the highest Secondary (a real CNA score is never dropped), null ONLY when no CVSS exists (Rejected/Awaiting) — base scores are null-never-0. KEV HONESTY: kev is {listed:true,dateAdded,dueDate,ransomware,requiredAction,catalogVersion} | {listed:false,note} | {listed:null,status:'unavailable'}; a not-listed result carries the not-in-KEV≠safe caveat (absence is NOT a clearance); if the KEV catalog cannot load, kev.listed degrades to NULL (never false) with fieldsUnavailable:['kev'], and a kevOnly filter during that outage THROWS (a KEV-membership filter is unanswerable without the catalog). PAGINATION is from NVD's EXACT totalResults, never page length. A genuine totalResults:0 is an honest found:false; a 403/429 rate breach THROWS rate_limited with the NVD_API_KEY tier disclosure; 404/5xx/timeout/off-host-redirect THROW (never a fake-empty). An OPTIONAL free NVD_API_KEY (env; https://nvd.nist.gov/developers/request-an-api-key) lifts the rate and is sent ONLY in the apiKey header — never a URL/label/_meta/log.",
|
|
5700
|
+
description: "Look up NIST NVD CVE records (keyless; services.nvd.nist.gov CVE API 2.0) — exact by `cveId` OR search by `keyword`/`cpeName`/`cvssV3Severity`/date range — each row JOINED with its CISA KEV status. Returns { results:[{ cveId, vulnStatus, rejected, published, lastModified, description, cvssMetrics:[{version,source,type,baseScore,baseSeverity,vectorString,exploitabilityScore,impactScore}], primaryCvss:{version,baseScore,baseSeverity,type}|null, cwes, references, kev }] } + honest _meta. Optional `kevOnly`, `resultsPerPage` (≤2000, def 50), `startIndex`. CVSS HONESTY: every ^cvssMetric key (V2/V30/V31/V40) is its own element — versions never conflated; V2 baseSeverity reads from metric level; primaryCvss is highest-version, type:Primary preferred but FALLS BACK to highest Secondary (real CNA score never dropped), null ONLY when no CVSS exists — base scores null-never-0. KEV HONESTY: kev is {listed:true,dateAdded,dueDate,ransomware,requiredAction,catalogVersion} | {listed:false,note} | {listed:null,status:'unavailable'}; not-listed ≠ safe (absence is NOT a clearance); if KEV catalog cannot load, kev.listed degrades to NULL (never false) with fieldsUnavailable:['kev']; a kevOnly filter during KEV outage THROWS. PAGINATION from NVD EXACT totalResults (never page length). Genuine totalResults:0 → honest found:false; 403/429 → rate_limited THROWS with NVD_API_KEY tier disclosure; 404/5xx/timeout/off-host THROW. Optional free NVD_API_KEY (env) lifts the rate — sent ONLY in the apiKey header.",
|
|
5689
5701
|
inputSchema: CveLookupInput,
|
|
5690
5702
|
handler: (input) => nvd.cveLookup(input),
|
|
5691
5703
|
}),
|
|
@@ -5698,16 +5710,14 @@ export const TOOLS: ToolDef[] = [
|
|
|
5698
5710
|
}),
|
|
5699
5711
|
defineTool({
|
|
5700
5712
|
name: "nist_800_53_controls",
|
|
5701
|
-
description:
|
|
5702
|
-
"Look up NIST SP 800-53 Rev 5 security & privacy CONTROLS (keyless) — the requirement backbone for FedRAMP / CMMC / RMF compliance work. Retrieve a control by `controlId` (exact, e.g. 'AC-2', 'SC-7', 'AC-2(1)'), a `family` (2-letter code 'AC'/'SC'/'IA' or a name substring 'Access Control'), and/or a `keyword` (case-insensitive substring over title + statement); `limit`/`offset` pagination. Each row: { id (e.g. 'AC-2'), family (e.g. 'AC — Access Control'), title, status ('withdrawn' | null), statement (the labelled requirement prose; NULL for a WITHDRAWN control, never ''), guidance (discussion), incorporatedInto:[control ids that superseded a withdrawn control, e.g. AC-13 → ['AC-2','AU-6']], enhancements:[{id,title}] (e.g. AC-2(1)) }. Complements cve_lookup + cisa_kev_lookup (the vulnerability side) with the CONTROL/requirement side. HONESTY: source is NIST's OFFICIAL OSCAL catalog published at github.com/usnistgov/oscal-content (authoritative first-party data served from GitHub, not a .gov API host — provenance disclosed in _meta); the exact OSCAL version + last-modified are surfaced in _meta (the catalog is fetched live from the MOVING 'main' branch, so control text can shift between point releases, e.g. 5.1.1 → 5.2.0 — cite the version, not just 'Rev 5'); a WITHDRAWN control (status:'withdrawn') has statement:null and is NOT an active requirement (see incorporatedInto for what replaced it); the catalog has no query API so filtering is CLIENT-SIDE and totalAvailable is the EXACT match count; this is the REQUIREMENT text only — applicability depends on the system's FIPS-199 impact baseline (Low/Moderate/High), which the catalog does not encode (disclosed); a download failure or an implausibly-truncated catalog (< 15 families) THROWS (never a fake-empty 'control not found').",
|
|
5713
|
+
description: "Look up NIST SP 800-53 Rev 5 security & privacy CONTROLS (keyless) — the requirement backbone for FedRAMP / CMMC / RMF compliance work. Complements cve_lookup + cisa_kev_lookup. Retrieve by `controlId` (exact, e.g. 'AC-2', 'SC-7', 'AC-2(1)'), `family` (2-letter 'AC'/'SC'/'IA' or name substring), and/or `keyword` (case-insensitive substring over title + statement); `limit`/`offset` pagination. Each row: { id, family, title, status ('withdrawn'|null), statement (requirement prose; NULL for a WITHDRAWN control, never ''), guidance (discussion), incorporatedInto:[ids that superseded a withdrawn control], enhancements:[{id,title}] }. HONESTY: source is NIST's OFFICIAL OSCAL catalog at github.com/usnistgov/oscal-content (authoritative first-party data served from GitHub, not a .gov API host — provenance disclosed in _meta); the exact OSCAL version + last-modified are surfaced in _meta (catalog fetched live from the MOVING 'main' branch, so control text can shift between point releases — cite the version); a WITHDRAWN control has statement:null and is NOT an active requirement (see incorporatedInto for what replaced it); filtering is CLIENT-SIDE and totalAvailable is the EXACT match count; applicability depends on the system's FIPS-199 impact baseline (Low/Moderate/High), which the catalog does not encode; a download failure or implausibly-truncated catalog (< 15 families) THROWS (never fake-empty).",
|
|
5703
5714
|
inputSchema: NistControlsInput,
|
|
5704
5715
|
handler: (input) => nistControls.searchControls(input),
|
|
5705
5716
|
}),
|
|
5706
5717
|
// ━━━ NPPES NPI Registry — Healthcare-Provider Vetting (1) ━━━ ADR-0036
|
|
5707
5718
|
defineTool({
|
|
5708
5719
|
name: "nppes_lookup_provider",
|
|
5709
|
-
description:
|
|
5710
|
-
"Keyless CMS/HHS NPPES NPI Registry lookup — the authoritative PUBLIC registry of every US healthcare provider (individual NPI-1 + organization NPI-2), for VA/HHS/CMS subcontractor/provider/teaming due-diligence (validate an NPI, confirm taxonomy/specialty, enumeration status, practice state, org/name match). Host npiregistry.cms.hhs.gov/api (version=2.1). Mode is inferred from `number` (no mode flag). EXACT-NPI mode (`number` given): the NPI is CMS-Luhn-validated client-side (Luhn over 80840+first-9) ⇒ a typo'd NPI is invalid_input, NEVER a fake 'does not exist'; ★the wire query carries `number` (+version) ALONE — any co-supplied filter (last_name/state/…) is DROPPED from the wire and checked CLIENT-SIDE (disclosed in data.filterMatch:{field:bool} + data.filtersDropped), because NPPES AND-combines a number with filters and a mismatch would falsely zero a real active provider into found:false. SEARCH mode: required-one of { first_name, last_name, organization_name, taxonomy_description, city, postal_code } (state + enumeration_type are REFINERS ONLY — rejected alone); a trailing '*' wildcard on a name/org field needs ≥2 leading literal chars. Returns EXACT-mode { found, provider:{ number, enumerationType, active, status, basic{…individual OR org fields, null-never-fabricated…}, taxonomies[{code,desc,primary,state,license,taxonomyGroup}], addresses[{purpose,address1,city,state,postalCode,telephone,fax,countryCode}], practiceLocations[…same, SEPARATE from addresses], identifiers[], otherNames[], endpoints[], createdEpoch, lastUpdatedEpoch }, filterMatch? } OR SEARCH-mode { providers:[…] } + honest _meta. HONESTY: active = basic.status==='A' (a deactivated/absent NPI is NOT active); epochs are ms numeric STRINGS → number|null (null-never-0); addresses[] and practiceLocations[] are kept SEPARATE (a provider can practice in a state that appears ONLY in practiceLocations); NPPES exposes NO match total, so a full page ⇒ totalAvailable is a disclosed LOWER BOUND (totalIsLowerBound) + a ~1,200-row-per-query reach cap (limit ≤ 200, skip ≤ 1,000 — OUR policy, a PER-QUERY cap only; cross-query enumeration is not architecturally prevented). A genuine {result_count:0} ⇒ honest found:false/empty; a {Errors:[…]} 200 body (no results key) ⇒ THROWS invalid_input (never a fake empty); any 4xx/5xx/timeout/off-host-redirect ⇒ THROWS; result_count !== results.length ⇒ schema_drift. ★NOT a fitness/exclusion/licensure/sanctions determination — cross-check SAM exclusions + OFAC; individual (NPI-1) records may surface personal/home addresses + phone/fax verbatim with NO enrichment. The caveat + reach-cap disclosure ride EVERY response.",
|
|
5720
|
+
description: "Keyless CMS/HHS NPPES NPI Registry — every US healthcare provider (NPI-1 individual + NPI-2 organization). EXACT-NPI mode (when `number` supplied): NPI is CMS-Luhn-validated — typo'd NPI is invalid_input, NEVER a fake 'does not exist'; wire carries `number`+version ALONE — co-supplied filters are DROPPED from wire and checked CLIENT-SIDE (filterMatch:{field:bool} + filtersDropped) because NPPES AND-combines number+filters and a mismatch would falsely zero a real active provider. SEARCH mode: required-one of {first_name, last_name, organization_name, taxonomy_description, city, postal_code} (state + enumeration_type are REFINERS ONLY — rejected alone); trailing '*' wildcard needs ≥2 leading literal chars. Returns EXACT-mode { found, provider:{number, enumerationType, active, basic, taxonomies, addresses, practiceLocations, identifiers, otherNames, endpoints, createdEpoch, lastUpdatedEpoch}, filterMatch? } OR SEARCH { providers:[…] } + honest _meta. HONESTY: active = basic.status==='A'; epochs are ms numeric STRINGS → number|null; addresses[] and practiceLocations[] kept SEPARATE (a provider may appear in practiceLocations ONLY); NPPES exposes NO match total — full page → totalAvailable is a LOWER BOUND (totalIsLowerBound) + reach cap (limit ≤ 200, skip ≤ 1,000). Genuine {result_count:0} → honest found:false; {Errors:[…]} 200 body THROWS; 4xx/5xx/timeout THROW; count mismatch → schema_drift. ★NOT a fitness/exclusion/licensure/sanctions determination — cross-check SAM + OFAC; NPI-1 records may surface personal/home addresses + phone/fax verbatim.",
|
|
5711
5721
|
inputSchema: NppesLookupInput,
|
|
5712
5722
|
handler: (input) => nppes.lookupProvider(input),
|
|
5713
5723
|
}),
|
|
@@ -5721,23 +5731,20 @@ export const TOOLS: ToolDef[] = [
|
|
|
5721
5731
|
}),
|
|
5722
5732
|
defineTool({
|
|
5723
5733
|
name: "cms_query_dataset",
|
|
5724
|
-
description:
|
|
5725
|
-
"Query a CMS Open Payments DKAN datastore distribution by datasetId + index (keyless; openpaymentsdata.cms.gov) — the healthcare industry-financial-relationship / COI-vetting + market-intelligence lane NPPES (provider identity) cannot answer. GET /api/1/datastore/query/{datasetId}/{index} with server-side `conditions` filters, an EXACT `count`, offset/limit pagination, and a `properties` projection. Returns { datasetId, index, results (mode), fields:[{name,type,mysqlType,description}] (from the DKAN schema), rows:[…verbatim…] } + honest _meta. A confirmed target: 2025 Research Payment Data 'f0d1de67-6852-4093-a036-c9328c256a05' index 0 (count 931959; + a recipient_state='CA' condition → 92097). ★HONESTY: `count` is the EXACT grand total (P1) → totalAvailable=count + real offset pagination (NOT a page-length lower bound); `conditions` are server-side and self-policing — a valid column narrows the count, a BAD column ⇒ HTTP 400 ⇒ invalid_input, so filtersDropped is ALWAYS empty (no silent-drop path, P4); limit ≤ 500 is the HARD API cap (a higher limit ⇒ invalid_input, no silent clamp); every column is text, so amounts (total_amount_of_payment_usdollars, …) arrive as STRINGS surfaced verbatim (a missing amount is null-never-0, P3). ★results:false = a COUNT/SCHEMA-discovery mode: no rows, pagination disabled (no livelock), but the EXACT count + every column's schema returned (count=true is ALWAYS on the wire — not a caller toggle). A genuine {count:0} ⇒ honest empty; a 400 (bad column/limit) / 404 (bad datasetId/index) / HTML (SPA/WAF) / 5xx / timeout / a missing schema anchor or non-array results (in results:true) ⇒ THROW (never a fake empty). ★SSRF: datasetId (36-char lowercase UUID) + index interpolate into the URL PATH (validated before interpolation). ★PII: Open Payments is PUBLIC transparency-BY-LAW data (in-scope per the NPPES precedent) naming physicians + amounts verbatim — bounded to targeted vetting (offset ≤ 2000 reach cap), NO enrichment, NO covered_recipient_npi→NPPES auto-join. NOT a conflict-of-interest finding / fitness / exclusion determination — cross-check SAM exclusions + OFAC + the OIG-LEIE. The caveat + reach-cap disclosure ride EVERY response.",
|
|
5734
|
+
description: "Query a CMS Open Payments DKAN datastore distribution by datasetId + index (keyless; openpaymentsdata.cms.gov). Returns { datasetId, index, results (mode), fields:[{name,type,mysqlType,description}], rows:[…verbatim…] } + honest _meta. ★HONESTY: `count` is the EXACT grand total → totalAvailable=count + real offset pagination (NOT a page-length lower bound). `conditions` are server-side self-policing — BAD column → HTTP 400 → invalid_input; filtersDropped is ALWAYS empty (no silent-drop path). limit ≤ 500 is the HARD API cap (higher → invalid_input, no silent clamp). Every column is text, amounts arrive as STRINGS verbatim (null-never-0). ★results:false = COUNT/SCHEMA-discovery mode: no rows, pagination disabled (no livelock), EXACT count + column schema returned. Genuine {count:0} → honest empty; 400/404/HTML/5xx/timeout/missing schema/non-array → THROW. ★SSRF: datasetId (36-char UUID) + index are validated before URL interpolation. ★PII: Open Payments is PUBLIC transparency-BY-LAW data — bounded to targeted vetting (offset ≤ 2000 reach cap), NO enrichment, NO covered_recipient_npi→NPPES auto-join. NOT a conflict-of-interest / fitness / exclusion determination — cross-check SAM + OFAC + OIG-LEIE. The caveat + reach-cap disclosure ride EVERY response.",
|
|
5726
5735
|
inputSchema: CmsQueryDatasetInput,
|
|
5727
5736
|
handler: (input) => cms.queryDataset(input),
|
|
5728
5737
|
}),
|
|
5729
5738
|
// ━━━ FAC Federal Audit Clearinghouse — Single Audit audit-risk vetting (2) ━━━ ADR-0038
|
|
5730
5739
|
defineTool({
|
|
5731
5740
|
name: "fac_search_audits",
|
|
5732
|
-
description:
|
|
5733
|
-
"Search entity Single Audit summaries from the Federal Audit Clearinghouse (keyless via the api.data.gov DEMO_KEY; api.fac.gov PostgREST `general` table) — the SUBCONTRACTOR / teaming AUDIT-RISK vetting entry point (2 CFR 200 Subpart F / Single Audit Act; every entity expending ≥$750K/yr in federal awards). Structured filters (all optional, AND-combined): `auditeeUei` (12-char SAM UEI — the PRIMARY join key to SAM/USAspending/EDGAR, → auditee_uei), `auditeeState` (2-letter → auditee_state), `auditYear` (int → audit_year), `totalExpendedMin`/`totalExpendedMax` (USD → total_amount_expended gte/lte). `limit` (≤100, def 25), `offset`. Returns { audits:[{ report_id, auditee_uei, audit_year, auditee_name, auditee_ein, auditee_state, auditee_city, total_amount_expended, fac_accepted_date }] } + honest _meta. Feed a row's report_id (or the UEI) to fac_get_findings for the audit-RISK flags. ★PII: a HARDCODED select-allowlist surfaces ONLY entity + audit-summary fields and DELIBERATELY EXCLUDES the auditee's personal-contact columns (email/phone/certifying-official name) — the vetting subject is the ENTITY; there is NO caller `select`/column param. HONESTY: totalAvailable is the EXACT Content-Range total (a response header under Prefer:count=exact; a '*'/absent/non-numeric denominator ⇒ totalAvailable:null + a page-fullness hedge, NEVER 0); total_amount_expended is null-never-0 (a missing amount is null, never 0); a bad column ⇒ PostgREST 400 ⇒ invalid_input (filtersDropped is ALWAYS empty); a genuine [] ⇒ honest empty; 400/403/5xx/timeout/HTML/non-array THROW (206 = success, never a fake empty). NOT a debarment/exclusion/fitness determination — an audit finding is the auditor's opinion; cross-check SAM exclusions + OFAC. Keyless-first via DEMO_KEY (~10 req/hr shared ceiling; set DATA_GOV_API_KEY for production — never logged).",
|
|
5741
|
+
description: "Search entity Single Audit summaries from the Federal Audit Clearinghouse (keyless via api.data.gov DEMO_KEY; api.fac.gov PostgREST `general` table) — the SUBCONTRACTOR / teaming AUDIT-RISK vetting entry point (2 CFR 200 Subpart F / Single Audit Act; every entity expending ≥$750K/yr in federal awards). Filters (all optional, AND-combined): `auditeeUei` (12-char SAM UEI — PRIMARY join key to SAM/USAspending/EDGAR), `auditeeState` (2-letter), `auditYear` (int), `totalExpendedMin`/`totalExpendedMax` (USD). `limit` (≤100, def 25), `offset`. Returns { audits:[{ report_id, auditee_uei, audit_year, auditee_name, auditee_ein, auditee_state, auditee_city, total_amount_expended, fac_accepted_date }] } + honest _meta. Feed report_id (or UEI) to fac_get_findings for the audit-RISK flags. ★PII: a HARDCODED select-allowlist surfaces ONLY entity + audit-summary fields and DELIBERATELY EXCLUDES personal-contact columns — NO caller `select`/column param. HONESTY: totalAvailable is the EXACT Content-Range total (a response header under Prefer:count=exact; '*'/absent/non-numeric denominator → totalAvailable:null + page-fullness hedge, NEVER 0); total_amount_expended is null-never-0; a bad column → PostgREST 400 → invalid_input (filtersDropped ALWAYS empty); genuine [] → honest empty; 400/403/5xx/timeout/HTML/non-array THROW. NOT a debarment/exclusion/fitness determination — cross-check SAM + OFAC. Keyless-first via DEMO_KEY (~10 req/hr shared; set DATA_GOV_API_KEY for production — never logged).",
|
|
5734
5742
|
inputSchema: FacSearchAuditsInput,
|
|
5735
5743
|
handler: (input) => fac.searchAudits(input),
|
|
5736
5744
|
}),
|
|
5737
5745
|
defineTool({
|
|
5738
5746
|
name: "fac_get_findings",
|
|
5739
|
-
description:
|
|
5740
|
-
"Drill into the audit-RISK findings for an entity from the Federal Audit Clearinghouse (keyless via the api.data.gov DEMO_KEY; api.fac.gov PostgREST `findings` table) — the risk-detail step after fac_search_audits. At least ONE of `auditeeUei` (12-char UEI → auditee_uei) or `reportId` (→ report_id, from a fac_search_audits row) is REQUIRED (an empty query is refused, never a whole-table scan); optional `auditYear` (int), `limit` (≤100, def 50), `offset`. Returns { findings:[{ report_id, auditee_uei, audit_year, award_reference, reference_number, is_material_weakness, is_modified_opinion, is_questioned_costs, is_repeat_finding, is_significant_deficiency, is_other_findings, is_other_matters, type_requirement, prior_finding_ref_numbers, riskFlags:{materialWeakness, modifiedOpinion, questionedCosts, repeatFinding, significantDeficiency, otherFindings, otherMatters} }] } + honest _meta. ★RISK-FLAG HONESTY: the is_* flags are surfaced VERBATIM as the auditor reported them (\"Y\"/\"N\") PLUS a typed riskFlags tri-state (\"Y\"→true / \"N\"→false / blank/absent/other → null=UNKNOWN) — a null flag is NEVER rendered as false/\"no material weakness\" (the false-CLEAR class). ★EMPTY ≠ CLEAN: an empty findings list does NOT confirm a clean audit — the entity may not have filed a Single Audit (below the $750K threshold), the audit may predate FAC coverage, or the UEI may be wrong; a disclosure note fires on any empty result telling you to confirm an ACCEPTED audit exists via fac_search_audits. ★PII: a HARDCODED select-allowlist (NO caller column param) surfaces only entity + audit-risk fields — no personal contact. totalAvailable is the EXACT Content-Range total ('*'/absent ⇒ null + hedge, never 0); a bad column ⇒ 400 ⇒ invalid_input; 400/403/5xx/timeout/HTML/non-array THROW (206 = success). NOT a debarment/determination — cross-check SAM exclusions + OFAC + the specific finding text. Keyless-first via DEMO_KEY (~10 req/hr; set DATA_GOV_API_KEY — never logged).",
|
|
5747
|
+
description: "Drill into audit-RISK findings for an entity from the Federal Audit Clearinghouse (keyless via api.data.gov DEMO_KEY; api.fac.gov PostgREST `findings` table) — the risk-detail step after fac_search_audits. At least ONE of `auditeeUei` (12-char UEI) or `reportId` is REQUIRED (empty query refused); optional `auditYear`, `limit` (≤100, def 50), `offset`. Returns { findings:[{ report_id, auditee_uei, audit_year, award_reference, reference_number, is_material_weakness, is_modified_opinion, is_questioned_costs, is_repeat_finding, is_significant_deficiency, is_other_findings, is_other_matters, type_requirement, prior_finding_ref_numbers, riskFlags:{materialWeakness, modifiedOpinion, questionedCosts, repeatFinding, significantDeficiency, otherFindings, otherMatters} }] } + honest _meta. ★RISK-FLAG HONESTY: is_* flags surfaced VERBATIM (\"Y\"/\"N\") PLUS typed riskFlags tri-state (\"Y\"→true / \"N\"→false / blank/absent → null=UNKNOWN) — null NEVER rendered as false (the false-CLEAR class). ★EMPTY ≠ CLEAN: empty findings does NOT confirm a clean audit — the entity may not have filed a Single Audit (below the $750K threshold), may predate FAC coverage, or UEI wrong; a disclosure note fires on any empty result; on empty, confirm an ACCEPTED audit via fac_search_audits. ★PII: HARDCODED select-allowlist, entity + audit-risk fields only. totalAvailable is EXACT Content-Range total ('*'/absent → null + hedge, never 0); 400/403/5xx/timeout/HTML/non-array THROW. NOT a debarment/determination — cross-check SAM + OFAC. DEMO_KEY ~10 req/hr — set DATA_GOV_API_KEY.",
|
|
5741
5748
|
inputSchema: FacGetFindingsInput,
|
|
5742
5749
|
handler: (input) => fac.getFindings(input),
|
|
5743
5750
|
}),
|
|
@@ -5816,22 +5823,19 @@ export const TOOLS: ToolDef[] = [
|
|
|
5816
5823
|
}),
|
|
5817
5824
|
defineTool({
|
|
5818
5825
|
name: "edgar_filing_index",
|
|
5819
|
-
description:
|
|
5820
|
-
"Bulk cross-filer SEC filing index for a quarter (keyless, from the www.sec.gov EDGAR full-index master.idx). Reads the WHOLE quarter's index (every filer's every filing — CIK|Company|Form|Date|Filename, ~370K rows), FULL-SCANS it, and returns offset-paginated filings matching CLIENT-SIDE filters with the EXACT total. Input `year` (>=1993, <= current year), `quarter` (1..4); optional `formType` (exact form, e.g. '8-K'), `cik` (numeric, leading-zero-safe), `companyContains` (LITERAL case-insensitive substring), `dateFrom`/`dateTo` (ISO YYYY-MM-DD), `limit` (<=1000, def 100), `offset`. Returns { year, quarter, indexFile, returned, totalAvailable, filings:[{ cik, cikPadded, companyName, formType, dateFiled, filename, filingUrl }] }. This is the BULK-ENUMERATION primitive (the per-filer edgar tools need a CIK you already hold; this sweeps a whole quarter by form/date/company, e.g. 'every 8-K in 2024 Q1'). HONESTY: totalAvailable is the EXACT match count over the full quarter scan — never a page length, never a byte-capped subset (SEC ignores HTTP Range); a 0-match result is a genuine EXACT ZERO (complete:true), NOT a truncation; a bounds-valid but unpublished quarter returns HTTP 403 and is surfaced as an AMBIGUOUS both-causes error (quarter-not-published OR the 10 req/s rate-block), never a bare rate-limit and never a fake-empty; a non-index / all-malformed body is refused as schema_drift; a future year / bad quarter is rejected pre-fetch (invalid_input, 0 fetch). The CURRENT quarter grows daily (totalAvailable is exact AS-OF-snapshot). filingUrl is a resolvable archive URL. NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — there is no authoritative CIK↔UEI join.",
|
|
5826
|
+
description: "Bulk sweep — per-filer edgar tools need a CIK. Bulk cross-filer SEC filing index for a quarter (keyless; www.sec.gov EDGAR full-index master.idx). Reads the WHOLE quarter's index (~370K rows: every filer's every filing — CIK|Company|Form|Date|Filename), full-scans it, and returns offset-paginated filings matching CLIENT-SIDE filters with the EXACT total. Input: `year` (≥1993, ≤current year), `quarter` (1..4); optional `formType` (exact, e.g. '8-K'), `cik` (numeric, leading-zero-safe), `companyContains` (LITERAL case-insensitive substring), `dateFrom`/`dateTo` (ISO YYYY-MM-DD), `limit` (≤1000, def 100), `offset`. Returns { year, quarter, indexFile, returned, totalAvailable, filings:[{ cik, cikPadded, companyName, formType, dateFiled, filename, filingUrl }] }. HONESTY: totalAvailable is the EXACT match count over the full quarter scan — never a page length, never a byte-capped subset (SEC ignores HTTP Range). A 0-match result is a genuine EXACT ZERO (complete:true), NOT a truncation. A bounds-valid but unpublished quarter returns HTTP 403 and is surfaced as an AMBIGUOUS error (quarter-not-published OR the 10 req/s rate-block) — never a bare rate-limit and never a fake-empty. A non-index or all-malformed body → schema_drift. A future year / bad quarter → invalid_input pre-fetch. The CURRENT quarter grows daily (totalAvailable is exact as-of-snapshot). filingUrl is a resolvable archive URL. NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — no authoritative CIK↔UEI join.",
|
|
5821
5827
|
inputSchema: EdgarFilingIndexInput,
|
|
5822
5828
|
handler: (input) => edgar.filingIndex(input),
|
|
5823
5829
|
}),
|
|
5824
5830
|
defineTool({
|
|
5825
5831
|
name: "edgar_daily_filing_index",
|
|
5826
|
-
description:
|
|
5827
|
-
"Per-DAY cross-filer SEC filing index (keyless, from the www.sec.gov EDGAR daily-index master.YYYYMMDD.idx). The per-day sibling of edgar_filing_index (~30× smaller): reads ONE calendar day's index (every filer's every filing that day — CIK|Company|Form|Date|File Name, ~8K rows), FULL-SCANS it, and returns offset-paginated filings matching CLIENT-SIDE filters with the EXACT total. Answers the monitoring/alerting question the quarterly tool cannot ('every 8-K filed on 2024-01-03', 'watch a CIK day-by-day'). Input `date` (required ISO YYYY-MM-DD, >=1994-01-01, not future); optional `formType` (exact form, e.g. '8-K'), `cik` (numeric, leading-zero-safe), `companyContains` (LITERAL case-insensitive substring), `limit` (<=1000, def 100), `offset`. Returns { found, date, year, quarter, indexFile, returned, totalAvailable, filings:[{ cik, cikPadded, companyName, formType, dateFiled, filename, filingUrl }] }. HONESTY: totalAvailable is the EXACT match count over the full day scan — never a page length, never a byte-capped subset (SEC ignores HTTP Range). The daily-index's pervasive-403 empty model is disambiguated via the quarter's index.json existence oracle, RECENCY-AWARE: a day NEWER than the newest published index (weekend/holiday/not-yet-disseminated recent trading day) ⇒ found:false, complete:FALSE, retryable not-yet-disseminated note (NEVER a confident empty); an unlisted day INSIDE the covered range (a real weekend/holiday) ⇒ found:false, complete:true genuine-absent; a LISTED day whose .idx 403s ⇒ honest rate_limited; the oracle itself inconclusive ⇒ ambiguous both-causes upstream_unavailable. A non-real/future date is rejected pre-fetch (invalid_input, 0 fetch); a non-index / all-malformed body is refused as schema_drift. dateFiled is normalized to ISO from the compact YYYYMMDD column. NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — there is no authoritative CIK↔UEI join.",
|
|
5832
|
+
description: "Per-day sibling of edgar_filing_index. Per-day cross-filer SEC filing index (keyless; www.sec.gov EDGAR daily-index master.YYYYMMDD.idx) — reads ONE calendar day's index (~8K rows), full-scans it, and returns offset-paginated filings matching CLIENT-SIDE filters with the EXACT total. Answers the monitoring/alerting question ('every 8-K filed on 2024-01-03'). Input: `date` (required ISO YYYY-MM-DD, ≥1994-01-01, not future); optional `formType` (exact), `cik` (numeric), `companyContains` (LITERAL case-insensitive), `limit` (≤1000, def 100), `offset`. Returns { found, date, year, quarter, indexFile, returned, totalAvailable, filings:[{ cik, cikPadded, companyName, formType, dateFiled, filename, filingUrl }] }. HONESTY: totalAvailable is EXACT match count — never a page length. The daily-index pervasive-403 model is disambiguated via the quarter's index.json existence oracle, RECENCY-AWARE: a day NEWER than the newest published index → found:false, complete:FALSE, retryable not-yet-disseminated note (NEVER a confident empty); an unlisted day INSIDE the covered range (real weekend/holiday) → found:false, complete:true; a LISTED day whose .idx 403s → honest rate_limited; oracle inconclusive → ambiguous upstream_unavailable. A non-real/future date → invalid_input pre-fetch; non-index/all-malformed body → schema_drift. dateFiled normalized from compact YYYYMMDD to ISO. NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — no authoritative CIK↔UEI join.",
|
|
5828
5833
|
inputSchema: EdgarDailyFilingIndexInput,
|
|
5829
5834
|
handler: (input) => edgar.dailyFilingIndex(input),
|
|
5830
5835
|
}),
|
|
5831
5836
|
defineTool({
|
|
5832
5837
|
name: "edgar_company_concept",
|
|
5833
|
-
description:
|
|
5834
|
-
"One filer × one XBRL concept × the COMPLETE reported time-series (keyless, from data.sec.gov companyconcept). The focused financial-TREND / entity-vetting primitive BETWEEN edgar_company_facts (many curated concepts for one filer) and edgar_xbrl_frames (one concept across ALL filers for one period) — 'track THIS filer's Assets/Revenues/NetIncomeLoss OVER TIME, and was it ever revised?'. Input `cikOrTicker` (CIK or resolvable ticker/name), `concept` (EXACT alnum XBRL tag, e.g. 'Assets'), optional `taxonomy` (us-gaap|dei|ifrs-full, def us-gaap), `unit` (CLIENT-SIDE key filter), `form`/`fy` (client-side), `canonicalOnly` (def false), `limit`/`offset`. Returns { found, cik, entityName, taxonomy, concept, label, description, unitsAvailable:[{unit,count}], rows:[{ unit, start, end, val, accn, fy, fp, form, filed, frame, canonical }] }. HONESTY: (M1) period identity is the (start,end) PAIR — every row carries `start` (null for INSTANT concepts, the ISO date for DURATION/flow concepts); the SAME `end` with a DIFFERENT `start` is a different-duration fact (a 3-month quarter vs the 12-month year), NOT a revision — a revision is only multiple rows sharing the same (start,end) with a differing accn/filed/val. DEFAULT returns ALL rows incl. the amendment/restatement history + a per-row `canonical` (frame-tagged = SEC's consolidated value); `canonicalOnly:true` dedups to one canonical row per (unit,start,end), fully disclosed, never a silent drop. Every row is unit-tagged (a USD amount is NEVER conflated with a share count); unitsAvailable discloses ALL units with their RAW counts even under a unit filter; val is null-never-0. A bad CIK/taxonomy/concept ⇒ upstream 404 ⇒ found:false (NEVER a fabricated val:0); a 5xx/timeout/non-JSON/units-shape-drift THROWS; a `unit` not present ⇒ honest empty + the available-units note (unit is CLIENT-SIDE, not a path segment). cik/taxonomy/concept are validated path segments (regex+enum, re-checked pre-fetch) — no injection surface. NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — there is no authoritative CIK↔UEI join.",
|
|
5838
|
+
description: "One filer × one XBRL concept × complete reported time-series (keyless; data.sec.gov companyconcept), including amendment/restatement history. Sits between edgar_company_facts (many concepts, one filer) and edgar_xbrl_frames (one concept, all filers). start=null for INSTANT concepts. Input: `cikOrTicker`, `concept` (EXACT alnum XBRL tag, e.g. 'Assets'), optional `taxonomy` (us-gaap|dei|ifrs-full, def us-gaap), `unit` (CLIENT-SIDE filter), `form`/`fy` (client-side), `canonicalOnly` (def false), `limit`/`offset`. Returns { found, cik, entityName, taxonomy, concept, label, description, unitsAvailable:[{unit,count}], rows:[{unit, start, end, val, accn, fy, fp, form, filed, frame, canonical}] }. HONESTY M1: period identity is the (start,end) PAIR — the SAME `end` with a DIFFERENT `start` is a different-duration fact (3-month vs 12-month), NOT a revision; a revision is multiple rows sharing the same (start,end) with differing accn/filed/val. DEFAULT returns ALL rows including restatement history + per-row `canonical`; `canonicalOnly:true` dedupes to one canonical row per (unit,start,end), fully disclosed, never a silent drop. Every row is unit-tagged; unitsAvailable discloses ALL units even under a unit filter; val is null-never-0. A bad CIK/taxonomy/concept → 404 → found:false (NEVER fabricated val:0); 5xx/timeout/non-JSON/shape-drift THROWS; `unit` not present → honest empty + available-units note (unit is CLIENT-SIDE, not a path segment). cik/taxonomy/concept are validated path segments (no injection). NOTE: EDGAR keys on CIK, NOT SAM UEI/DUNS — no authoritative CIK↔UEI join.",
|
|
5835
5839
|
inputSchema: EdgarCompanyConceptInput,
|
|
5836
5840
|
handler: (input) => edgar.companyConcept(input),
|
|
5837
5841
|
}),
|
|
@@ -5839,7 +5843,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
5839
5843
|
defineTool({
|
|
5840
5844
|
name: "socrata_query",
|
|
5841
5845
|
description:
|
|
5842
|
-
"Query rows from an allowlisted Socrata/SODA open-data portal (keyless; ~a dozen US state portals + USAC E-rate on one identical API — state spend/checkbook/contract/vendor-payment datasets). Input `domain` (curated allowlist enum — the SSRF host guard), `datasetId` (4x4, from socrata_discover_datasets), optional SoQL `select`/`where`/`order`/`q`, `limit` (≤1000, def 100), `offset`, `withTotal` (def true). HONESTY: SODA's row response has no total, so a count(*) companion supplies an exact totalAvailable; if it fails the rows still return with totalAvailable:null + a note (hasMore is then inferred from page-fill, never a false complete). Genuine-empty ⇒ complete:true/total:0; an outage/400/404 THROWS (never a fake empty). Value fields are strings.",
|
|
5846
|
+
"Query rows from an allowlisted Socrata/SODA open-data portal (keyless; ~a dozen US state portals + USAC E-rate on one identical API — state spend/checkbook/contract/vendor-payment datasets). State procurement mirrors: NY ehig-g5x3, NJ ubnu-tqu7, WA s8d5-pj78, MA cthru.data.socrata.com pegc-naaa (~49M payment rows). Full map: read resource samgov://data-map/state-local. Input `domain` (curated allowlist enum — the SSRF host guard), `datasetId` (4x4, from socrata_discover_datasets), optional SoQL `select`/`where`/`order`/`q`, `limit` (≤1000, def 100), `offset`, `withTotal` (def true). HONESTY: SODA's row response has no total, so a count(*) companion supplies an exact totalAvailable; if it fails the rows still return with totalAvailable:null + a note (hasMore is then inferred from page-fill, never a false complete). Genuine-empty ⇒ complete:true/total:0; an outage/400/404 THROWS (never a fake empty). Value fields are strings.",
|
|
5843
5847
|
inputSchema: SocrataQueryInput,
|
|
5844
5848
|
handler: (input) => socrata.query(input),
|
|
5845
5849
|
}),
|
|
@@ -5854,7 +5858,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
5854
5858
|
defineTool({
|
|
5855
5859
|
name: "ckan_query",
|
|
5856
5860
|
description:
|
|
5857
|
-
"Query rows from an allowlisted CKAN datastore resource (keyless;
|
|
5861
|
+
"Query rows from an allowlisted CKAN datastore resource (keyless; state/city spend/checkbook/procurement/vendor tables on the CKAN Action API). VA eVA PO lines: host=data.virginia.gov resourceId=3c7f1bde-35b0-4fbf-b89c-978a19124d53. Full map: read resource samgov://data-map/state-local. Input `host` (curated allowlist enum — the SSRF host guard: data.ca.gov, data.virginia.gov, data.boston.gov), `resourceId` (36-char lowercase UUID, from ckan_discover_datasets), optional `q` (full-text), `filters` (constrained object {field:value} we JSON.stringify), `sort`, `limit` (≤1000, def 100), `offset`. HONESTY: CKAN's envelope carries a real result.total — the DEFAULT is an EXACT total (exact totalAvailable + hasMore); the rare estimated total (total_was_estimated:true) is disclosed via totalIsEstimated + a note and does NOT drive pagination (it can be above OR below the truth). Genuine-empty ⇒ complete:true/total:0; an outage/404/409 or success:false THROWS (never a fake empty). Values are typed per result.fields[].type.",
|
|
5858
5862
|
inputSchema: CkanQueryInput,
|
|
5859
5863
|
handler: (input) => ckan.query(input),
|
|
5860
5864
|
}),
|
|
@@ -5882,30 +5886,26 @@ export const TOOLS: ToolDef[] = [
|
|
|
5882
5886
|
}),
|
|
5883
5887
|
defineTool({
|
|
5884
5888
|
name: "fdic_bank_failures",
|
|
5885
|
-
description:
|
|
5886
|
-
"Historical FDIC-insured bank failures & assistance transactions (keyless FDIC BankFind, api.fdic.gov/banks/failures) — B2G counterparty / entity due-diligence: a failed or FDIC-assisted institution is a red flag, and CERT links a failure back to fdic_search_institutions / fdic_institution_financials. Exact-key filters: `state` (2-letter → PSTALP — NOTE the /failures state field is PSTALP, NOT STALP), `failYear` (→ FAILYR; e.g. 2023 → the 5 real 2023 failures incl. Silicon Valley Bank & First Republic Bank), `cert` (→ CERT, the STABLE entity key). `limit` (≤1000, def 100), `offset` (≤100000), `sortBy` (allowlisted enum FAILDATE/COST/QBFASSET/QBFDEP/NAME/FAILYR, def FAILDATE), `sortOrder` (def DESC → most-recent first). Returns { failures:[{ name, cert, failDate, failYear, city, state, resolutionType, resolutionFund, estimatedLossUSD, depositsUSD, assetsUSD, id }] }. NO name/city filter — FDIC's /failures `search` param is IGNORED (it returns the whole dataset), so name/city are SHOWN in each row but NOT searchable; to find a specific bank's failure, resolve its CERT via fdic_search_institutions then filter here by `cert`. HONESTY: totalAvailable is the EXACT meta.total (stable across offset — never the page length); failDate is normalized from FDIC's M/D/YYYY to ISO YYYY-MM-DD (an unrecognized value is surfaced raw + disclosed, never nulled/fabricated); COST/QBFDEP/QBFASSET are $thousands normalized to whole USD ×1000 (null-never-0 — a genuine 0 = a fully-assisted no-loss stays 0, a NEGATIVE COST = a net DIF recovery/gain not a loss, absent → null); the ONLY honest empty is meta.total:0/data:[] ⇒ complete:true/total:0, every other envelope (400 errors[]/404/non-JSON/missing meta or data) THROWS (never a fake empty); the point-in-time snapshot build time is disclosed. NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
|
|
5889
|
+
description: "Historical FDIC-insured bank failures & assistance transactions (keyless; api.fdic.gov/banks/failures). CERT links a failure back to fdic_search_institutions / fdic_institution_financials. Filters: `state` (2-letter → PSTALP — NOTE: /failures state field is PSTALP, NOT STALP), `failYear` (→FAILYR), `cert` (→CERT, the STABLE entity key). `limit` (≤1000), `offset` (≤100000), `sortBy` (FAILDATE/COST/QBFASSET/QBFDEP/NAME/FAILYR, def FAILDATE), `sortOrder` (def DESC). Returns { failures:[{ name, cert, failDate, failYear, city, state, resolutionType, resolutionFund, estimatedLossUSD, depositsUSD, assetsUSD, id }] }. NO name/city filter — FDIC /failures `search` param is IGNORED (returns the whole dataset); to find a specific bank's failure, resolve its CERT via fdic_search_institutions. HONESTY: totalAvailable is EXACT meta.total (stable across offset). failDate normalized from FDIC's M/D/YYYY to ISO YYYY-MM-DD (unrecognized → surfaced raw + disclosed, never nulled). COST/QBFDEP/QBFASSET are $thousands normalized to whole USD ×1000 (null-never-0 — genuine 0 = a no-loss assisted transaction stays 0; NEGATIVE COST = a net DIF recovery/gain, not a loss; absent → null). ONLY honest empty is meta.total:0/data:[] → complete:true/total:0; any other envelope (400 errors[]/404/non-JSON/missing meta or data) THROWS (never fake-empty). The point-in-time snapshot build time is disclosed. NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
|
|
5887
5890
|
inputSchema: FdicBankFailuresInput,
|
|
5888
5891
|
handler: (input) => fdic.bankFailures(input),
|
|
5889
5892
|
}),
|
|
5890
5893
|
defineTool({
|
|
5891
5894
|
name: "fdic_institution_history",
|
|
5892
|
-
description:
|
|
5893
|
-
"Institution-level STRUCTURAL-CHANGE event log for FDIC-insured banks (keyless FDIC BankFind, api.fdic.gov/banks/history) — the full lineage of mergers, absorptions, consolidations, failures, name/location/charter/regulator changes, branch open/close, trust-power grants & FRS-membership changes. Completes the FDIC entity cluster (directory + financials + failures + history). Killer feature: CERT-linked MERGER LINEAGE — a merger/failure row carries the acquiring / outgoing / surviving institution's CERT + name, each linking back to fdic_search_institutions / fdic_institution_financials / fdic_bank_failures. Exact-key filters (all optional, AND-combined): `cert` (→ CERT, the STABLE entity key & PRIMARY lookup; e.g. 3510 → Bank of America's 13,794 rows), `changeCode` (→ CHANGECODE; e.g. 223 = merger, 211 = failure, 721 = branch closing, 520 = location change), `effYear` (→ EFFYEAR), `state` (2-letter → PSTALP — NOTE the /history state field is PSTALP, NOT STALP). `limit` (≤1000, def 100), `offset` (≤100000), `sortBy` (allowlisted enum EFFDATE/PROCDATE/CHANGECODE/TRANSNUM, def EFFDATE), `sortOrder` (def DESC → newest change first). Returns { history:[{ cert, instName, state, changeCode, changeDescription, effectiveDate, processDate, effYear, transNum, acquirerCert, acquirerName, outgoingCert, outgoingName, survivingCert, survivingName, id }] }. NO name/city filter — FDIC's /history `search` param returns 0 for INSTNAME (a false-empty), so names are SHOWN in each row but NOT searchable; to find a specific bank's history, resolve its CERT via fdic_search_institutions then filter here by `cert`. HONESTY: totalAvailable is the EXACT meta.total (stable across offset — never the page length); changeDescription is FDIC's OWN co-served CHANGECODE_DESC passed through verbatim (the numeric changeCode is authoritative — never a hand-map); effectiveDate/processDate are normalized from FDIC's YYYY-MM-DDT00:00:00 to ISO YYYY-MM-DD (an unrecognized value is surfaced raw + disclosed, never nulled/fabricated); the acquirer/outgoing/surviving CERTs are null on a non-merger event (null-never-0 — a real absence, never a fabricated 0; *_UNINUM's 0 sentinel is NOT surfaced); the ONLY honest empty is meta.total:0/data:[] ⇒ complete:true/total:0, every other envelope (400 errors[]/404/non-JSON/missing meta or data) THROWS (never a fake empty); the point-in-time snapshot build time is disclosed. NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
|
|
5895
|
+
description: "Institution-level STRUCTURAL-CHANGE event log for FDIC-insured banks (keyless; api.fdic.gov/banks/history) — mergers, absorptions, consolidations, failures, name/location/charter/regulator changes, branch open/close, trust-power grants & FRS-membership. CERT-linked merger lineage: each row carries acquiring/outgoing/surviving institution CERT + name, linking to fdic_search_institutions / fdic_bank_failures. Filters (all optional, AND-combined): `cert` (→CERT, PRIMARY lookup), `changeCode` (→CHANGECODE; e.g. 223=merger, 211=failure, 721=branch closing), `effYear` (→EFFYEAR), `state` (2-letter → PSTALP — NOTE: /history uses PSTALP, NOT STALP). `limit` (≤1000), `offset` (≤100000), `sortBy` (EFFDATE/PROCDATE/CHANGECODE/TRANSNUM), `sortOrder`. Returns { history:[{ cert, instName, state, changeCode, changeDescription, effectiveDate, processDate, effYear, transNum, acquirerCert, acquirerName, outgoingCert, outgoingName, survivingCert, survivingName, id }] }. NO name/city filter — FDIC /history `search` returns 0 for INSTNAME (a false-empty); resolve CERT via fdic_search_institutions first. HONESTY: totalAvailable is EXACT meta.total; changeDescription is FDIC's CHANGECODE_DESC verbatim (changeCode is authoritative — never hand-mapped); effectiveDate/processDate normalized to ISO YYYY-MM-DD (unrecognized → surfaced raw); acquirer/outgoing/surviving CERTs null on non-merger events (null-never-0); ONLY honest empty is meta.total:0/data:[] → complete:true/total:0; any other envelope THROWS (never fake-empty). NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
|
|
5894
5896
|
inputSchema: FdicInstitutionHistoryInput,
|
|
5895
5897
|
handler: (input) => fdic.institutionHistory(input),
|
|
5896
5898
|
}),
|
|
5897
5899
|
defineTool({
|
|
5898
5900
|
name: "fdic_industry_summary",
|
|
5899
|
-
description:
|
|
5900
|
-
"FDIC industry & state banking-sector ANNUAL AGGREGATES — the FDIC's own roll-ups (keyless FDIC BankFind, api.fdic.gov/banks/summary). The FIRST aggregate/statistical FDIC tool (the other 4 are per-ENTITY, keyed on CERT): total assets, deposits, net income, equity & net interest income + structural counts (institutions, offices, branches, employees) for the whole US banking industry OR one state/territory in one year, split by charter class. Answers 'how big is the US (or a state's) banking industry this year, and how many institutions?' — a question the entity tools cannot express without summing thousands of rows. Exact-key filters (all optional, AND-combined): `year` (→ YEAR; e.g. 2023 → 121 rows), `state` (2-or-3-letter → STALP — NOTE the /summary state field is STALP, NOT PSTALP; accepts a jurisdiction code TX/CA/DC/GU/PR… OR a ROLL-UP code USA/US/OT/PI), `charterClass` (CB = commercial banks, SI = savings institutions; omit for both — there is NO combined row). `limit` (≤1000, def 100), `offset` (≤100000), `sortBy` (allowlisted enum YEAR/ASSET/DEP/NETINC/BANKS, def YEAR), `sortOrder` (def DESC → newest year / largest first). Returns { summary:[{ year, charterClass, charterClassCode, geography, stateCode, stateFips, scope, isRollup, institutionCount, officeCount, branchCount, employeeCount, totalAssetsUSD, totalDepositsUSD, netIncomeUSD, totalEquityUSD, netInterestIncomeUSD, id }] }. ★ROLL-UP HONESTY: each row crosses charter × geography; STALP ∈ {USA,US,OT,PI} are GEOGRAPHIC AGGREGATES (scope national_total/national_states_dc/territories_total/pacific_islands, isRollup:true), every other STALP is a jurisdiction (isRollup:false) — NEVER sum a roll-up row with jurisdiction rows or across scopes (national_total = national_states_dc + territories_total; a geography's total = its CB row + its SI row), read the national_total (USA) row directly for one national figure; a roll-up is NOT a state. ★NIM is net interest INCOME (a $ sum surfaced as netInterestIncomeUSD), NOT the margin ratio; this endpoint has NO ratio fields (ROA/ROE — derive from netIncomeUSD/totalAssetsUSD/totalEquityUSD). NO name/city filter — FDIC's /summary `search` param is ignored (returns the whole year); drill to institutions via fdic_search_institutions. HONESTY: totalAvailable is the EXACT meta.total (stable across offset — never the page length); money (ASSET/DEP/NETINC/EQ/NIM) is $thousands → whole USD ×1000 (null-never-0 — a genuine 0 like American Samoa's zero commercial banks stays 0, absent → null), counts (BANKS/OFFICES/BRANCHES/employees) pass through un-scaled (a count ×1000 is a fabrication); a non-int year is rejected pre-fetch (a malformed year is a live HTTP-200 total:0 false-empty); the ONLY honest empty is meta.total:0/data:[] ⇒ complete:true/total:0, every other envelope (400 errors[]/404/non-JSON/missing meta or data) THROWS (never a fake empty); the point-in-time snapshot build time is disclosed. NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
|
|
5901
|
+
description: "FDIC banking-sector ANNUAL AGGREGATES — total assets, deposits, net income, equity & net interest income + institution/office/branch/employee counts for the whole US OR one state, split by charter class (keyless; api.fdic.gov/banks/summary). Filters (all optional): `year` (→YEAR), `state` (→STALP — NOTE: /summary uses STALP, NOT PSTALP; accepts TX/CA/DC/GU/PR or ROLL-UP codes USA/US/OT/PI), `charterClass` (CB=commercial, SI=savings; omit for both). `limit` (≤1000), `offset` (≤100000), `sortBy` (YEAR/ASSET/DEP/NETINC/BANKS), `sortOrder`. Returns { summary:[{ year, charterClass, charterClassCode, geography, stateCode, stateFips, scope, isRollup, institutionCount, officeCount, branchCount, employeeCount, totalAssetsUSD, totalDepositsUSD, netIncomeUSD, totalEquityUSD, netInterestIncomeUSD, id }] }. ★ROLL-UP HONESTY: STALP ∈ {USA,US,OT,PI} are GEOGRAPHIC AGGREGATES (isRollup:true). NEVER sum a roll-up row with jurisdiction rows or across scopes — USA is the one national figure; a roll-up is NOT a state. ★netInterestIncomeUSD is net interest INCOME ($ sum), NOT the margin ratio; NO ratio fields (ROA/ROE); derive from netIncomeUSD/totalAssetsUSD/totalEquityUSD. NO name/city filter — FDIC /summary `search` is ignored; drill via fdic_search_institutions. HONESTY: totalAvailable is EXACT meta.total; money ($thousands → whole USD ×1000, null-never-0; genuine 0 stays 0; absent → null); counts pass through unscaled; non-int year rejected pre-fetch; ONLY honest empty is meta.total:0/data:[] → complete:true/total:0; any other envelope THROWS. NOTE: FDIC keys on CERT, not SAM UEI/DUNS.",
|
|
5901
5902
|
inputSchema: FdicIndustrySummaryInput,
|
|
5902
5903
|
handler: (input) => fdic.industrySummary(input),
|
|
5903
5904
|
}),
|
|
5904
5905
|
// ━━━ FDIC BankFind Suite — WITHIN-SOURCE DEPTH: counterparty risk ratios + branch deposits (2) ━━━ ADR-0040
|
|
5905
5906
|
defineTool({
|
|
5906
5907
|
name: "fdic_risk_ratios",
|
|
5907
|
-
description:
|
|
5908
|
-
"FDIC counterparty RISK RATIOS for ONE institution by certificate number (keyless FDIC BankFind, api.fdic.gov/banks/financials) — the SOUNDNESS lane the balance-sheet tools cannot express: profitability (ROA/pretax ROA/ROE), net interest margin, efficiency ratio, asset quality (net charge-offs to loans), capital adequacy (leverage, tier-1 risk-based, total risk-based ratios) + the tier-1 capital LEVEL. Input `cert` (REQUIRED FDIC certificate number, from fdic_search_institutions), `reportDate` (optional YYYYMMDD quarter-end → REPDTE; omit for the full quarterly time-series), `limit` (≤1000, def 100), `offset` (≤100000), `sortBy` (allowlisted enum REPDTE/ROA/ROE/RBCRWAJ/EEFFR, def REPDTE), `sortOrder` (def DESC → newest quarter first). Returns { cert, ratios:[{ cert, reportDate, cblrFramework, returnOnAssetsPct, preTaxReturnOnAssetsPct, returnOnEquityPct, netInterestMarginPct, efficiencyRatioPct, netChargeOffsToLoansPct, leverageRatioPct, tier1RiskBasedCapitalRatioPct, totalRiskBasedCapitalRatioPct, tier1CapitalUSD, id }] }. ★UNITS-IN-THE-KEY: every *Pct field is an FDIC-published PERCENTAGE surfaced VERBATIM (no scaling, no recompute) — do NOT read it as a dollar amount or ×1000-scale it; tier1CapitalUSD is a DOLLAR amount (FDIC publishes it in $thousands, normalized ×1000). ★NULL-NEVER-0: a not-reported ratio is null (never 0% — a false 'no return / no capital'). ★CBLR (community-bank-leverage) banks (cblrFramework:true) do NOT report the risk-based capital ratios — FDIC returns a literal 0 for the total risk-based ratio, which this tool maps to null for BOTH tier1RiskBasedCapitalRatioPct and totalRiskBasedCapitalRatioPct (a null there is a normal framework artifact, read alongside leverageRatioPct — NOT a 0% capital red flag). No ratio is recomputed; each is exactly FDIC's published Call-Report figure. HONESTY: totalAvailable is the EXACT meta.total (stable across offset); the ONLY honest empty is meta.total:0/data:[] ⇒ complete:true/total:0, every other envelope (400 errors[]/404/non-JSON/missing meta or data) THROWS (never a fake empty); the snapshot build time is disclosed. NOTE: reported regulatory metrics, NOT a soundness rating or failure prediction; FDIC keys on CERT, not SAM UEI/DUNS.",
|
|
5908
|
+
description: "FDIC risk ratios for ONE institution by certificate number (keyless; api.fdic.gov/banks/financials) — profitability (ROA/pretax ROA/ROE), net interest margin, efficiency ratio, asset quality (net charge-offs to loans), capital adequacy (leverage, tier-1 risk-based, total risk-based ratios) + tier-1 capital level. Input: `cert` (REQUIRED FDIC certificate, from fdic_search_institutions), `reportDate` (optional YYYYMMDD quarter-end → REPDTE; omit for full quarterly time-series), `limit` (≤1000), `offset`, `sortBy` (REPDTE/ROA/ROE/RBCRWAJ/EEFFR), `sortOrder`. Returns { cert, ratios:[{ cert, reportDate, cblrFramework, returnOnAssetsPct, preTaxReturnOnAssetsPct, returnOnEquityPct, netInterestMarginPct, efficiencyRatioPct, netChargeOffsToLoansPct, leverageRatioPct, tier1RiskBasedCapitalRatioPct, totalRiskBasedCapitalRatioPct, tier1CapitalUSD, id }] }. ★UNITS: every *Pct field is FDIC-published PERCENTAGE verbatim (no scaling); tier1CapitalUSD is $thousands × 1000. ★NULL-NEVER-0: not-reported ratio is null (never 0). ★CBLR banks (cblrFramework:true) do NOT report risk-based capital ratios — FDIC returns literal 0 for totalRiskBased only; this tool maps 0→null for BOTH tier1RiskBased and totalRiskBased; null is a framework artifact, not a 0% red flag — read alongside leverageRatioPct. No ratio is recomputed; each is FDIC's published Call-Report figure verbatim. HONESTY: totalAvailable is EXACT meta.total; ONLY honest empty is meta.total:0/data:[] → complete:true/total:0; any other envelope THROWS. NOTE: regulatory metrics, NOT a soundness rating. FDIC keys on CERT, not SAM UEI/DUNS.",
|
|
5909
5909
|
inputSchema: FdicRiskRatiosInput,
|
|
5910
5910
|
handler: (input) => fdic.riskRatios(input),
|
|
5911
5911
|
}),
|
|
@@ -5919,8 +5919,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
5919
5919
|
// ━━━ USITC Harmonized Tariff Schedule — keyless import-tariff / duty-rate lookup (1) ━━━ ADR-0039
|
|
5920
5920
|
defineTool({
|
|
5921
5921
|
name: "hts_lookup",
|
|
5922
|
-
description:
|
|
5923
|
-
"Look up US import-tariff classification + duty rates from the USITC Harmonized Tariff Schedule (keyless; hts.usitc.gov/reststop/search) — the IMPORT-TARIFF / supply-chain PRICE lane a product-reseller / supply-chain bidder needs to price a hardware or commodity contract (extends the THIN Price lane with a NON-labor cost input, a sibling of gsa_benchmark_labor_rates). A single `query` serves BOTH modes: a KEYWORD (e.g. 'laptop', 'cotton shirt') OR an HTS number (e.g. '8471.30' / '8471.30.01.00') — both ride the `keyword=` search. Returns { query, lines:[{ htsno, statisticalSuffix, indent, description, units, columnOneGeneral, specialPreferential, columnTwo, additionalDuties, footnotes, quotaQuantity, effectivePeriod, status, isChapter99 }] } + honest _meta. ★DUTY-RATE HONESTY (the crux): columnOneGeneral (Column-1 General), specialPreferential (Special/preferential/FTA), and columnTwo (Column-2) are AUTHORITATIVE VERBATIM TEXT surfaced as strings — 'Free', a percentage ('35%'), a specific rate ('0.47¢/kg'), a compound/range, or null — NEVER coerced to a number (a coerced 0/NaN would fabricate a false 'duty-free'); an empty Special ('') → null = NO special-program rate published (NEVER read as Free). ★HIERARCHY (M1): a lookup returns rows across levels; the rate is stated ONCE at a shallower level (usually the 6/8-digit subheading) and inherits DOWNWARD to the blank statistical-suffix lines — to find a specific line's rate, read UP to the nearest ANCESTOR line (shallower indent, same htsno prefix) with a non-empty rate; a blank deepest line is NOT no/unknown duty. ★ADDITIONAL DUTIES (S1): the per-line `additionalDuties` is frequently null even when Section 301/232 duties apply — the real additional duty rides the Chapter-99 rows (isChapter99:true, htsno beginning '99') returned alongside the base line + the footnotes; they STACK on the base rate. ★COMPLETENESS (M2): the endpoint returns the FULL match array with NO server-side total and NO working pagination (offset is IGNORED) → totalAvailable is the EXACT served array length and paging is CLIENT-SIDE; there is no fixed cap (a single-char/common fragment can return 10,000–16,000+ rows / several MB), so `query` must be ≥3 non-whitespace chars (a 1–2 char query is rejected invalid_input before the fetch). `limit` (≤200, def 50), `offset`. A no-match ⇒ honest empty; a 404/5xx/timeout/non-array/HTML(→schema_drift) ⇒ THROWS (never a fake empty); a transient 400 on the validated query ⇒ upstream_unavailable (retryable). NOT a binding CBP classification ruling and NOT a landed-cost quote — the duty owed depends on country of origin + trade program + Section 301/232 / Chapter-99 additional duties + footnotes; confirm via CBP (CROSS / eRulings). The not-a-ruling caveat rides EVERY response.",
|
|
5922
|
+
description: "Look up US import-tariff classification + duty rates from the USITC Harmonized Tariff Schedule (keyless; hts.usitc.gov/reststop/search). A single `query` serves BOTH modes: KEYWORD (e.g. 'laptop') OR HTS number (e.g. '8471.30'). Returns { query, lines:[{ htsno, statisticalSuffix, indent, description, units, columnOneGeneral, specialPreferential, columnTwo, additionalDuties, footnotes, quotaQuantity, effectivePeriod, status, isChapter99 }] } + honest _meta. ★DUTY-RATE HONESTY: columnOneGeneral, specialPreferential, and columnTwo are AUTHORITATIVE VERBATIM TEXT — 'Free', '35%', '0.47¢/kg', compound/range, or null — NEVER coerced to a number (0/NaN fabricates a false 'duty-free'); empty Special ('') → null = NO special rate (NEVER read as Free). ★HIERARCHY: rate stated at a shallower level and inherits downward; read to the nearest ancestor with a non-empty rate; blank deepest ≠ 'no duty'. ★ADDITIONAL DUTIES: `additionalDuties` is frequently null even when Section 301/232 duties apply — the real additional duty rides Chapter-99 rows (isChapter99:true) and STACKS on the base rate. ★COMPLETENESS: endpoint returns the FULL match array; offset IGNORED → totalAvailable is the EXACT served array length; paging CLIENT-SIDE. query must be ≥3 non-whitespace chars (1-2 → invalid_input). limit (≤200, def 50), offset. No-match → honest empty; 404/5xx/timeout/non-array/HTML → schema_drift THROW; transient 400 → upstream_unavailable. NOT a binding CBP ruling or landed-cost quote — duty owed depends on country of origin + trade program + Ch-99 stacking; confirm via CBP (CROSS/eRulings).",
|
|
5924
5923
|
inputSchema: HtsLookupInput,
|
|
5925
5924
|
handler: (input) => usitc.htsLookup(input),
|
|
5926
5925
|
}),
|
|
@@ -5935,8 +5934,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
5935
5934
|
// rides ONLY in the POST body (v2) — never a URL/header/label/_meta/log.
|
|
5936
5935
|
defineTool({
|
|
5937
5936
|
name: "bls_timeseries",
|
|
5938
|
-
description:
|
|
5939
|
-
"Fetch US Bureau of Labor Statistics time series — the PRICING / ESCALATION layer (keyless; api.bls.gov Public Data API v1, POST/JSON batch). CPI-U & ECI drive federal contract escalation / economic-price-adjustment (EPA) clauses; PPI benchmarks materials pricing; CES employment/wages give labor-rate context (next to gsa_benchmark_labor_rates + sam wage determinations). Inputs (at least one of series/seriesId REQUIRED; both combinable): `series` — a FROZEN 9-key CURATED enum (typo-proof; each carries meaning + units): cpi_u_all/cpi_u_core (CPI-U index, NSA — the escalation reference), ppi_final_demand (PPI index), eci_total_comp/eci_wages (★12-MONTH % CHANGE, NOT an index — a consumer misreads 3.4 as an index level otherwise), unemployment_rate/labor_force_participation (percent, SA), employment_total_nonfarm (thousands of persons, SA), avg_hourly_earnings (dollars/hour, SA). `seriesId` — raw BLS IDs (charclass ^[A-Z0-9]{1,20}$; the OEWS/local-area/regional passthrough; units:null for a raw ID). `startYear`/`endYear` (1900..currentYear+1; default a ~10-year window; span CLAMPED to the tier cap ~10y and disclosed). Returns { series:[{ seriesId, key, meaning, units, observations:[{ year, period, periodName, value, valueUnavailable, footnotes, latest }], observationCount, coveredRange }] } + honest _meta. HONESTY: each `value` is PARSED number|null — the BLS \"-\" unavailable marker (e.g. the 2025 lapse-in-appropriations gap) → null NEVER 0, with valueUnavailable:true + the footnote reason on the observation AND lifted into _meta.notes (a data gap is DISCLOSED, never a silent null and never a fabricated 0); a genuine \"0\" stays 0. A non-SUCCESS status THROWS (never a fake-empty): REQUEST_NOT_PROCESSED (the v1 ~25/day limit) ⇒ rate_limited with the tier disclosure; REQUEST_FAILED ⇒ upstream_unavailable/invalid_input surfacing message[]. A non-JSON 200 or a SUCCESS body missing Results.series ⇒ schema_drift. An empty data[] on SUCCESS ⇒ observations:[] + an ambiguity note (a curated key = a genuine empty range; a raw seriesId = EITHER genuine-empty OR a nonexistent/typo'd ID — verify it). Every response discloses the active tier (v1 keyless ~25/day, 25 series/query, ~10y span | v2 with a free BLS_API_KEY ~500/day) + the per-series units caveat. An OPTIONAL free BLS_API_KEY (env; https://data.bls.gov/registrationEngine/) lifts to v2 and is sent ONLY in the request body — never a URL/header/log.",
|
|
5937
|
+
description: "Fetch one or more BLS time-series over a year range (keyless v1 default; optional free BLS_API_KEY lifts to v2; api.bls.gov POST). At least one of `series` (curated enum key) or `seriesId` (raw ^[A-Z0-9]{1,25}$; covers 25-char OEWS IDs) is required; both may be combined. Optional `startYear`/`endYear` (defaults to active tier's span cap). Returns { series:[{ seriesId, key, meaning, units, observations:[{year, period, periodName, value:number|null, valueUnavailable:bool, footnotes:[{code,text}], latest:bool}], observationCount, coveredRange:{from,to} }] } + honest _meta. HONESTY: BLS '-' unavailable marker → value:null (NEVER 0); valueUnavailable:true on the observation + footnote reason lifted into _meta.notes so the gap is DISCLOSED, never silent. Each series carries its own units label — an ECI '…A' series is a 12-month PERCENT CHANGE, NOT an index level; CPI/PPI are index levels; CES nonfarm employment is thousands of persons. Do NOT compare values across series without reading each units label. status !== 'REQUEST_SUCCEEDED' THROWS: REQUEST_NOT_PROCESSED (v1 daily limit) → rate_limited retryable; REQUEST_FAILED → upstream_unavailable. Series count refused over active tier cap (v1: 25 series/~10yr; v2: 50 series/~20yr) — overflow is NEVER silently dropped. Span is CLAMPED to tier cap before the fetch and disclosed. totalAvailable is null (batch fetch has no upstream total). A typo'd seriesId returns an empty series with 'Invalid Series' upstream message — NOT a real available series. BLS_API_KEY rides ONLY in the POST body, never URL/label/_meta/log.",
|
|
5940
5938
|
inputSchema: BlsTimeseriesInput,
|
|
5941
5939
|
handler: (input) => bls.timeseries(input),
|
|
5942
5940
|
}),
|
|
@@ -5944,13 +5942,12 @@ export const TOOLS: ToolDef[] = [
|
|
|
5944
5942
|
// The LEVEL layer next to bls_timeseries's ESCALATION layer: mean/median annual &
|
|
5945
5943
|
// hourly wages + employment by SOC occupation × geography — the highest-value B2G
|
|
5946
5944
|
// BLS slice (labor-rate benchmarking) that bls_timeseries structurally cannot reach
|
|
5947
|
-
// (OEWS IDs are 25 chars
|
|
5948
|
-
// ID INTERNALLY from validated structured inputs; REUSES the same POST/JSON transport
|
|
5945
|
+
// (OEWS IDs are 25 chars; bls_timeseries raw-seriesId cap is now widened to 25). BUILDS
|
|
5946
|
+
// the 25-char series ID INTERNALLY from validated structured inputs; REUSES the same POST/JSON transport
|
|
5949
5947
|
// + parseBlsBody status-throw + mapObservation ("-"→null-never-0) + tier/key seam.
|
|
5950
5948
|
defineTool({
|
|
5951
5949
|
name: "bls_oews_wages",
|
|
5952
|
-
description:
|
|
5953
|
-
"Benchmark US occupational wages & employment from BLS OEWS (Occupational Employment & Wage Statistics) — the LEVEL layer for labor-rate benchmarking (keyless; api.bls.gov Public Data API, POST/JSON batch). The actual mean/median annual & hourly wage a labor category commands, by area — next to gsa_benchmark_labor_rates (GSA CALC), sam wage determinations, and bls_timeseries (the CPI/ECI escalation layer). OEWS series IDs are 25 chars (area×occupation×industry×datatype), EXCEEDING bls_timeseries's raw-seriesId cap, so this tool BUILDS the ID INTERNALLY from validated structured inputs. Inputs (at least one of occupation/soc REQUIRED; all arrays batch into ONE POST — the cartesian product area×occupation×datatype is capped at the active tier's series cap and refused over-cap WITH THE COUNT NAMED, never silently truncated): `occupation` — a CURATED 16-key SOC enum (typo-proof; e.g. software_developer=15-1252, civil_engineer=17-2051, management_analyst=13-1111); `soc` — raw 6-digit HYPHENLESS SOC codes for the ~830-SOC long tail (use 151252, not 15-1252); `area` — default [\"national\"]; each is \"national\", a 2-letter USPS state code (CA/TX/DC…), or a 5-digit CBSA metro code (19100 = Dallas-Fort Worth); `datatype` — default [\"annual_mean\"]: annual_mean/annual_median (dollars/year), hourly_mean/hourly_median (dollars/hour), employment (count jobs). NO year input — OEWS is ANNUAL and the API serves only the latest release; the tool requests a recent window internally and DISCLOSES the reference year. Returns { results:[{ area:{type,code,label}, occupation:{soc,key,label}, measure:{key,code,units}, value:number|null, valueUnavailable, referenceYear, referencePeriod, footnotes, seriesId }] } + honest _meta. HONESTY: (H1) OEWS is an ANNUAL point-in-time snapshot (reference May <year>, period A01), NOT monthly/current-quarter — disclosed every call; (H2) a built ID that returns empty/absent ⇒ value:null, valueUnavailable:FALSE (the occupation is not surveyed/estimated there OR the cell is suppressed for confidentiality) + the not-published note + the surfaced upstream \"Series does not exist\" message + the ID in fieldsUnavailable — NEVER a fabricated 0; a PRESENT \"-\" in-band value ⇒ null + valueUnavailable:true + footnote; (H3) each row's measure.units labels the datatype (never read an employment count as a wage); (H4) the API returns real numerics (no top-code); a non-SUCCESS status THROWS (REQUEST_NOT_PROCESSED ⇒ rate_limited with the tier disclosure; a non-JSON 200 ⇒ schema_drift). Every response discloses the active tier. An OPTIONAL free BLS_API_KEY lifts to v2 and is sent ONLY in the request body — never a URL/header/log.",
|
|
5950
|
+
description: "BLS OEWS occupational wage/employment benchmarking by area × occupation × datatype (keyless; api.bls.gov). Builds validated 25-char series IDs from structured inputs and batches them into one POST — NO year input (OEWS serves only the latest annual release). Inputs: `occupation` (curated enum, e.g. 'software_developer') or `soc` (raw 6-digit SOC, ^[0-9]{6}$ NO hyphen — at least ONE required); `area` (default 'national', 2-letter USPS state, or 5-digit CBSA metro code); `datatype` (default 'annual_mean'; annual_mean/annual_median/hourly_mean/hourly_median/employment). Returns { results:[{ area:{type,code,label}, occupation:{soc,key,label}, measure:{key,code,units}, value:number|null, valueUnavailable:bool, referenceYear, referencePeriod, footnotes, seriesId }] } + honest _meta. ★H1: OEWS is an ANNUAL point-in-time snapshot (reference May <year>); BLS API serves ONLY the most recent release, may lag ~1 year. NOT monthly/current-quarter. ★H2: a built-ID with no published value (occupation not surveyed or suppressed in that area) → value:null (NOT a tool error). ★H3: measure.units is set from datatype — annual_mean/annual_median=dollars/year; hourly_mean/hourly_median=dollars/hour; employment=count. NEVER mislabeled. ★H4: the API returns real numerics (no '#' top-code). area×occupation×datatype is capped at the tier's series cap (v1 25 / v2 50) and refused over-cap with the count named — never silently truncated; REQUEST_NOT_PROCESSED → rate_limited THROWS. Active tier (v1 keyless ~25/day or v2 BLS_API_KEY ~500/day) and series-cap limits disclosed.",
|
|
5954
5951
|
inputSchema: BlsOewsWagesInput,
|
|
5955
5952
|
handler: (input) => bls.oewsWages(input),
|
|
5956
5953
|
}),
|
|
@@ -5967,8 +5964,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
5967
5964
|
// → honest empty. NEW gate key "bls_qcew"; NO BLS_API_KEY on this keyless path.
|
|
5968
5965
|
defineTool({
|
|
5969
5966
|
name: "bls_qcew",
|
|
5970
|
-
description:
|
|
5971
|
-
"BLS QCEW (Quarterly Census of Employment & Wages) — county×NAICS MARKET-SIZE / wages / location-quotient (keyless; data.bls.gov/cew Open Data Access CSV, a SECOND un-rate-limited BLS domain — NOT the ~25/day api.bls.gov timeseries API). Answers the market-size / competition-density question no other tool can: for ONE area_fips (county/state/metro/US) OR ONE NAICS × quarter — establishment COUNT (market size / competitor density), county×NAICS employment, average weekly wage (labor cost), and the LOCATION QUOTIENT (lq_* = concentration vs the national average; >1.00 = more concentrated / higher competition density). Inputs: `mode` (REQUIRED {area,industry}); `area` (area_fips ^[0-9A-Za-z]{1,6}$ — REQUIRED path segment for mode=area, else an optional client-side narrow); `industry` (NAICS ^[0-9]{1,6}$ DIGIT-ONLY — REQUIRED path segment for mode=industry, else an optional narrow; a hyphenated 31-33 404s, use the digit aggregate); `year` (REQUIRED 1990..current), `quarter` (REQUIRED 1|2|3|4); client-side `ownership`(own_code)/`aggregationLevel`(agglvl_code)/`sizeCode`; `limit` (≤1000, def 50)/`offset`. Wire: GET data.bls.gov/cew/data/api/{year}/{quarter}/{mode}/{code}.csv. Returns { found, mode, area|industry, year, quarter, rows:[{ area_fips, own_code, industry_code, agglvl_code, size_code, base:{ disclosed, disclosureCode, qtrly_estabs, month1/2/3_emplvl, total_qtrly_wages, taxable_qtrly_wages, qtrly_contributions, avg_wkly_wage }, locationQuotient:{ disclosed, disclosureCode, lq_qtrly_estabs, lq_… }, overTheYear:{ disclosed, disclosureCode, oty_qtrly_estabs_chg, oty_…_pct_chg } }] } + honest _meta. ★DISCLOSURE-SUPPRESSION HONESTY (the crux): each row carries THREE disclosure codes (base/lq/oty), each governing its block. QCEW encodes a SUPPRESSED (confidential) employment/wage value as a literal 0 — so under 'N' the confidential emplvl/wage/avg-wkly fields map to null (WITHHELD, never a fabricated $0), while the establishment COUNT (qtrly_estabs / lq_qtrly_estabs) AND its over-the-year change (oty_qtrly_estabs_chg / _pct_chg) stay DISCLOSED (real); under '-' the WHOLE block incl. the estabs field(s) → null; under blank a genuine reported/NEGATIVE 0 SURVIVES (the disclosed federal taxable=0/contrib=0 and the oty_*_chg=0 'no change'). NEVER a blanket 0→null. A null carries disclosed:false + the raw disclosureCode; a suppression note fires whenever any page row is suppressed. HONESTY: totalAvailable is the EXACT filtered row count (fetch-once + client-side limit/offset — QCEW does not paginate; never the page length); a per-tuple HTTP 404 ⇒ honest empty (found:false, the HTML 404 body NEVER parsed as CSV); a 5xx/timeout ⇒ THROW; a 200 non-CSV / a renamed/±column header / a wrong field-count row ⇒ schema_drift THROW (symmetric drift guard). The file MIXES aggregation levels + ownerships — a do-NOT-sum-across-agglvl/ownership note rides every response. PUBLIC AGGREGATE stats (the suppression mechanism keeps small-cell data non-identifying — no PII). Keyless, un-rate-limited; NO BLS_API_KEY is read.",
|
|
5967
|
+
description: "BLS QCEW (Quarterly Census of Employment & Wages) — county×NAICS market-size / wages / location-quotient (keyless; data.bls.gov/cew Open Data Access CSV, un-rate-limited). Inputs: `mode` (REQUIRED {area,industry}); `area` (area_fips ^[0-9A-Za-z]{1,6}$ — REQUIRED for mode=area); `industry` (NAICS ^[0-9]{1,6}$ DIGIT-ONLY — REQUIRED for mode=industry; hyphenated 31-33 404s, use digit aggregate); `year` (REQUIRED), `quarter` (REQUIRED 1|2|3|4); client-side `ownership`/`aggregationLevel`/`sizeCode`; `limit`/`offset`. Returns { found, mode, rows:[{ area_fips, own_code, industry_code, agglvl_code, size_code, base:{disclosed, disclosureCode, qtrly_estabs, month1/2/3_emplvl, total_qtrly_wages, taxable_qtrly_wages, qtrly_contributions, avg_wkly_wage}, locationQuotient:{disclosed, disclosureCode, lq_…}, overTheYear:{disclosed, disclosureCode, oty_…} }] }. ★DISCLOSURE HONESTY: each row has three disclosure codes (base/lq/oty). QCEW encodes SUPPRESSED values as literal 0 — under 'N': confidential emplvl/wage/avg-wkly → null (WITHHELD), estab count + oty-estab change stay DISCLOSED; under '-': WHOLE block → null; under blank: genuine reported/NEGATIVE 0 SURVIVES. NEVER blanket 0→null; null carries disclosed:false + raw disclosureCode; suppression note fires on any suppressed row. HONESTY: totalAvailable is EXACT filtered row count (fetch-once; QCEW does not paginate); per-tuple HTTP 404 → honest empty; 5xx/timeout THROW; 200 non-CSV/renamed header/wrong field-count → schema_drift THROW. Do-NOT-sum-across-agglvl/ownership note rides every response.",
|
|
5972
5968
|
inputSchema: BlsQcewInput,
|
|
5973
5969
|
handler: (input) => bls.qcew(input),
|
|
5974
5970
|
}),
|
|
@@ -6030,8 +6026,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6030
6026
|
// (silent no-op). The 15,000-record retrieval window is disclosed, not hidden.
|
|
6031
6027
|
defineTool({
|
|
6032
6028
|
name: "nih_reporter_search_projects",
|
|
6033
|
-
description:
|
|
6034
|
-
"Search awarded NIH RePORTER research-GRANT projects (keyless; api.reporter.nih.gov v2, POST/JSON — the FIRST non-GET getJson-port consumer) — the NEW federal research-funding recipient-enrichment axis (who receives NIH research money, by organization / state, joinable to SAM/USAspending via primary_uei). Structured, LIVE-CONFIRMED-narrowing criteria ONLY, AND-combined in a module-built body (NO raw passthrough): orgStates (UPPERCASE 2-letter USPS enum — the SSRF + silent-zero guard; a lowercase/unknown code silently returns zeros), orgNames (≤512 each, ≤20), fiscalYears (int array 1985..currentYear+1, ≤20), limit (1..500, def 50), offset (0..14,999, def 0). Returns { projects:[{ projectNum, projectTitle, fiscalYear, awardAmount, organization:{ name, state, primaryUei, primaryDuns, ueis, duns }, principalInvestigators, contactPiName, fundingIc }] } + honest _meta. HONESTY: (M2) records are RESEARCH GRANTS, NOT procurement contracts — primary_uei joins to SAM/USAspending recipients but the award nature differs (disclosed in every _meta.notes); totalAvailable = the EXACT meta.total (NEVER the page size, NEVER a lower bound); NIH caps keyless retrieval at the first 15,000 records (offset 0..14,999) — offset ≥ 15,000 ⇒ invalid_input, and past the window the count stays exact while records are UNREACHABLE (disclosed in a note; nextOffset is never a dead-end). Disclose-not-refuse: an unscoped query still returns the first page + the exact total + a narrow-your-criteria note. agencyIcCodes is intentionally NOT a filter (NIH silently drops it — it would be a false 'applied'). Genuine-empty (total:0) ⇒ complete:true/total:0; an outage/5xx/timeout THROWS; a 400 (bad offset/limit/type) ⇒ invalid_input; a 200 body that isn't {meta,results} or a non-numeric meta.total ⇒ schema_drift (never a fake empty). awardAmount is number|null (a real $0 award is 0, an absent amount is null).",
|
|
6029
|
+
description: "Search awarded NIH RePORTER research-grant projects (keyless; api.reporter.nih.gov v2, POST/JSON), joinable to SAM/USAspending via primary_uei. LIVE-CONFIRMED-narrowing criteria ONLY, AND-combined: orgStates (UPPERCASE 2-letter USPS — lowercase/unknown code silently returns zeros), orgNames (≤512 chars each, ≤20 names), fiscalYears (int array, 1985..currentYear+1, ≤20), limit (1..500), offset (0..14,999). Returns { projects:[{ projectNum, projectTitle, fiscalYear, awardAmount, organization:{name, state, primaryUei, primaryDuns, ueis, duns}, principalInvestigators, contactPiName, fundingIc }] } + honest _meta. HONESTY: records are RESEARCH GRANTS, NOT procurement contracts — primaryUei joins SAM/USAspending but the award nature differs (disclosed in every _meta.notes). totalAvailable = EXACT meta.total (NEVER the page size, NEVER a lower bound). NIH caps keyless retrieval at the first 15,000 records (offset 0..14,999); offset ≥ 15,000 → invalid_input; past the window the count stays exact while records are UNREACHABLE (disclosed in a note). An unscoped query returns the first page + exact total + narrow-your-criteria note. agencyIcCodes is NOT a filter (NIH silently drops it — would be a false 'applied'). awardAmount is number|null (genuine $0 is 0, absent is null). Genuine total:0 → complete:true/total:0; outage/5xx THROWS; 400 (bad offset/limit/type) → invalid_input; 200 not {meta,results} or non-numeric meta.total → schema_drift.",
|
|
6035
6030
|
inputSchema: NihSearchProjectsInput,
|
|
6036
6031
|
handler: (input) => nih.searchProjects(input),
|
|
6037
6032
|
}),
|
|
@@ -6047,8 +6042,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6047
6042
|
// HTTP 200 loud-fails (never a fake empty); grant≠contract in every response.
|
|
6048
6043
|
defineTool({
|
|
6049
6044
|
name: "nsf_search_awards",
|
|
6050
|
-
description:
|
|
6051
|
-
"Search awarded NSF research-GRANT awards (keyless; api.nsf.gov/services/v1/awards.json) — the NEW federal research-funding recipient-enrichment axis (who receives NSF research money, by organization / UEI / PI / state, joinable to SAM/USAspending via ueiNumber/parentUeiNumber). The grant-SIBLING of nih_reporter_search_projects on a different agency. LIVE-CONFIRMED-narrowing filters ONLY, module-built into a URLSearchParams query (NO raw passthrough): keyword (free text; MULTI-WORD is OR-tokenized — 'machine learning' = machine OR learning, disclosed in _meta.notes), awardeeStateCode (UPPERCASE 2-letter USPS enum — the SSRF + silent-zero guard; a non-state typo silently returns 0), awardeeName, ueiNumber (12-char UEI — an EXACT SAM/USAspending join), parentUeiNumber (parent-org roll-up), pdPIName, dateStart/dateEnd (STRICT mm/dd/yyyy on the award ACTION date — a wrong format is silently mis-parsed), limit (1..100, def 25 → rpp), offset (0..9999). Returns { awards:[{ id, title, agency, cfdaNumber, transType, awardee:{ name, city, stateCode, ueiNumber, parentUeiNumber }, performanceSite, principalInvestigator:{ fullName, firstName, lastName, middleInitial, email, id }, coPrincipalInvestigators, programOfficer, amounts:{ fundsObligatedAmt, estimatedTotalAmt, fundsObligatedByYear }, dates, program, activeAward, historicalAward }] } (abstract EXCLUDED — use nsf_get_award) + honest _meta. HONESTY: NSF Awards are RESEARCH GRANTS, NOT procurement contracts (ueiNumber joins to SAM/USAspending but the award nature differs — disclosed every response); totalAvailable = the EXACT metadata.totalCount below 10,000 and SATURATES at 10,000 (an ES track_total_hits cap ⇒ totalIsLowerBound:true + a note — the true total is ≥10,000 and only the first 10,000 are retrievable); NSF caps keyless retrieval at offset+rpp ≤ 10,000 (offset ≥ 10,000 ⇒ invalid_input; the outgoing rpp is clamped so a page never crosses the window). fundsObligatedAmt/estimatedTotalAmt arrive as STRINGS → number|null (a real $0 is 0, absent is null). Genuine-empty (totalCount:0) ⇒ complete:true/total:0; a serviceNotification at HTTP 200 (bad param / deep offset) ⇒ invalid_input/upstream_unavailable THROWS (never a fake empty); an outage/5xx/timeout THROWS; a 200 body that isn't {response:{award,metadata}} or a non-numeric totalCount ⇒ schema_drift. Feed a row's id to nsf_get_award for the full record + abstractText.",
|
|
6045
|
+
description: "Search awarded NSF research-grant awards (keyless; api.nsf.gov/services/v1/awards.json), joinable to SAM/USAspending via ueiNumber/parentUeiNumber. Filters: keyword (MULTI-WORD is OR-tokenized — 'machine learning' = machine OR learning, disclosed), awardeeStateCode (UPPERCASE 2-letter USPS — non-state typo silently returns 0), awardeeName, ueiNumber (12-char UEI — EXACT SAM/USAspending join), parentUeiNumber, pdPIName, dateStart/dateEnd (strict mm/dd/yyyy — wrong format silently mis-parsed), limit (1..100), offset (0..9999). Returns { awards:[{ id, title, agency, cfdaNumber, transType, awardee:{name, city, stateCode, ueiNumber, parentUeiNumber}, principalInvestigator, coPrincipalInvestigators, programOfficer, amounts:{fundsObligatedAmt, estimatedTotalAmt, fundsObligatedByYear}, dates, program, activeAward, historicalAward }] } (abstract EXCLUDED — use nsf_get_award) + honest _meta. HONESTY: NSF Awards are RESEARCH GRANTS, NOT procurement contracts (ueiNumber joins SAM/USAspending but the award nature differs — disclosed every response). totalAvailable = EXACT metadata.totalCount below 10,000; SATURATES at 10,000 (ES track_total_hits cap → totalIsLowerBound:true + note; first 10,000 only retrievable). NSF caps retrieval at offset+rpp ≤ 10,000 (offset ≥ 10,000 → invalid_input). fundsObligatedAmt/estimatedTotalAmt: STRINGS → number|null (genuine $0 is 0, absent is null). Genuine totalCount:0 → complete:true/total:0; serviceNotification at HTTP 200 → THROWS; outage/5xx THROWS; 200 not {response:{award,metadata}} or non-numeric totalCount → schema_drift.",
|
|
6052
6046
|
inputSchema: NsfSearchAwardsInput,
|
|
6053
6047
|
handler: (input) => nsf.searchAwards(input),
|
|
6054
6048
|
}),
|
|
@@ -6073,8 +6067,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6073
6067
|
// AND-tokenized (disclosed); trial≠federal-award caveat in every response.
|
|
6074
6068
|
defineTool({
|
|
6075
6069
|
name: "clinicaltrials_search_studies",
|
|
6076
|
-
description:
|
|
6077
|
-
"Search federally-registered clinical-research studies with LEAD-SPONSOR / COLLABORATOR / ORGANIZATION / FUNDING-SOURCE entity enrichment (keyless; clinicaltrials.gov/api/v2/studies) — the trial-REGISTRATION axis of the research-funding entity layer (the sponsor/collaborator NAMES overlap the pharma/biotech/university/agency entities in NIH RePORTER / NSF Awards / SAM / USAspending). LIVE-CONFIRMED-narrowing filters ONLY, module-built into a URLSearchParams query (NO raw passthrough): query.term (broad free-text), sponsor (→query.spons — a fuzzy sponsor NAME search), condition (→query.cond), location (→query.locn), overallStatus (a frozen 14-value enum → filter.overallStatus), funderType (a frozen 4-value enum nih/fed/industry/other → aggFilters — the FEDERAL-funding axis), pageSize (1..1000, def 20), pageToken (the OPAQUE cursor). Returns { studies:[{ nctId, briefTitle, orgStudyId, organization:{ name, class }, leadSponsor:{ name, class }, collaborators:[{ name, class }], fundingClass, overallStatus, startDate, studyType, phases, conditions }] } (briefSummary EXCLUDED — use clinicaltrials_get_study) + honest _meta. HONESTY: countTotal=true is ALWAYS sent ⇒ totalAvailable = the EXACT filter-respecting UNCAPPED total (NEVER studies.length; a missing/non-number totalCount ⇒ schema_drift; a genuine 0 ⇒ 0, never null); pagination is an OPAQUE cursor (offset/nextOffset null; nextCursor = nextPageToken passed back verbatim as pageToken; terminal = token absent; a bad token ⇒ HTTP 400 THROWS). funderType is re-validated IN the handler — an UNLISTED value silently returns totalCount:0 at HTTP 200 (a fake-empty trap) ⇒ invalid_input pre-fetch (0 fetch); funderType is an OVERLAPPING facet (counts MUST NOT be summed). A MULTI-WORD query.term/sponsor/condition is AND-conjunctive (ALL tokens must co-occur — disclosed). A registered trial is NOT a federal award and leadSponsor.name is FREE TEXT (not a UEI) ⇒ a NOMINAL name match only (disclosed every response). Genuine-empty (totalCount:0, no token) ⇒ complete:true/total:0; a bad overallStatus/pageToken/nctId ⇒ HTTP 400/404 THROWS; an outage/5xx ⇒ THROWS (never a fake empty). Feed a row's nctId to clinicaltrials_get_study for the full record + briefSummary.",
|
|
6070
|
+
description: "Search federally-registered clinical-research studies with LEAD-SPONSOR / COLLABORATOR / FUNDING-SOURCE enrichment (keyless; clinicaltrials.gov/api/v2/studies). Filters: query.term (broad free-text), sponsor (→query.spons, fuzzy NAME search), condition (→query.cond), location (→query.locn), overallStatus (frozen 14-value enum), funderType (frozen 4-value enum nih/fed/industry/other), pageSize (1..1000), pageToken (OPAQUE cursor). Returns { studies:[{ nctId, briefTitle, orgStudyId, organization:{name,class}, leadSponsor:{name,class}, collaborators:[{name,class}], fundingClass, overallStatus, startDate, studyType, phases, conditions }] } (briefSummary EXCLUDED — use clinicaltrials_get_study) + honest _meta. HONESTY: countTotal=true ALWAYS sent → totalAvailable = EXACT uncapped total (NEVER studies.length; missing/non-number totalCount → schema_drift). Pagination is OPAQUE cursor (nextCursor = nextPageToken back as pageToken; terminal = token absent; bad token → HTTP 400 THROWS). funderType re-validated in handler — UNLISTED value silently returns totalCount:0 at HTTP 200 (fake-empty trap) → invalid_input pre-fetch. funderType is an OVERLAPPING facet (counts MUST NOT be summed). MULTI-WORD query.term/sponsor/condition is AND-conjunctive (all tokens must co-occur — disclosed). Registered trial is NOT a federal award; leadSponsor.name is FREE TEXT (not a UEI) → NOMINAL name match only — disclosed every response. Genuine totalCount:0 → complete:true/total:0; bad overallStatus/pageToken → HTTP 400/404 THROWS; outage/5xx THROWS. Feed nctId to clinicaltrials_get_study.",
|
|
6078
6071
|
inputSchema: ClinicaltrialsSearchStudiesInput,
|
|
6079
6072
|
handler: (input) => clinicaltrials.searchStudies(input),
|
|
6080
6073
|
}),
|
|
@@ -6096,8 +6089,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6096
6089
|
// every response. NO free-text ⇒ no tokenization.
|
|
6097
6090
|
defineTool({
|
|
6098
6091
|
name: "clinicaltrials_facet_counts",
|
|
6099
|
-
description:
|
|
6100
|
-
"Aggregate/statistical view: EXACT per-value STUDY counts over the WHOLE ClinicalTrials.gov registry for one or more whitelisted ENUM fields (keyless; clinicaltrials.gov/api/v2/stats/field/values) — the DISTRIBUTION sibling of clinicaltrials_search_studies (which gives the exact FILTERED total for a query). Input `fields`: 1..11 ENUM fields (deduped) — OverallStatus, StudyType, Phase, LeadSponsorClass (★ the funding-SOURCE-class distribution: NIH/FED/OTHER_GOV/INDUSTRY/OTHER/NETWORK/INDIV/UNKNOWN/AMBIG — richer than, and distinct from, the search tool's 4-value funderType filter), Sex, DesignAllocation, DesignPrimaryPurpose, DesignInterventionModel, DesignMasking, DesignObservationalModel, DesignTimePerspective. Module-built comma-joined into fields=<…> (NO raw passthrough). Returns { facets:[{ field, fieldPath, valueType, uniqueValuesCount, missingStudiesCount, returned, truncated, overlapping, values:[{ value, studiesCount }] }] } + honest _meta. HONESTY: each studiesCount/uniqueValuesCount is EXACT (typeof-checked to a NUMBER before num() — a non-number ⇒ schema_drift, NEVER a silent 0); a non-ENUM shape for a whitelisted field (e.g. a BOOLEAN {trueCount,falseCount}) ⇒ schema_drift (never read as empty). [M1] _meta.totalAvailable/returned count DISTINCT FIELD VALUES across the requested facet(s), NOT studies (a mandatory unit note points to facets[].values[].studiesCount / clinicaltrials_search_studies for a study count). These counts cover the ENTIRE registry and are NOT filterable — /stats/field/values rejects query.*/filter.*/countTotal/pageSize (HTTP 400) — a scope note cross-links the search tool for filtered totals. The returned<uniqueValuesCount⇒truncated invariant discloses the endpoint's hard 250-value cap the instant it binds (never for these v1 ENUM fields — all complete). Phase is ARRAY-valued (a study can carry several) ⇒ overlapping:true + a not-a-partition note (counts MUST NOT be summed); scalar fields partition the registry minus missingStudiesCount. A high missingStudiesCount ⇒ a note that the shown buckets cover a MINORITY of the registry. MANDATORY CAVEAT every response: a facet count is a distribution over trial REGISTRATIONS, NOT federal awards; LeadSponsorClass is the funding-SOURCE class, not a UEI-keyed award join. An unlisted field ⇒ invalid_input pre-fetch (0 fetch); a 404/400/5xx ⇒ THROWS (never a fake-empty distribution).",
|
|
6092
|
+
description: "Aggregate EXACT per-value study counts over the WHOLE ClinicalTrials.gov registry for 1..11 whitelisted ENUM fields (keyless; clinicaltrials.gov/api/v2/stats/field/values). Input `fields` (deduped): OverallStatus, StudyType, Phase, LeadSponsorClass (NIH/FED/OTHER_GOV/INDUSTRY/OTHER/NETWORK/INDIV/UNKNOWN/AMBIG — richer than the 4-value funderType in the search tool), Sex, DesignAllocation, DesignPrimaryPurpose, DesignInterventionModel, DesignMasking, DesignObservationalModel, DesignTimePerspective. Returns { facets:[{ field, fieldPath, valueType, uniqueValuesCount, missingStudiesCount, returned, truncated, overlapping, values:[{value, studiesCount}] }] } + honest _meta. HONESTY: each studiesCount/uniqueValuesCount is EXACT (typeof NUMBER — non-number → schema_drift); non-ENUM shape for a whitelisted field → schema_drift. _meta.totalAvailable/returned count DISTINCT FIELD VALUES, NOT studies — see facets[].values[].studiesCount / clinicaltrials_search_studies for study counts. Counts cover the ENTIRE registry and are NOT filterable — /stats/field/values rejects query.*/filter.*/pageSize (HTTP 400). returned<uniqueValuesCount → truncated (hard cap 250). Phase is ARRAY-valued (overlapping:true, MUST NOT sum counts); scalar fields partition the registry minus missingStudiesCount. High missingStudiesCount → buckets cover a MINORITY of the registry. MANDATORY CAVEAT: facet counts are distributions over trial REGISTRATIONS, NOT federal awards; LeadSponsorClass is the funding-SOURCE class, not a UEI-keyed award join. Unlisted field → invalid_input pre-fetch; 404/400/5xx → THROWS.",
|
|
6101
6093
|
inputSchema: ClinicaltrialsFacetCountsInput,
|
|
6102
6094
|
handler: (input) => clinicaltrials.facetCounts(input),
|
|
6103
6095
|
}),
|
|
@@ -6197,8 +6189,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6197
6189
|
// is a planned, separately-guarded addition).
|
|
6198
6190
|
defineTool({
|
|
6199
6191
|
name: "arcgis_hub_discover_datasets",
|
|
6200
|
-
description:
|
|
6201
|
-
"Discover ArcGIS Hub datasets by keyword — the SLED/GIS open-data layer that Socrata and CKAN do NOT cover (keyless; hub.arcgis.com/api/v3/datasets). Much US state/local/regional/tribal open data (GIS, infrastructure, permits, zoning, boundaries, procurement) is published on ArcGIS Hub. Input `query` (→q, REQUIRED, ≥2 non-whitespace chars — a broad whole-Hub scan is refused), `openDataOnly` (default TRUE → filter[openData]=true, the B2G-relevant designated-open-data subset; false broadens to all shared items), `limit` (1..100, def 20 → page[size]), `offset` (0-based → page[start]=offset+1). Returns { query, openDataOnly, datasets:[{ id, name, description, owner, orgName, source, region, type, sector, keywords, downloadable, hasApi, created, modified, landingPage, itemId }] } + honest _meta. ★PROVENANCE (the crux — a DIFFERENT trust posture from our other sources): ArcGIS Hub is a GLOBAL, OPEN publishing platform — results include NON-US and NON-GOVERNMENTAL publishers. This is a DISCOVERY aid, NOT a curated official-source allowlist (unlike socrata_query): the per-row owner/orgName/source/region are surfaced VERBATIM so you can VET the publisher before relying on the data, and the global-platform caveat rides EVERY response. DISCOVERY ONLY — metadata + links; to read rows follow the dataset on its own ArcGIS endpoint (a guarded row-query tool is a planned addition). HONESTY: totalAvailable = the EXACT Hub match count (meta.total, NEVER data.length — P1); pagination is a 0-based offset (nextOffset when more remain); every scalar null-never-empty, booleans null-preserving, counts null-never-0; a genuine no-match ⇒ complete:true/returned:0; a 429 ⇒ rate_limited / 5xx/timeout ⇒ upstream_unavailable THROWS (never a fake empty); a 200 non-JSON / non-array data ⇒ schema_drift.",
|
|
6192
|
+
description: "Discover ArcGIS Hub datasets by keyword — the SLED/GIS open-data layer that Socrata and CKAN do NOT cover (keyless; hub.arcgis.com/api/v3/datasets). Input `query` (→q, REQUIRED, ≥2 non-whitespace chars — broad whole-Hub scan refused), `openDataOnly` (default TRUE → filter[openData]=true, the B2G-relevant designated-open-data subset; false broadens to all shared items), `limit` (1..100, def 20 → page[size]), `offset` (0-based → page[start]=offset+1). Returns { query, openDataOnly, datasets:[{ id, name, description, owner, orgName, source, region, type, sector, keywords, downloadable, hasApi, created, modified, landingPage, itemId }] } + honest _meta. ★PROVENANCE (the crux): ArcGIS Hub is a GLOBAL, OPEN publishing platform — results include NON-US and NON-GOVERNMENTAL publishers. This is a DISCOVERY aid, NOT a curated official-source allowlist (unlike socrata_query): the per-row owner/orgName/source/region are surfaced VERBATIM so you can VET the publisher, and the global-platform caveat rides EVERY response. DISCOVERY ONLY — metadata + links; to read rows, arcgis_feature_query covers only its curated allowlist; other datasets must be followed on their own endpoint. HONESTY: totalAvailable = EXACT Hub match count (meta.total, NEVER data.length); pagination is 0-based offset; scalars null-never-empty, booleans null-preserving; genuine no-match → complete:true/returned:0; 429 → rate_limited / 5xx/timeout → upstream_unavailable THROWS; 200 non-JSON / non-array → schema_drift.",
|
|
6202
6193
|
inputSchema: ArcgisHubDiscoverInput,
|
|
6203
6194
|
handler: (input) => arcgisHub.discoverDatasets(input),
|
|
6204
6195
|
}),
|
|
@@ -6224,13 +6215,13 @@ export const TOOLS: ToolDef[] = [
|
|
|
6224
6215
|
}),
|
|
6225
6216
|
// ━━━ Bonfire (Euna) — keyless per-org open-opportunity RSS (SLED bids) ━━━
|
|
6226
6217
|
// SLED bid campaign. Thousands of US state/local govs on Bonfire expose a keyless
|
|
6227
|
-
// RSS of open opportunities. Ships a curated
|
|
6218
|
+
// RSS of open opportunities. Ships a curated 195-org live-verified seed directory
|
|
6228
6219
|
// (Bonfire's authoritative org API is auth-gated → out of bounds). Fixed-suffix
|
|
6229
6220
|
// SSRF (.bonfirehub.com). RSS = the complete open set (totalAvailable honest).
|
|
6230
6221
|
defineTool({
|
|
6231
6222
|
name: "bonfire_list_organizations",
|
|
6232
6223
|
description:
|
|
6233
|
-
"List US governments on the Bonfire (Euna) eProcurement platform — the directory for bonfire_search_opportunities (keyless). Bonfire hosts thousands of US state/local governments' open-bid portals, each with a keyless RSS feed. Filter the curated seed by `state` (2-letter) / `query` (case-insensitive name substring); `limit`(1..200)/`offset`. Output: { organizations:[{ org, name, state }] }. Feed a result's `org` to bonfire_search_opportunities. ★HONESTY: this is a CURATED, live-verified SEED of
|
|
6224
|
+
"List US governments on the Bonfire (Euna) eProcurement platform — the directory for bonfire_search_opportunities (keyless). Bonfire hosts thousands of US state/local governments' open-bid portals, each with a keyless RSS feed. Filter the curated seed by `state` (2-letter) / `query` (case-insensitive name substring); `limit`(1..200)/`offset`. Output: { organizations:[{ org, name, state }] }. Feed a result's `org` to bonfire_search_opportunities. ★HONESTY: this is a CURATED, live-verified SEED of 195 US orgs — Bonfire has NO keyless org-list API (its authoritative directory is auth-gated, out of bounds), and Euna markets up to ~900 US orgs, so the seed is PARTIAL (disclosed in _meta); probe `{slug}.bonfirehub.com/opportunities/rss` to extend. totalAvailable = the exact filtered seed count.",
|
|
6234
6225
|
inputSchema: BonfireListOrganizationsInput,
|
|
6235
6226
|
handler: (input) => bonfire.listOrganizations(input),
|
|
6236
6227
|
}),
|
|
@@ -6314,8 +6305,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6314
6305
|
// vintage enum is the (benchmark,vintage) UNION ([M2]); GEOIDs stay strings.
|
|
6315
6306
|
defineTool({
|
|
6316
6307
|
name: "census_geocode_address",
|
|
6317
|
-
description:
|
|
6318
|
-
"Resolve a one-line US address → its matched address(es) + the Census GEOGRAPHIES that drive set-aside / place-of-performance analysis (US Census Geocoder, keyless; geocoding.geo.census.gov/geocoder/geographies/onelineaddress) — the NEW territory/geospatial domain. Input `address` (≤500 chars), optional `benchmark` (default Public_AR_Current) / `vintage` (default Current_Current). Returns { matches:[{ matchedAddress, coordinates:{x,y}, tigerLineId, addressComponents, geographies:{ state, county, congressionalDistrict, censusTract, censusBlock, place, cbsaOrCsa, stateLegislativeUpper, stateLegislativeLower } }], matchCount, vintageResolved } + honest _meta. Each geography = { layerKey (the RAW vintage-versioned key, e.g. '119th Congressional Districts'), geoid (a STRING — leading zeros survive: '0102'), name }. HONESTY: genuine-empty (addressMatches:[]) ⇒ matchCount:0/complete:true (NOT an error; verify spelling + add city/state/ZIP); MULTIPLE matches are ALL surfaced (each with its own geographies) + a note; a historical vintage can return >1 layer per type (e.g. 111th+113th Congressional Districts with DISTINCT GEOIDs for a redistricted place) ⇒ BOTH surfaced (chosen + alternates[]) + a mandatory note (NEVER silently dropped); the resolved benchmark/vintage is echoed + a 'Current is a MOVING vintage' note; an invalid/missing benchmark/vintage ⇒ HTTP 400 THROWS (never a fake empty); an outage/5xx ⇒ THROWS. MANDATORY CAVEAT every response: these are a NOMINAL input, NOT an authoritative HUBZone / Opportunity-Zone / set-aside determination (those require SBA's HUBZone map / Treasury's OZ-tract list). Feed censusTract.geoid / county.geoid onward to those authoritative sources.",
|
|
6308
|
+
description: "Resolve a one-line US address → matched address(es) + the Census GEOGRAPHIES for set-aside / place-of-performance analysis (US Census Geocoder, keyless; geocoding.geo.census.gov/geocoder/geographies/onelineaddress). Input: `address` (≤500 chars), optional `benchmark` (default Public_AR_Current), `vintage` (default Current_Current). Returns { matches:[{ matchedAddress, coordinates:{x,y}, tigerLineId, addressComponents, geographies:{ state, county, congressionalDistrict, censusTract, censusBlock, place, cbsaOrCsa, stateLegislativeUpper, stateLegislativeLower } }], matchCount, vintageResolved }. Each geography = { layerKey (raw vintage-versioned key), geoid (STRING — leading zeros survive, e.g. '0102'), name }. HONESTY: genuine empty (addressMatches:[]) → matchCount:0/complete:true (NOT an error; verify spelling + add city/state/ZIP). MULTIPLE matches are ALL surfaced (each with its own geographies) + a note. A historical vintage can return >1 layer per type with DISTINCT GEOIDs → BOTH surfaced (chosen + alternates[]) + a mandatory note (NEVER silently dropped). The resolved benchmark/vintage is echoed + a 'Current is a MOVING vintage' note. Invalid/missing benchmark/vintage → HTTP 400 THROWS (never fake-empty); outage/5xx THROWS. MANDATORY CAVEAT every response: these are a NOMINAL input, NOT an authoritative HUBZone / Opportunity-Zone / set-aside determination — feed censusTract.geoid / county.geoid to SBA's HUBZone map / Treasury's OZ-tract list.",
|
|
6319
6309
|
inputSchema: CensusGeocodeAddressInput,
|
|
6320
6310
|
handler: (input) => census.geocodeAddress(input),
|
|
6321
6311
|
}),
|
|
@@ -6335,8 +6325,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6335
6325
|
// a negative number / never 0). The 2D-array body is parsed by header name.
|
|
6336
6326
|
defineTool({
|
|
6337
6327
|
name: "census_business_patterns",
|
|
6338
|
-
description:
|
|
6339
|
-
"Market sizing by NAICS × geography — establishments, employment, and annual payroll from the US Census County Business Patterns (CBP) API (api.census.gov/data/{year}/cbp). ★REQUIRES a free CENSUS_API_KEY: the Census Data API has NO keyless tier, so without the key this tool THROWS an honest config error (get one at https://api.census.gov/data/key_signup.html; call api_key_status to see every source's key requirement). Input: optional `naics` (2–6 digit NAICS-2017, e.g. '5415'; omit to aggregate all sectors), `geography` (us|state|county, default us; county REQUIRES `state`), `state` (2-digit FIPS, e.g. '06'), `year` (default '2023'), optional `limit` (client-side top-N; CBP has no server pagination). Returns { rows:[{ name, geoId, naicsCode, naicsLabel, establishments, employees, annualPayrollUsd, state }] } + honest _meta. HONESTY: establishments/employees are integer counts and annualPayrollUsd is annual US dollars (×1000 from the source's $1,000-unit PAYANN); large-negative suppression sentinels map to null — NEVER a negative number and NEVER 0 (a genuine 0 stays 0; note CBP primarily uses noise-infusion + suppression flags, surfaced as reported — see the tool's suppression note); geoId/naicsCode/state are STRINGS (leading zeros survive). CBP returns the COMPLETE geography set for the filter (no pagination) ⇒ totalAvailable = the row count, complete:true. A missing/invalid key ⇒ invalid_input (a 302 to the Missing-Key page); a header-only body ⇒ honest empty (returned:0); a 5xx ⇒ THROWS; a 200 non-JSON ⇒ schema_drift. The key rides ONLY in the &key= query param — never logged or echoed.",
|
|
6328
|
+
description: "Market sizing by NAICS × geography — establishments, employment, and annual payroll from the US Census County Business Patterns (CBP) API (api.census.gov/data/{year}/cbp). ★REQUIRES a free CENSUS_API_KEY: the Census Data API has NO keyless tier, so without the key this tool THROWS an honest config error (get one at https://api.census.gov/data/key_signup.html; call api_key_status to check). Input: optional `naics` (2–6 digit NAICS-2017, e.g. '5415'; omit to aggregate all sectors), `geography` (us|state|county, default us; county REQUIRES `state`), `state` (2-digit FIPS, e.g. '06'), `year` (default '2023'), optional `limit` (client-side top-N). Returns { rows:[{ name, geoId, naicsCode, naicsLabel, establishments, employees, annualPayrollUsd, state }] } + honest _meta. HONESTY: establishments/employees are integer counts and annualPayrollUsd is annual US dollars (×1000 from source's $1,000-unit PAYANN); large-negative suppression sentinels map to null — NEVER a negative number and NEVER 0 (genuine 0 stays 0; CBP primarily uses noise-infusion + suppression flags, surfaced as reported); geoId/naicsCode/state are STRINGS (leading zeros survive). CBP returns the COMPLETE geography set (no pagination) → totalAvailable = row count, complete:true. Missing/invalid key → invalid_input (302 to Missing-Key page); header-only body → honest empty; 5xx → THROWS; 200 non-JSON → schema_drift. Key rides ONLY in the &key= query param.",
|
|
6340
6329
|
inputSchema: CensusBusinessPatternsInput,
|
|
6341
6330
|
handler: (input) => censusEconomic.businessPatterns(input),
|
|
6342
6331
|
}),
|
|
@@ -6361,8 +6350,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6361
6350
|
// SPECIFIC ANNUAL VINTAGE (surfaced in a _meta note; update yearly).
|
|
6362
6351
|
defineTool({
|
|
6363
6352
|
name: "cms_medicare_provider_services",
|
|
6364
|
-
description:
|
|
6365
|
-
"Look up Medicare Part-B provider utilization — for a given provider (NPI) or state, the HCPCS services rendered, beneficiaries served, and submitted / Medicare-allowed / Medicare-paid amounts (CMS 'Medicare Physician & Other Practitioners — by Provider and Service', keyless; data.cms.gov data-API). The demand-side complement to nppes_lookup_provider (who providers ARE → what they BILL) for healthcare-market / competitor / teaming due-diligence. Input: `npi` (10-digit) OR `state` (2-letter) — at least ONE is REQUIRED (the table is 9.78M rows; an all-empty query is refused; providerType/hcpcsCode alone are NOT enough to scope); optional `providerType` (exact CMS specialty, e.g. 'Family Practice'), `hcpcsCode` (e.g. '97110', 'G0463'), `size` (1–100, default 25), `offset`. Returns { services:[{ npi, providerName, credentials, providerType, city, state, zip, hcpcsCode, hcpcsDescription, totalBeneficiaries, totalServices, avgSubmittedCharge, avgMedicareAllowed, avgMedicarePayment }] } + honest _meta. ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (…/data-viewer/stats → found_rows, e.g. VA=278254), NEVER the returned-rows length; if that count fails, totalAvailable is null + a disclosing note (never length-faked). offset/size pagination (hasMore = offset+returned < total). Aggregate/payment values are numeric-string → number|null (a genuine 0 stays 0, absent → null, never 0-faked); NPI/HCPCS/names are null-never-empty-string. A genuine no-match ⇒ honest empty (returned:0); a 4xx ⇒ invalid_input/not_found; a 5xx ⇒ THROWS; a 200 non-array/non-JSON ⇒ schema_drift. These are public PROVIDER-level AGGREGATE figures (no patient identifiers) for ONE annual vintage (the dataset year is disclosed in _meta) — a utilization snapshot, NOT a fraud/quality/fitness determination. KEYLESS — no key is sent.",
|
|
6353
|
+
description: "Medicare Part-B provider utilization — HCPCS services rendered, beneficiaries served, and submitted / Medicare-allowed / Medicare-paid amounts (CMS 'Medicare Physician & Other Practitioners — by Provider and Service', keyless; data.cms.gov data-API). Input: `npi` (10-digit) OR `state` (2-letter) — at least ONE is REQUIRED (the table is 9.78M rows; an all-empty query is refused; providerType/hcpcsCode alone are NOT enough to scope). Optional `providerType` (exact CMS specialty, e.g. 'Family Practice'), `hcpcsCode` (e.g. '97110'), `size` (1–100, def 25), `offset`. Returns { services:[{ npi, providerName, credentials, providerType, city, state, zip, hcpcsCode, hcpcsDescription, totalBeneficiaries, totalServices, avgSubmittedCharge, avgMedicareAllowed, avgMedicarePayment }] } + honest _meta. ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (…/data-viewer/stats → found_rows); if that count fails, totalAvailable is null + a disclosing note (never length-faked). hasMore = offset+returned < total. Aggregate/payment values: numeric-string → number|null (genuine 0 stays 0, absent → null, never 0-faked); NPI/HCPCS/names are null-never-empty-string. Genuine no-match → honest empty; 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array/non-JSON → schema_drift. PUBLIC PROVIDER-LEVEL AGGREGATE figures (no patient identifiers) for ONE annual vintage (dataset year disclosed in _meta) — utilization snapshot, NOT a fraud/quality/fitness determination.",
|
|
6366
6354
|
inputSchema: CmsMedicareProviderServicesInput,
|
|
6367
6355
|
handler: (input) => cmsUtilization.providerServices(input),
|
|
6368
6356
|
}),
|
|
@@ -6375,8 +6363,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6375
6363
|
// AND-combined server-side. REQUIRE state OR facilityName (never scanned unscoped).
|
|
6376
6364
|
defineTool({
|
|
6377
6365
|
name: "cms_hospital_compare",
|
|
6378
|
-
description:
|
|
6379
|
-
"Look up Medicare-certified hospitals by US state and/or facility-name fragment — location, type, ownership, emergency-services flag, and CMS star rating (CMS Hospital Compare 'Hospital General Information', keyless; data.cms.gov provider-data datastore-query API, ~5,432 hospitals). A healthcare-facility directory / market-map lane (WHERE hospitals are and HOW CMS rates them). Input: `state` (2-letter, EXACT) OR `facilityName` (a name fragment, case-insensitive substring/contains match) — at least ONE is REQUIRED (an all-empty query is refused; hospitalType alone is NOT enough to scope); optional `hospitalType` (substring, e.g. 'Acute', 'Critical Access'), `size` (1–100, default 25), `offset`. Returns { hospitals:[{ facilityId, facilityName, address, city, state, zip, county, phone, hospitalType, ownership, emergencyServices, overallRating }] } + honest _meta. ★HONESTY: totalAvailable is the response's EXACT top-level `count` for the filter set (VA=96), NEVER the returned-rows length; offset/size pagination (hasMore = offset+returned < count). overallRating is CMS's 1–5 star rating as a number; 'Not Available'/blank/non-numeric ⇒ null (NEVER 0). emergencyServices normalizes 'Yes'⇒true / 'No'⇒false / else null (never a fabricated false). IDs/names/addresses are null-never-empty-string. A genuine no-match ⇒ honest empty (returned:0); a 4xx ⇒ invalid_input/not_found; a 5xx ⇒ THROWS; a 200 non-array body or one missing count/results ⇒ schema_drift. Filters are applied SERVER-SIDE (AND-combined) — nothing is silently dropped. This is a summary star rating, NOT a clinical-quality or fitness determination. KEYLESS — no key is sent.",
|
|
6366
|
+
description: "Look up Medicare-certified hospitals by state and/or facility-name fragment — location, type, ownership, emergency-services flag, and CMS star rating (CMS Hospital Compare 'Hospital General Information', keyless; data.cms.gov provider-data datastore-query API, ~5,432 hospitals). Input: `state` (2-letter, EXACT) OR `facilityName` (case-insensitive substring) — at least ONE is REQUIRED (all-empty query refused; hospitalType alone is NOT enough to scope); optional `hospitalType` (substring, e.g. 'Acute', 'Critical Access'), `size` (1–100, default 25), `offset`. Returns { hospitals:[{ facilityId, facilityName, address, city, state, zip, county, phone, hospitalType, ownership, emergencyServices, overallRating }] } + honest _meta. ★HONESTY: totalAvailable is the response's EXACT top-level `count` for the filter set, NEVER the returned-rows length. overallRating is CMS's 1–5 star rating; 'Not Available'/blank/non-numeric → null (NEVER 0). emergencyServices normalizes 'Yes'→true / 'No'→false / else null. IDs/names/addresses are null-never-empty-string. Genuine no-match → honest empty; 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array or missing count/results → schema_drift. Filters applied SERVER-SIDE (AND-combined). Summary star rating, NOT a clinical-quality or fitness determination.",
|
|
6380
6367
|
inputSchema: CmsHospitalCompareInput,
|
|
6381
6368
|
handler: (input) => cmsHospital.hospitalCompare(input),
|
|
6382
6369
|
}),
|
|
@@ -6389,8 +6376,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6389
6376
|
// per dataset → coalesced (null if none — never empty-string, never fabricated).
|
|
6390
6377
|
defineTool({
|
|
6391
6378
|
name: "cms_facility_directory",
|
|
6392
|
-
description:
|
|
6393
|
-
"Look up Medicare/Medicaid-certified healthcare FACILITIES by type — nursing homes, home health agencies, hospices, or dialysis facilities — with their name, address, city, state, zip, and ownership (CMS provider-data, keyless; data.cms.gov datastore-query API, four datasets). A healthcare-facility directory / market-map lane that generalizes cms_hospital_compare beyond hospitals. Input: `facilityType` (REQUIRED enum — 'nursing_home' ~14,695 / 'home_health' ~12,460 / 'hospice' ~6,852 / 'dialysis' ~7,490; selects the dataset id via a constant map, the value never enters the URL path), optional `state` (2-letter, EXACT), `facilityName` (a name fragment, case-insensitive substring/contains match against the dataset's primary-name column), `size` (1–100, default 25), `offset`. Returns { facilities:[{ name, address, city, state, zip, facilityType, ownership }] } + honest _meta. ★HONESTY: totalAvailable is the response's EXACT top-level `count` for the filter set, NEVER the returned-rows length; offset/size pagination (hasMore = offset+returned < count). name/address/ownership column names DIFFER across the four datasets, so each is COALESCED over per-dataset candidates (name: provider_name/facility_name/legal_business_name; address: address/provider_address/address_line_1; ownership: ownership_type/type_of_ownership/profit_or_nonprofit) — a field absent in the chosen dataset is null (unknown), NEVER an empty string and NEVER fabricated. facilityType is echoed on each row. A genuine no-match ⇒ honest empty (returned:0); an invalid facilityType ⇒ invalid_input (blocked by the enum); a 4xx ⇒ invalid_input/not_found; a 5xx ⇒ THROWS; a 200 non-array body or one missing count/results ⇒ schema_drift. Filters are applied SERVER-SIDE (AND-combined) — nothing is silently dropped. This is a facility directory, NOT a clinical-quality or fitness determination. KEYLESS — no key is sent.",
|
|
6379
|
+
description: "Medicare/Medicaid-certified healthcare facilities by type — nursing homes, home health agencies, hospices, or dialysis facilities — with name, address, city, state, zip, and ownership (CMS provider-data, keyless; data.cms.gov datastore-query API, four datasets). Input: `facilityType` (REQUIRED enum — 'nursing_home' ~14,695 / 'home_health' ~12,460 / 'hospice' ~6,852 / 'dialysis' ~7,490; selects the dataset id via a constant map, the value NEVER enters the URL path). Optional `state` (2-letter, EXACT), `facilityName` (case-insensitive substring), `size` (1–100, def 25), `offset`. Returns { facilities:[{ name, address, city, state, zip, facilityType, ownership }] } + honest _meta. ★HONESTY: totalAvailable is the response's EXACT top-level `count` for the filter set — NEVER the returned-rows length; hasMore = offset+returned < count. name/address/ownership column names DIFFER across the four datasets → each is COALESCED over per-dataset candidates (name: provider_name/facility_name/legal_business_name; address: address/provider_address/address_line_1; ownership: ownership_type/type_of_ownership/profit_or_nonprofit) — a field absent in the chosen dataset is null (NEVER an empty string, NEVER fabricated). facilityType is echoed on each row. Filters applied SERVER-SIDE (AND-combined) — nothing silently dropped. Genuine no-match → honest empty; invalid facilityType → invalid_input (enum-blocked); 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array or missing count/results → schema_drift. NOT a clinical-quality or fitness determination.",
|
|
6394
6380
|
inputSchema: CmsFacilityDirectoryInput,
|
|
6395
6381
|
handler: (input) => cmsFacility.facilityDirectory(input),
|
|
6396
6382
|
}),
|
|
@@ -6402,8 +6388,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6402
6388
|
// scanned unscoped). The dataset UUID is a SPECIFIC ANNUAL VINTAGE (update yearly).
|
|
6403
6389
|
defineTool({
|
|
6404
6390
|
name: "cms_dmepos_suppliers",
|
|
6405
|
-
description:
|
|
6406
|
-
"Look up Medicare DMEPOS (Durable Medical Equipment, Devices & Supplies) SUPPLIERS — for a given supplier (NPI) or state, the supplier's identity plus aggregate Medicare figures: HCPCS codes billed, beneficiaries served, claims, services, and submitted / Medicare-allowed / Medicare-paid amounts (CMS 'Medicare DMEPOS — by Supplier', keyless; data.cms.gov data-API). The supply-side complement to cms_medicare_provider_services for healthcare-market / competitor / teaming due-diligence on equipment suppliers. Input: `npi` (10-digit) OR `state` (2-letter) — at least ONE is REQUIRED (an all-empty query is refused; the supplier table is never scanned unscoped); optional `size` (1–100, default 25), `offset`. Returns { suppliers:[{ npi, supplierName, credentials, entityType, city, state, zip, totalHcpcsCodes, totalBeneficiaries, totalClaims, totalServices, submittedCharges, medicareAllowed, medicarePayment }] } + honest _meta. ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (…/data-viewer/stats → found_rows), NEVER the returned-rows length; if that count fails, totalAvailable is null + a disclosing note (never length-faked). offset/size pagination (hasMore = offset+returned < total). Aggregate/payment values are numeric-string → number|null (a genuine 0 stays 0, absent → null, never 0-faked); NPI/entityType/names are null-never-empty-string; supplierName joins Last_Name_Org + First_Name ('Last, First' for individuals, the org name alone for organizations). A genuine no-match ⇒ honest empty (returned:0); a 4xx ⇒ invalid_input/not_found; a 5xx ⇒ THROWS; a 200 non-array/non-JSON ⇒ schema_drift. These are public SUPPLIER-level AGGREGATE figures (no patient identifiers) for ONE annual vintage (disclosed in _meta) — a utilization snapshot, NOT a fraud/quality/fitness determination. KEYLESS — no key is sent.",
|
|
6391
|
+
description: "Look up Medicare DMEPOS (Durable Medical Equipment) SUPPLIERS — supplier identity plus aggregate Medicare figures: HCPCS codes billed, beneficiaries served, claims, services, submitted / Medicare-allowed / Medicare-paid amounts (CMS 'Medicare DMEPOS — by Supplier', keyless; data.cms.gov data-API). The supply-side complement to cms_medicare_provider_services. Input: `npi` (10-digit) OR `state` (2-letter) — at least ONE is REQUIRED (all-empty query refused); optional `size` (1–100, default 25), `offset`. Returns { suppliers:[{ npi, supplierName, credentials, entityType, city, state, zip, totalHcpcsCodes, totalBeneficiaries, totalClaims, totalServices, submittedCharges, medicareAllowed, medicarePayment }] } + honest _meta. ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (…/data-viewer/stats → found_rows), NEVER the returned-rows length; if that count fails, totalAvailable is null + a disclosing note (never length-faked). Aggregate/payment values are numeric-string → number|null (genuine 0 stays 0, absent → null); NPI/entityType/names are null-never-empty-string; supplierName coalesces Last_Name_Org + First_Name. Genuine no-match → honest empty; 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array/non-JSON → schema_drift. PUBLIC SUPPLIER-LEVEL AGGREGATE figures (no patient identifiers) for ONE annual vintage — NOT a fraud/quality/fitness determination.",
|
|
6407
6392
|
inputSchema: CmsDmeposSuppliersInput,
|
|
6408
6393
|
handler: (input) => cmsSupplier.dmeposSuppliers(input),
|
|
6409
6394
|
}),
|
|
@@ -6415,8 +6400,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6415
6400
|
// P1 pattern; filter VALUES ride via URLSearchParams (bracket key + value encoded).
|
|
6416
6401
|
defineTool({
|
|
6417
6402
|
name: "cms_revoked_providers",
|
|
6418
|
-
description:
|
|
6419
|
-
"Search CMS's PUBLIC 'Revoked Medicare Providers & Suppliers' list — the legally-published register of Medicare enrollment revocations, with the revoked provider/supplier's identity, provider type, revocation reason, effective date, and re-enrollment-bar expiration (CMS 'Revoked Providers and Suppliers', keyless; data.cms.gov data-API, ~7,059 rows). A vetting / due-diligence lane in the SAME class as the OFAC / SAM-exclusions lists — for screening a counterparty before teaming or subcontracting. Input (ALL optional — the ~7K-row list is safe to page unfiltered): `npi` (10-digit → NPI), `state` (2-letter → STATE_CD, exact), `lastName` (→ LAST_NAME, exact), `size` (1–100, default 25), `offset`. Returns { revocations:[{ enrollmentId, npi, name, state, providerType, revocationReason, revocationEffectiveDate, reenrollmentBarExpiration }] } + honest _meta (which notes this is CMS's public revocation/exclusion list — a due-diligence signal, NOT a current-eligibility, guilt, or fitness determination). ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (…/data-viewer/stats → found_rows), NEVER the returned-rows length; if that count fails, totalAvailable is null + a disclosing note (never length-faked). offset/size pagination (hasMore = offset+returned < total). name coalesces ORG_NAME (organizations) else FIRST_NAME + LAST_NAME (individuals) — null if none, never a fabricated empty; NPI/reasons/dates are strings (null-never-empty-string). A genuine no-match ⇒ honest empty (returned:0); a 4xx ⇒ invalid_input/not_found; a 5xx ⇒ THROWS; a 200 non-array/non-JSON ⇒ schema_drift. KEYLESS — no key is sent.",
|
|
6403
|
+
description: "Search CMS's PUBLIC 'Revoked Medicare Providers & Suppliers' list — the legally-published register of Medicare enrollment revocations, with the revoked provider's identity, provider type, revocation reason, effective date, and re-enrollment-bar expiration (CMS 'Revoked Providers and Suppliers', keyless; data.cms.gov data-API, ~7,059 rows). A vetting lane in the same class as OFAC / SAM-exclusions lists. Input (ALL optional — the ~7K-row list is safe to page unfiltered): `npi` (10-digit), `state` (2-letter, EXACT), `lastName` (→ LAST_NAME, exact), `size` (1–100, default 25), `offset`. Returns { revocations:[{ enrollmentId, npi, name, state, providerType, revocationReason, revocationEffectiveDate, reenrollmentBarExpiration }] } + honest _meta (which notes this is CMS's public revocation list — a due-diligence signal, NOT a current-eligibility, guilt, or fitness determination). ★HONESTY: totalAvailable is the EXACT count from a SEPARATE stats sub-query (found_rows), NEVER the returned-rows length (null + note if count fails). name coalesces ORG_NAME else FIRST_NAME + LAST_NAME; NPI/reasons/dates null-never-empty. Genuine no-match → honest empty; 4xx → invalid_input/not_found; 5xx → THROWS; 200 non-array/non-JSON → schema_drift. KEYLESS.",
|
|
6420
6404
|
inputSchema: CmsRevokedProvidersInput,
|
|
6421
6405
|
handler: (input) => cmsSupplier.revokedProviders(input),
|
|
6422
6406
|
}),
|
|
@@ -6448,8 +6432,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6448
6432
|
// empty, never a throw. The optional key rides &api_key= ONLY.
|
|
6449
6433
|
defineTool({
|
|
6450
6434
|
name: "openfda_enforcement",
|
|
6451
|
-
description:
|
|
6452
|
-
"Search openFDA recall/enforcement records — drug/device/food product recalls with the recalling firm, product, reason, FDA classification (Class I/II/III), status, and geography (openFDA /{category}/enforcement.json; api.fda.gov). KEYLESS (an OPTIONAL free OPENFDA_API_KEY only RAISES the rate limit — keyless works at ~1000 requests/day; it NEVER throws for a missing key; get one at https://open.fda.gov/apis/authentication/; call api_key_status to see every source's key requirement). Input: `category` (drug|device|food, default drug), and STRUCTURED filters — `firm` (→recalling_firm), `product` (→product_description), `reason` (→reason_for_recall), `classification` (Class I|II|III), `status` (e.g. Ongoing/Terminated/Completed), `state` (2-letter, e.g. 'CA') — the tool safely assembles + escapes these into the openFDA search= Lucene string (NO raw passthrough — injection-safe), plus `limit` (1..100, default 25) and `skip` (offset ≥0). Returns { recalls:[{ recallingFirm, productDescription, reasonForRecall, classification, status, state, city, recallInitiationDate, recallNumber, voluntaryMandated, distributionPattern }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination via hasMore/nextOffset — never results.length); every scalar (dates included, recall_initiation_date is a YYYYMMDD string) is null-never-empty-string. ★A no-match query returns openFDA HTTP 404 NOT_FOUND ⇒ an HONEST EMPTY (returned:0, totalAvailable:0), NOT an error; a 400 syntax error ⇒ invalid_input surfacing openFDA's message; a 5xx ⇒ THROWS; a 200 non-JSON ⇒ schema_drift. The optional key rides ONLY the &api_key= query param — never logged or echoed.",
|
|
6435
|
+
description: "Search openFDA recall/enforcement records — drug/device/food product recalls with the recalling firm, product, reason, FDA classification (Class I/II/III), status, and geography (api.fda.gov/{category}/enforcement.json). KEYLESS — an OPTIONAL free OPENFDA_API_KEY only raises the rate limit; keyless works at ~1000 requests/day and NEVER throws for a missing key. Input: `category` (drug|device|food, default drug), structured filters — `firm` (→recalling_firm), `product` (→product_description), `reason` (→reason_for_recall), `classification` (Class I|II|III), `status` (e.g. Ongoing/Terminated), `state` (2-letter) — safely assembled + escaped into openFDA search= Lucene string (NO raw passthrough), plus `limit` (1..100, def 25) and `skip`. Returns { recalls:[{ recallingFirm, productDescription, reasonForRecall, classification, status, state, city, recallInitiationDate, recallNumber, voluntaryMandated, distributionPattern }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination via hasMore/nextOffset — never results.length). Every scalar (recall_initiation_date is a YYYYMMDD string) is null-never-empty-string. ★A no-match query returns openFDA HTTP 404 NOT_FOUND → HONEST EMPTY (returned:0, totalAvailable:0), NOT an error; 400 → invalid_input surfacing openFDA's message; 5xx → THROWS; 200 non-JSON → schema_drift. Optional key rides ONLY the &api_key= query param.",
|
|
6453
6436
|
inputSchema: OpenfdaEnforcementInput,
|
|
6454
6437
|
handler: (input) => openfda.enforcement(input),
|
|
6455
6438
|
}),
|
|
@@ -6463,15 +6446,13 @@ export const TOOLS: ToolDef[] = [
|
|
|
6463
6446
|
// empty, never a throw. The optional key rides &api_key= ONLY.
|
|
6464
6447
|
defineTool({
|
|
6465
6448
|
name: "openfda_device_clearances",
|
|
6466
|
-
description:
|
|
6467
|
-
"Search openFDA 510(k) DEVICE CLEARANCES — the FDA's premarket-notification (510(k)) clearances for medical devices, with the applicant/manufacturer, device name, clearance number (K-number), decision (date + description), clearance type, product code, advisory committee, and geography (openFDA /device/510k.json; api.fda.gov). KEYLESS (an OPTIONAL free OPENFDA_API_KEY only RAISES the rate limit — keyless works at ~1000 requests/day; it NEVER throws for a missing key; get one at https://open.fda.gov/apis/authentication/; call api_key_status to see every source's key requirement). Input: STRUCTURED filters — `applicant` (→applicant), `deviceName` (→device_name), `productCode` (→product_code), `clearanceType` (→clearance_type, e.g. Traditional/Special/Abbreviated), `kNumber` (→k_number, e.g. 'K123456'), `state` (2-letter, e.g. 'CA') — the tool safely assembles + escapes these into the openFDA search= Lucene string (NO raw passthrough — injection-safe), plus `limit` (1..100, default 25) and `skip` (offset ≥0). Returns { clearances:[{ applicant, deviceName, kNumber, decisionDate, decisionDescription, clearanceType, productCode, advisoryCommittee, state }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination via hasMore/nextOffset — never results.length); every scalar (dates included, decision_date is a YYYY-MM-DD string) is null-never-empty-string. ★A no-match query returns openFDA HTTP 404 NOT_FOUND ⇒ an HONEST EMPTY (returned:0, totalAvailable:0), NOT an error; a 400 syntax error ⇒ invalid_input surfacing openFDA's message; a 5xx ⇒ THROWS; a 200 non-JSON ⇒ schema_drift. The optional key rides ONLY the &api_key= query param — never logged or echoed.",
|
|
6449
|
+
description: "Search openFDA 510(k) DEVICE CLEARANCES — FDA premarket-notification clearances for medical devices, with the applicant/manufacturer, device name, clearance number (K-number), decision (date + description), clearance type, product code, advisory committee, and geography (openFDA /device/510k.json; api.fda.gov). KEYLESS (optional free OPENFDA_API_KEY only raises the rate limit — keyless works at ~1000 requests/day; NEVER throws for a missing key). Input: STRUCTURED filters — `applicant`, `deviceName`, `productCode`, `clearanceType` (e.g. Traditional/Special/Abbreviated), `kNumber` (e.g. 'K123456'), `state` (2-letter) — safely escaped into the openFDA search= Lucene string (NO raw passthrough), plus `limit` (1..100, default 25) and `skip` (offset ≥0). Returns { clearances:[{ applicant, deviceName, kNumber, decisionDate, decisionDescription, clearanceType, productCode, advisoryCommittee, state }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination via hasMore/nextOffset — never results.length); every scalar is null-never-empty-string; decision_date is a YYYY-MM-DD string. ★A no-match query returns openFDA HTTP 404 → HONEST EMPTY (returned:0/total:0), NOT an error; 400 → invalid_input surfacing openFDA's message; 5xx → THROWS; 200 non-JSON → schema_drift. Optional key rides ONLY in &api_key= param.",
|
|
6468
6450
|
inputSchema: OpenfdaDeviceClearancesInput,
|
|
6469
6451
|
handler: (input) => openfdaDevice.deviceClearances(input),
|
|
6470
6452
|
}),
|
|
6471
6453
|
defineTool({
|
|
6472
6454
|
name: "openfda_drug_approvals",
|
|
6473
|
-
description:
|
|
6474
|
-
"Search openFDA Drugs@FDA DRUG APPROVALS — FDA-approved drug applications (NDA/ANDA/BLA) with the sponsor, application number, each approved product (brand + generic/active-ingredient name, dosage form, route, marketing status), and the submission/approval history (openFDA /drug/drugsfda.json; api.fda.gov). Answers 'what drugs did sponsor X get approved, and which are still marketed' — pharma vendor product/approval intelligence. KEYLESS (an OPTIONAL free OPENFDA_API_KEY only RAISES the rate limit — keyless works at ~1000 requests/day; NEVER throws for a missing key; api_key_status lists every source's key requirement). Input: STRUCTURED filters — `sponsorName` (→sponsor_name), `brandName` (→products.brand_name), `activeIngredient` (→products.active_ingredients.name), `applicationNumber` (→application_number) — safely escaped into the openFDA search= Lucene string (NO raw passthrough — injection-safe), plus `limit` (1..100, default 25) and `skip` (offset ≥0). Returns { applications:[{ applicationNumber, sponsorName, products:[{ brandName, genericIngredients:[{name,strength}], dosageForm, route, marketingStatus }], submissions:[{ submissionType, submissionNumber, submissionStatus, submissionStatusDate, submissionClass }] }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination — never results.length); every scalar is null-never-empty-string; a 'Discontinued' marketingStatus is NOT an approval revocation (disclosed in _meta). ★A no-match query returns openFDA HTTP 404 NOT_FOUND ⇒ an HONEST EMPTY (returned:0/total:0), NOT an error; a 400 ⇒ invalid_input surfacing openFDA's message; a 5xx ⇒ THROWS; a 200 non-JSON ⇒ schema_drift. The optional key rides ONLY the &api_key= query param — never logged or echoed.",
|
|
6455
|
+
description: "Search openFDA Drugs@FDA DRUG APPROVALS — FDA-approved drug applications (NDA/ANDA/BLA) with the sponsor, application number, approved products (brand + generic/active-ingredient name, dosage form, route, marketing status), and submission/approval history (openFDA /drug/drugsfda.json; api.fda.gov). KEYLESS (optional free OPENFDA_API_KEY only raises the rate limit — ~1000 req/day keyless; NEVER throws for a missing key). Input: STRUCTURED filters — `sponsorName`, `brandName`, `activeIngredient`, `applicationNumber` — safely escaped into the openFDA search= Lucene string (NO raw passthrough), plus `limit` (1..100, default 25) and `skip` (offset ≥0). Returns { applications:[{ applicationNumber, sponsorName, products:[{ brandName, genericIngredients:[{name,strength}], dosageForm, route, marketingStatus }], submissions:[{ submissionType, submissionNumber, submissionStatus, submissionStatusDate, submissionClass }] }] } + honest _meta. HONESTY: totalAvailable is openFDA's EXACT meta.results.total (skip/limit pagination — never results.length); every scalar is null-never-empty-string; a 'Discontinued' marketingStatus is NOT an approval revocation (disclosed in _meta). ★A no-match query returns openFDA HTTP 404 → HONEST EMPTY (returned:0/total:0), NOT an error; 400 → invalid_input; 5xx → THROWS; 200 non-JSON → schema_drift. Optional key rides ONLY in &api_key= param.",
|
|
6475
6456
|
inputSchema: OpenfdaDrugApprovalsInput,
|
|
6476
6457
|
handler: (input) => openfdaDrugsfda.drugApprovals(input),
|
|
6477
6458
|
}),
|
|
@@ -6504,8 +6485,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6504
6485
|
// (disclosed) rather than silently fetch the entire dataset.
|
|
6505
6486
|
defineTool({
|
|
6506
6487
|
name: "cpsc_recalls",
|
|
6507
|
-
description:
|
|
6508
|
-
"Look up U.S. CPSC consumer-product RECALLS — the recall title, hazard description, remedy, affected products, manufacturers, retailers, injuries, and country of manufacture (CPSC SaferProducts /RestWebServices/Recall; www.saferproducts.gov). The consumer-goods / import product-safety lane alongside nhtsa_recalls (vehicles) and openfda (medical). KEYLESS — no API key is required or accepted. Inputs (ALL optional): `dateStart`/`dateEnd` (YYYY-MM-DD recall date range), `productName` (substring), `manufacturer` (substring), `recallNumber` (a specific CPSC recall number). Returns { recalls:[{ recallNumber, recallDate, title, description, url, products:[names], numberOfUnits, manufacturers:[names], retailers:[names], hazards:[descriptions], remedies:[descriptions], injuries:[names], manufacturerCountries:[names] }] } + honest _meta. HONESTY: the CPSC response is a bare array with NO count field and NO pagination — it returns the COMPLETE matching set, so totalAvailable = the number of returned recalls and complete:true (never a fabricated total). ★With NO filter given, results are bounded to a DEFAULT ~90-day recent window (RecallDateStart, disclosed in _meta.notes) rather than a silent whole-dataset fetch. An empty result ⇒ an HONEST EMPTY (returned:0), NOT an error; a 4xx ⇒ invalid_input; a 5xx/timeout ⇒ THROWS; a 200 non-JSON OR a non-array body ⇒ schema_drift. Nested arrays are flattened to name/description strings (an empty {} object is skipped, never fabricated); NumberOfUnits is free text kept as a string; dates are strings; every scalar is null-never-empty-string. Fixed host www.saferproducts.gov (SSRF-guarded); dates are ^\\d{4}-\\d{2}-\\d{2}$ and recallNumber is letters/digits/hyphen only.",
|
|
6488
|
+
description: "Look up U.S. CPSC consumer-product RECALLS — recall title, hazard description, remedy, affected products, manufacturers, retailers, injuries, and country of manufacture (CPSC SaferProducts /RestWebServices/Recall; www.saferproducts.gov). KEYLESS. Siblings: nhtsa_recalls (vehicles), openfda_enforcement. Inputs (ALL optional): `dateStart`/`dateEnd` (YYYY-MM-DD recall date range), `productName` (substring), `manufacturer` (substring), `recallNumber`. Returns { recalls:[{ recallNumber, recallDate, title, description, url, products:[names], numberOfUnits, manufacturers:[names], retailers:[names], hazards:[descriptions], remedies:[descriptions], injuries:[names], manufacturerCountries:[names] }] } + honest _meta. HONESTY: the CPSC response is a bare array with NO count field and NO pagination — it returns the COMPLETE matching set, so totalAvailable = number of returned recalls and complete:true (never a fabricated total). ★With NO filter given, results are bounded to a DEFAULT ~90-day recent window (RecallDateStart, disclosed in _meta.notes) rather than a silent whole-dataset fetch. Empty result → HONEST EMPTY (returned:0), NOT an error; 4xx → invalid_input; 5xx/timeout → THROWS; 200 non-JSON or non-array → schema_drift. Nested arrays are flattened to name/description strings; NumberOfUnits kept as a string; every scalar is null-never-empty-string. Fixed host (SSRF-guarded); dates are ^\\d{4}-\\d{2}-\\d{2}$ and recallNumber is letters/digits/hyphen only.",
|
|
6509
6489
|
inputSchema: CpscRecallsInput,
|
|
6510
6490
|
handler: (input) => cpsc.recalls(input),
|
|
6511
6491
|
}),
|
|
@@ -6527,8 +6507,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6527
6507
|
// DataValue is a comma-formatted string; suppression codes ((NA)/(D)/(NM)/(L)/*) → null.
|
|
6528
6508
|
defineTool({
|
|
6529
6509
|
name: "bea_regional_data",
|
|
6530
|
-
description:
|
|
6531
|
-
"Regional (county / state / MSA) economic data — GDP by industry and personal income — from the US Bureau of Economic Analysis (BEA) Regional Economic Accounts (apps.bea.gov/api/data, dataset 'Regional'). ★REQUIRES a free BEA_API_KEY: the BEA Data API has NO keyless tier, so without the key this tool THROWS an honest config error (get one at https://apps.bea.gov/API/signup/; call api_key_status to see every source's key requirement). Input: `tableName` (required, e.g. 'CAGDP2' county GDP by industry, 'SAGDP2N' state GDP, 'CAINC1'/'SAINC1' personal income), `geoFips` (required — 'STATE' for all states, a county FIPS like '06075', or an MSA code), `lineCode` (required — an integer industry line like '1', or 'ALL'), optional `year` ('LAST5' default, a 4-digit year, or 'ALL'), `frequency` ('A' annual default, or 'Q'). Returns { rows:[{ geoFips, geoName, timePeriod, lineCode, dataValue, unitOfMeasure, unitMult, noteRef }], notes:[{ noteRef, noteText }] } + honest _meta. ★HONESTY (the crux): a missing/invalid key — or ANY bad parameter — returns HTTP 200 carrying an Error object (NOT an HTTP error status); this is detected and surfaced as invalid_input carrying BEA's APIErrorDescription — NEVER a fake empty. dataValue is parsed from BEA's comma-formatted string ('1,234,567'→1234567); BEA suppression/not-available codes ((NA)/(D)/(NM)/(L)/*) map to null — NEVER 0 (a genuine 0 stays 0). unitMult (a power-of-10 multiplier) and unitOfMeasure are reported ALONGSIDE the raw dataValue — the value is NOT multiplied in (apply unitMult yourself). BEA returns the COMPLETE set for the filter (no pagination) ⇒ totalAvailable = the row count, complete:true; a genuine empty Data:[] ⇒ honest empty (returned:0); a 5xx ⇒ THROWS; a 200 non-JSON ⇒ schema_drift. The key rides ONLY in the UserID= query param — never logged or echoed.",
|
|
6510
|
+
description: "Regional (county / state / MSA) GDP by industry and personal income from the BEA Regional Economic Accounts (apps.bea.gov/api/data, dataset 'Regional'). ★REQUIRES a free BEA_API_KEY — NO keyless tier; without the key this tool THROWS an honest config error (get one at https://apps.bea.gov/API/signup/; call api_key_status to check). Input: `tableName` (required, e.g. 'CAGDP2' county GDP, 'SAGDP2N' state GDP, 'CAINC1'/'SAINC1' personal income), `geoFips` (required — 'STATE', county FIPS like '06075', or MSA code), `lineCode` (required — integer industry line or 'ALL'), optional `year` ('LAST5' default, 4-digit year, or 'ALL'), `frequency` ('A'/'Q'). Returns { rows:[{ geoFips, geoName, timePeriod, lineCode, dataValue, unitOfMeasure, unitMult, noteRef }], notes:[{ noteRef, noteText }] } + honest _meta. ★HONESTY: a missing/invalid key OR ANY bad parameter returns HTTP 200 carrying an Error object — detected and surfaced as invalid_input carrying BEA's APIErrorDescription, NEVER a fake empty. dataValue parsed from BEA's comma-formatted string ('1,234,567' → 1234567). BEA suppression codes (NA)/(D)/(NM)/(L)/* → null (NEVER 0; genuine 0 stays 0). unitMult and unitOfMeasure reported ALONGSIDE raw dataValue — NOT pre-multiplied in. BEA returns the COMPLETE filter result (no pagination) → complete:true. Genuine empty Data:[] → honest empty; 5xx → THROWS; 200 non-JSON → schema_drift. Key rides ONLY in the UserID= query param.",
|
|
6532
6511
|
inputSchema: BeaRegionalDataInput,
|
|
6533
6512
|
handler: (input) => bea.regionalData(input),
|
|
6534
6513
|
}),
|
|
@@ -6541,8 +6520,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6541
6520
|
// never-0; standardRate/isOconus are STRING booleans coerced to real booleans.
|
|
6542
6521
|
defineTool({
|
|
6543
6522
|
name: "gsa_perdiem_rates",
|
|
6544
|
-
description:
|
|
6545
|
-
"Look up GSA Federal Travel PER-DIEM rates — the max lodging + Meals & Incidental Expenses (M&IE) reimbursement ceilings for official U.S. government travel (api.gsa.gov /travel/perdiem/v2, keyed — DATA_GOV_API_KEY or the shared DEMO_KEY). Input: EITHER `city` (e.g. 'Washington') + `state` (2-letter, e.g. 'DC') OR `zip` (5-digit) — supplying BOTH, or NEITHER, ⇒ invalid_input with 0 fetch; optional `year` (default: the current U.S. federal fiscal year). Returns { rates:[{ city, county, state, zip, year, isOconus, standardRate, mealsUsd, monthlyLodgingUsd:[{ month (1-12), monthName, lodgingUsd }] }] } + honest _meta. HONESTY: lodgingUsd (the API's monthly `value`) is the MAX nightly lodging ceiling for that month — it VARIES SEASONALLY (hence a per-month array), and mealsUsd is the daily M&IE ceiling; both are integer US dollars, null-when-withheld (NEVER 0 — a genuine 0 is preserved). standardRate/isOconus are booleans coerced from the API's string 'true'/'false' (an unrecognized value ⇒ null, never a fabricated false); the months array is preserved AS-IS (never padded to 12). The API returns the COMPLETE rate set (no pagination) ⇒ totalAvailable = the row count, complete:true. A genuine no-match (rates:[]/rate:[]) ⇒ honest empty (returned:0); the API's `errors` field non-null ⇒ invalid_input carrying the message (never a fake empty); a 429 (DEMO_KEY ~10 req/hr, hit quickly) ⇒ rate_limited THROWS; a 5xx/timeout ⇒ upstream_unavailable THROWS; a 200 non-JSON ⇒ schema_drift. DEMO_KEY ~10 req/hr shared ceiling — set DATA_GOV_API_KEY (free at api.data.gov/signup) for 1000/hr. The key rides ONLY in the X-Api-Key header (never the URL/_meta).",
|
|
6523
|
+
description: "Look up GSA Federal Travel PER-DIEM rates — the max lodging + Meals & Incidental Expenses (M&IE) reimbursement ceilings for official U.S. government travel (api.gsa.gov /travel/perdiem/v2, keyed — DATA_GOV_API_KEY or the shared DEMO_KEY). Input: EITHER `city` + `state` (2-letter) OR `zip` (5-digit) — supplying BOTH or NEITHER → invalid_input with 0 fetch; optional `year` (default: current federal fiscal year). Returns { rates:[{ city, county, state, zip, year, isOconus, standardRate, mealsUsd, monthlyLodgingUsd:[{ month (1-12), monthName, lodgingUsd }] }] } + honest _meta. HONESTY: lodgingUsd is the MAX nightly lodging ceiling for that month — VARIES SEASONALLY (hence a per-month array); mealsUsd is the daily M&IE ceiling; both are integer US dollars, null-when-withheld (NEVER 0 — genuine 0 preserved). standardRate/isOconus are booleans coerced from the API's string 'true'/'false' (unrecognized → null, never fabricated false); months array preserved AS-IS (never padded to 12). API returns COMPLETE rate set (no pagination) → totalAvailable = row count, complete:true. Genuine no-match → honest empty; `errors` field non-null → invalid_input; 429 (DEMO_KEY ~10 req/hr) → rate_limited THROWS; set DATA_GOV_API_KEY (free, api.data.gov/signup) for 1000/hr. 5xx/timeout → upstream_unavailable THROWS; 200 non-JSON → schema_drift. Key rides ONLY in the X-Api-Key header.",
|
|
6546
6524
|
inputSchema: GsaPerdiemRatesInput,
|
|
6547
6525
|
handler: (input) => gsaPerdiem.perdiemRates(input),
|
|
6548
6526
|
}),
|
|
@@ -6562,8 +6540,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6562
6540
|
}),
|
|
6563
6541
|
defineTool({
|
|
6564
6542
|
name: "dol_get_dataset",
|
|
6565
|
-
description:
|
|
6566
|
-
"Fetch records from ONE US DOL dataset (apiprod.dol.gov /v4/get/{agency}/{endpoint}/json). ★REQUIRES a free DOL_API_KEY: the DOL DATA endpoint has NO keyless tier, so without the key this tool THROWS an honest config error (get one at https://dataportal.dol.gov/registration; the dataset CATALOG — dol_list_datasets — and agency list stay keyless). Input: `agency` (required — the `agencyAbbr` from dol_list_datasets, e.g. 'WHD', 'OSHA', 'ILAB'; rides the PATH, ^[A-Za-z0-9_]+$), `table` (required — the dataset's `apiUrl` endpoint from dol_list_datasets, e.g. 'Child_Labor_Report__2016_to_2022'; rides the PATH, ^[A-Za-z0-9_]+$), optional `limit` (default 10, max 100), `offset`, `filterField`+`filterValue` (a paired equality filter → a DOL filter_object), `fields` (best-effort column selection). Returns { records:[…verbatim dataset rows…] } + honest _meta. HONESTY: records are surfaced VERBATIM (the data-record envelope is key-gated and unverified, so field names/values are preserved as-is — a genuine 0 stays 0, a missing field stays null; the tool never coerces or fabricates). totalAvailable is a real count field ONLY when the response carries one, else null (an honest unknown — `returned` is NEVER passed off as the total); offset pagination (a full page ⇒ hasMore, page forward to confirm). A missing/invalid key (401/403) ⇒ invalid_input carrying the DOL_API_KEY guidance (never empty); a 400 ⇒ invalid_input; a genuine empty ⇒ honest empty (returned:0); a 429 ⇒ rate_limited THROWS (Retry-After honored); a 5xx/timeout ⇒ upstream_unavailable THROWS; a 200 non-JSON / no row array ⇒ schema_drift. The key rides ONLY in the X-API-KEY request header — never the URL / _meta / a log.",
|
|
6543
|
+
description: "Fetch records from ONE US DOL dataset (apiprod.dol.gov /v4/get/{agency}/{endpoint}/json). ★REQUIRES a free DOL_API_KEY: the DOL DATA endpoint has NO keyless tier — without the key this tool THROWS an honest config error (get one at https://dataportal.dol.gov/registration; dol_list_datasets stays keyless). Input: `agency` (required — the `agencyAbbr` from dol_list_datasets, e.g. 'WHD'/'OSHA'/'ILAB'; rides the PATH, ^[A-Za-z0-9_]+$), `table` (required — the dataset's `apiUrl` endpoint from dol_list_datasets; rides the PATH, ^[A-Za-z0-9_]+$), optional `limit` (def 10, max 100), `offset`, `filterField`+`filterValue` (paired equality filter), `fields` (best-effort column selection). Returns { records:[…verbatim dataset rows…] } + honest _meta. HONESTY: records are surfaced VERBATIM (field names/values preserved as-is — genuine 0 stays 0, missing field stays null; never coerced or fabricated). totalAvailable is a real count ONLY when the response carries one, else null (honest unknown — `returned` is NEVER passed off as the total). A full page → hasMore; page forward to confirm. Missing/invalid key (401/403) → invalid_input carrying DOL_API_KEY guidance (never empty); 400 → invalid_input; genuine empty → honest empty; 429 → rate_limited THROWS (Retry-After honored); 5xx/timeout → upstream_unavailable THROWS; 200 non-JSON / no row array → schema_drift. Key rides ONLY in the X-API-KEY request header — never URL/_meta.",
|
|
6567
6544
|
inputSchema: DolGetDatasetInput,
|
|
6568
6545
|
handler: (input) => dol.getDataset(input),
|
|
6569
6546
|
}),
|
|
@@ -6576,8 +6553,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6576
6553
|
// page-based pagination. income/expenses are null-or-decimal-string ⇒ null-never-0.
|
|
6577
6554
|
defineTool({
|
|
6578
6555
|
name: "lda_search_filings",
|
|
6579
|
-
description:
|
|
6580
|
-
"Search US Senate LDA (Lobbying Disclosure Act) filings — who is paid HOW MUCH to lobby WHICH federal agency on WHICH issue (lda.senate.gov/api/v1/filings, KEYLESS — anonymous access works; an optional free LDA_API_KEY only raises the rate limit). All inputs optional: `registrantName` (the lobbying firm/in-house filer), `clientName` (who it's for), `lobbyistName`, `filingYear` (4-digit), `filingType` (short code, e.g. 'Q1'/'RR'/'YE'), `agency` (NOTE: /filings/ has NO server-side agency filter — the LDA API silently ignores it, so it is reported in _meta.filtersDropped and NOT applied; government entities are nested per activity in lobbyingActivities[].governmentEntities), `issue` (specific lobbying issues text), `page` (1-based, default 1), `pageSize` (1..25, default 25). Returns { filings:[{ filingUuid, filingType, filingYear, filingPeriod, incomeUsd, expensesUsd, registrant, client, lobbyingActivities:[{ issueCode, description, governmentEntities:[names] }], documentUrl, postedDate, terminationDate }] } + honest _meta. HONESTY: totalAvailable is the API's REAL total match count (the corpus is ~1.95M filings) — NOT the rows on this page; pagination is page-based (pass the next page number when hasMore). incomeUsd/expensesUsd are parsed from the null-or-decimal-string income/expenses — null (not reported) ⇒ null, NEVER 0 (a genuine 0 stays 0); a filing reports EITHER income OR expenses, so the other is typically null. Missing lobbying_activities/government_entities ⇒ empty arrays (never fabricated). A genuine no-match (results:[]) ⇒ honest empty (returned:0); a 400 (bad filter) ⇒ invalid_input surfacing the API's message; a 429 ⇒ rate_limited THROWS (Retry-After honored, never routed around); a 5xx/timeout ⇒ upstream_unavailable THROWS; a 200 non-JSON / non-array results / non-number count ⇒ schema_drift. The optional key rides ONLY in the Authorization: Token header (never the URL/_meta).",
|
|
6556
|
+
description: "Search US Senate LDA (Lobbying Disclosure Act) filings — who is paid how much to lobby which federal agency on which issue (lda.senate.gov/api/v1/filings, KEYLESS — anonymous access works; optional free LDA_API_KEY only raises the rate limit). Filters (all optional): `registrantName` (the lobbying firm/in-house filer), `clientName`, `lobbyistName`, `filingYear` (4-digit), `filingType` (e.g. 'Q1'/'RR'/'YE'), `agency` (NOTE: /filings/ has NO server-side agency filter — the LDA API silently ignores it, so it is reported in _meta.filtersDropped and NOT applied; government entities are nested per activity in lobbyingActivities[].governmentEntities), `issue`, `page` (1-based), `pageSize` (1..25). Returns { filings:[{ filingUuid, filingType, filingYear, filingPeriod, incomeUsd, expensesUsd, registrant, client, lobbyingActivities:[{issueCode, description, governmentEntities:[names]}], documentUrl, postedDate, terminationDate }] } + honest _meta. HONESTY: totalAvailable is the API's REAL total match count (corpus ~1.95M filings) — NOT the rows on this page; pagination is page-based. incomeUsd/expensesUsd parsed from null-or-decimal-string — null (not reported) → null, NEVER 0 (genuine 0 stays 0); a filing reports EITHER income OR expenses, so the other is typically null. Missing lobbying_activities/government_entities → empty arrays. Genuine no-match → honest empty; 400 → invalid_input; 429 → rate_limited THROWS (Retry-After honored); 5xx/timeout THROWS; 200 non-JSON/non-array results/non-number count → schema_drift. Token rides ONLY in the Authorization: Token header.",
|
|
6581
6557
|
inputSchema: LdaSearchFilingsInput,
|
|
6582
6558
|
handler: (input) => lda.searchFilings(input),
|
|
6583
6559
|
}),
|
|
@@ -6591,8 +6567,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6591
6567
|
// (nextCursor extracted from `next`, host re-asserted). type=o FIXED.
|
|
6592
6568
|
defineTool({
|
|
6593
6569
|
name: "courtlistener_search_opinions",
|
|
6594
|
-
description:
|
|
6595
|
-
"Search US FEDERAL COURT OPINIONS (case law / litigation) via CourtListener (www.courtlistener.com/api/rest/v4/search, type=o). ★PROVENANCE: the DATA is US federal court PUBLIC RECORDS, but the API is CourtListener, run by the Free Law Project (a NON-PROFIT) — this is NOT a .gov API; CourtListener republishes these records KEYLESS because the .gov primary source (PACER) is PAYWALLED. KEYLESS (anonymous access works; an optional free COURTLISTENER_API_TOKEN only raises the rate limit; get one at https://www.courtlistener.com/help/api/rest/; call api_key_status to see every source's key requirement). All inputs optional: `query` (full-text → q), `court` (a court id, ^[a-z0-9]+$ — e.g. 'uscfc' US Court of Federal Claims for contract claims/bid protests, 'cafc' Federal Circuit for contract/patent appeals, 'scotus'), `dateFiledAfter`/`dateFiledBefore` (ISO ^\\d{4}-\\d{2}-\\d{2}$ → filed_after/filed_before), `natureOfSuit` (folded into the q query — no verified dedicated filter, disclosed in notes), `cursor` (opaque continuation — pass back _meta.nextCursor), `order` (→ order_by, default 'dateFiled desc'). Returns { opinions:[{ caseName, court, courtId, dateFiled, docketNumber, natureOfSuit, status, judge, citation, absoluteUrl }] } + honest _meta. HONESTY: totalAvailable is the API's REAL `count` (the total match count for the filter) — NOT the rows on this page; pagination is an OPAQUE CURSOR (offset/nextOffset are null/meaningless — pass _meta.nextCursor back as `cursor`; nextCursor:null/hasMore:false = last page). CourtListener v4 stops counting on deep cursor pages (count:null) ⇒ totalAvailable:null is DISCLOSED, never faked as results.length. dateFiled is a date STRING; citation may be an array/object ⇒ flattened to a safe string/string[] (never fabricated); judge/natureOfSuit/docketNumber are null when absent (never ''); absoluteUrl is the full https://www.courtlistener.com link. A genuine no-match (results:[]) ⇒ honest empty (returned:0); a 400 (bad param) ⇒ invalid_input surfacing the API's message; a 429 (unauth throttle) ⇒ rate_limited THROWS (Retry-After honored, never routed around); a 5xx/timeout ⇒ upstream_unavailable THROWS; a 200 non-JSON / non-array results / a count that is neither a number nor null ⇒ schema_drift; an off-host `next` is REFUSED (SSRF). The optional token rides ONLY in the Authorization: Token header (never the URL/_meta).",
|
|
6570
|
+
description: "Search US federal court opinions via CourtListener (www.courtlistener.com/api/rest/v4/search, type=o). ★PROVENANCE: DATA is US federal court PUBLIC RECORDS; the API is CourtListener (Free Law Project, NON-PROFIT) — NOT a .gov API; the .gov primary source (PACER) is PAYWALLED. KEYLESS (optional free COURTLISTENER_API_TOKEN only raises the rate limit). Filters (all optional): `query` (full-text → q), `court` (^[a-z0-9]+$ — e.g. 'uscfc' US Court of Federal Claims, 'cafc' Federal Circuit, 'scotus'), `dateFiledAfter`/`dateFiledBefore` (ISO YYYY-MM-DD), `natureOfSuit` (folded into q — no verified dedicated filter, disclosed in notes), `cursor` (opaque continuation), `order` (default 'dateFiled desc'). Returns { opinions:[{ caseName, court, courtId, dateFiled, docketNumber, natureOfSuit, status, judge, citation, absoluteUrl }] } + honest _meta. HONESTY: totalAvailable is the API's REAL `count` (total match count) — NOT rows on this page. Pagination is OPAQUE CURSOR (pass _meta.nextCursor back as `cursor`; nextCursor:null/hasMore:false = last page). CourtListener v4 stops counting on deep cursor pages (count:null) → totalAvailable:null DISCLOSED, never faked as results.length. dateFiled is a date STRING; citation → flattened to string/string[]; judge/natureOfSuit/docketNumber null when absent; absoluteUrl is the full CL link. Genuine no-match → honest empty; 400 → invalid_input; 429 → rate_limited (Retry-After honored); 5xx/timeout THROWS; 200 non-JSON/count not number or null → schema_drift; off-host `next` REFUSED (SSRF). Token rides ONLY in the Authorization: Token header.",
|
|
6596
6571
|
inputSchema: CourtlistenerSearchOpinionsInput,
|
|
6597
6572
|
handler: (input) => courtlistener.searchOpinions(input),
|
|
6598
6573
|
}),
|
|
@@ -6613,8 +6588,7 @@ export const TOOLS: ToolDef[] = [
|
|
|
6613
6588
|
}),
|
|
6614
6589
|
defineTool({
|
|
6615
6590
|
name: "nonprofit_financials",
|
|
6616
|
-
description:
|
|
6617
|
-
"Fetch ONE US tax-exempt nonprofit's IRS Form 990 profile + FINANCIALS by EIN via ProPublica Nonprofit Explorer (projects.propublica.org/nonprofits/api/v2/organizations/{ein}.json). ★PROVENANCE: the DATA is IRS Form 990 filings (federal tax-exempt public records) but the API is ProPublica Nonprofit Explorer, run by ProPublica (a NON-PROFIT newsroom) — NOT a .gov API; ProPublica republishes these records KEYLESS because the IRS has no clean query API. KEYLESS (no key). Input: `ein` (required — the Employer Identification Number, 1..9 digits ^\\d{1,9}$, e.g. '530196605' American National Red Cross; rides the URL path). Returns { organization:{ ein, name, address, city, state, zip, nteeCode, subsectionCode, rulingDate, statusCode }, filings:[{ taxYear, formType, revenueUsd, expensesUsd, assetsUsd, liabilitiesUsd, pdfUrl }] } + honest _meta. HONESTY: the four Form 990 figures (revenueUsd/expensesUsd/assetsUsd/liabilitiesUsd, from totrevenue/totfuncexpns/totassetsend/totliabend) ride null-never-0 coercion — a genuine reported 0 stays 0, an absent figure ⇒ null (NEVER 0-faked); ein/codes are strings; rulingDate is a date string. totalAvailable = filings.length (the COMPLETE Form 990 filing set from the one detail document — no pagination). An unknown EIN (HTTP 404) ⇒ not_found (NEVER a fabricated empty org); a 4xx ⇒ invalid_input; a 429 ⇒ rate_limited THROWS; a 5xx/timeout ⇒ upstream_unavailable THROWS; a 200 non-JSON / non-object organization / non-array filings_with_data ⇒ schema_drift. Data is IRS Form 990 data via ProPublica Nonprofit Explorer, disclosed in _meta.source and a note.",
|
|
6591
|
+
description: "Fetch ONE US tax-exempt nonprofit's IRS Form 990 profile + FINANCIALS by EIN via ProPublica Nonprofit Explorer (projects.propublica.org/nonprofits/api/v2/organizations/). ★PROVENANCE: the DATA is IRS Form 990 filings (federal tax-exempt public records) but the API is ProPublica Nonprofit Explorer (NON-PROFIT newsroom) — NOT a .gov API; ProPublica republishes these records KEYLESS (no key). Input: `ein` (required — the Employer Identification Number, 1..9 digits, e.g. '530196605' for American National Red Cross; rides the URL path). Returns { organization:{ ein, name, address, city, state, zip, nteeCode, subsectionCode, rulingDate, statusCode }, filings:[{ taxYear, formType, revenueUsd, expensesUsd, assetsUsd, liabilitiesUsd, pdfUrl }] } + honest _meta. HONESTY: the four Form 990 figures (revenueUsd/expensesUsd/assetsUsd/liabilitiesUsd) ride null-never-0 coercion — genuine reported 0 stays 0, absent → null (NEVER 0-faked); ein/codes are strings; rulingDate is a date string. totalAvailable = filings.length (COMPLETE filing set — no pagination). Unknown EIN (HTTP 404) → not_found (NEVER fabricated empty org); 4xx → invalid_input; 429 → rate_limited THROWS; 5xx/timeout → upstream_unavailable THROWS; 200 non-JSON / non-object org / non-array filings → schema_drift. Data source disclosed in _meta.source.",
|
|
6618
6592
|
inputSchema: NonprofitFinancialsInput,
|
|
6619
6593
|
handler: (input) => nonprofit.financials(input),
|
|
6620
6594
|
}),
|
|
@@ -6668,20 +6642,73 @@ async function main() {
|
|
|
6668
6642
|
},
|
|
6669
6643
|
});
|
|
6670
6644
|
|
|
6645
|
+
// ── Toolset resolution (MCP_SAM_GOV_TOOLSETS) ────────────────────
|
|
6646
|
+
const toolsetEnv = process.env.MCP_SAM_GOV_TOOLSETS;
|
|
6647
|
+
const allToolNames = TOOLS.map((t) => t.name);
|
|
6648
|
+
const tsResult = resolveToolsets(toolsetEnv, allToolNames);
|
|
6649
|
+
|
|
6650
|
+
// Warn on unknown names.
|
|
6651
|
+
for (const u of tsResult.unknown) {
|
|
6652
|
+
console.error(
|
|
6653
|
+
`[mcp-sam-gov] WARNING: unknown toolset name '${u}' in MCP_SAM_GOV_TOOLSETS — valid names: ${ALL_TOOLSET_NAMES.join(", ")}. Ignoring.`,
|
|
6654
|
+
);
|
|
6655
|
+
}
|
|
6656
|
+
if (tsResult.fellBack) {
|
|
6657
|
+
console.error(
|
|
6658
|
+
`[mcp-sam-gov] WARNING: no valid toolset names found in MCP_SAM_GOV_TOOLSETS='${toolsetEnv}' — falling back to all tools.`,
|
|
6659
|
+
);
|
|
6660
|
+
}
|
|
6661
|
+
|
|
6662
|
+
const loadedToolNames = tsResult.loaded;
|
|
6663
|
+
const isAllTools = tsResult.sets.length === 1 && tsResult.sets[0] === "all";
|
|
6664
|
+
|
|
6665
|
+
// Build instructions: add a one-liner about loaded/available toolsets when
|
|
6666
|
+
// not using the default (all). The default instructions stay byte-identical.
|
|
6667
|
+
// B2: unknown toolset names are reported here (not only stderr) because
|
|
6668
|
+
// stderr is invisible in Claude Desktop.
|
|
6669
|
+
const BASE_INSTRUCTIONS =
|
|
6670
|
+
"This server wraps US government open data (keyless-first). State/local dataset IDs (Socrata, CKAN, etc.) are listed in the MCP resource samgov://data-map/state-local. If a tool result looks wrong, a tool stays broken, or the user wants a capability this server lacks, help improve it: call the `feedback` tool — or use the `report` URL present on schema_drift / upstream_unavailable errors — to get a PREFILLED GitHub issue link, and offer it to the user to open and submit. Nothing is posted automatically; the user submits. Never include secrets or personal data in a report (the repo is public).";
|
|
6671
|
+
|
|
6672
|
+
let serverInstructions: string;
|
|
6673
|
+
if (isAllTools && tsResult.unknown.length === 0) {
|
|
6674
|
+
// Default: byte-identical to main.
|
|
6675
|
+
serverInstructions = BASE_INSTRUCTIONS;
|
|
6676
|
+
} else {
|
|
6677
|
+
const parts: string[] = [];
|
|
6678
|
+
if (!isAllTools) {
|
|
6679
|
+
// List loaded sets.
|
|
6680
|
+
parts.push(`Loaded toolsets: ${tsResult.sets.join(", ")}.`);
|
|
6681
|
+
// List other available sets with a 2–4 word hint each (NB5).
|
|
6682
|
+
const otherSets = ALL_TOOLSET_NAMES
|
|
6683
|
+
.filter((s) => !tsResult.sets.includes(s))
|
|
6684
|
+
.map((s) => `${s} (${TOOLSET_HINTS[s]})`);
|
|
6685
|
+
if (otherSets.length > 0) {
|
|
6686
|
+
parts.push(`Other available sets (set MCP_SAM_GOV_TOOLSETS to enable): ${otherSets.join("; ")}.`);
|
|
6687
|
+
}
|
|
6688
|
+
}
|
|
6689
|
+
// B2: report unknown names in instructions so they are visible in Claude Desktop.
|
|
6690
|
+
if (tsResult.unknown.length > 0) {
|
|
6691
|
+
parts.push(`Unknown toolset name(s) ignored: ${tsResult.unknown.join(", ")} — valid names: ${ALL_TOOLSET_NAMES.join(", ")}.`);
|
|
6692
|
+
}
|
|
6693
|
+
if (tsResult.fellBack) {
|
|
6694
|
+
parts.push("No valid toolset names found; fell back to all tools.");
|
|
6695
|
+
}
|
|
6696
|
+
serverInstructions = `${BASE_INSTRUCTIONS} ${parts.join(" ")}`;
|
|
6697
|
+
}
|
|
6698
|
+
|
|
6671
6699
|
const server = new Server(
|
|
6672
6700
|
{ name: SERVER_NAME, version: SERVER_VERSION },
|
|
6673
6701
|
{
|
|
6674
|
-
capabilities: { tools: {} },
|
|
6702
|
+
capabilities: { tools: {}, resources: {} },
|
|
6675
6703
|
// Surfaced to the agent at initialize. Tells it how to route real-usage
|
|
6676
6704
|
// friction back to the project WITHOUT the server ever posting anything.
|
|
6677
|
-
instructions:
|
|
6678
|
-
"This server wraps US government open data (keyless-first). If a tool result looks wrong, a tool stays broken, or the user wants a capability this server lacks, help improve it: call the `feedback` tool — or use the `report` URL present on schema_drift / upstream_unavailable errors — to get a PREFILLED GitHub issue link, and offer it to the user to open and submit. Nothing is posted automatically; the user submits. Never include secrets or personal data in a report (the repo is public).",
|
|
6705
|
+
instructions: serverInstructions,
|
|
6679
6706
|
},
|
|
6680
6707
|
);
|
|
6681
6708
|
|
|
6682
6709
|
server.setRequestHandler(ListToolsRequestSchema, async () => {
|
|
6683
6710
|
return {
|
|
6684
|
-
tools: TOOLS.map((t) => ({
|
|
6711
|
+
tools: filterToolsFor(TOOLS, loadedToolNames).map((t) => ({
|
|
6685
6712
|
name: t.name,
|
|
6686
6713
|
description: t.description,
|
|
6687
6714
|
inputSchema: zodToJsonSchema(t.inputSchema),
|
|
@@ -6690,8 +6717,60 @@ async function main() {
|
|
|
6690
6717
|
};
|
|
6691
6718
|
});
|
|
6692
6719
|
|
|
6720
|
+
// ── MCP Resources: state & local data map ──────────────────────────────
|
|
6721
|
+
const DATA_MAP_URI = "samgov://data-map/state-local";
|
|
6722
|
+
const DATA_MAP_NAME = "State & local data map";
|
|
6723
|
+
const DATA_MAP_DESCRIPTION =
|
|
6724
|
+
"Jurisdiction → verified tool call → row count for every allowlisted state/local Socrata, CKAN, Tableau and Open Checkbook dataset.";
|
|
6725
|
+
|
|
6726
|
+
server.setRequestHandler(ListResourcesRequestSchema, async () => {
|
|
6727
|
+
return {
|
|
6728
|
+
resources: [
|
|
6729
|
+
{
|
|
6730
|
+
uri: DATA_MAP_URI,
|
|
6731
|
+
name: DATA_MAP_NAME,
|
|
6732
|
+
description: DATA_MAP_DESCRIPTION,
|
|
6733
|
+
mimeType: "text/markdown",
|
|
6734
|
+
},
|
|
6735
|
+
],
|
|
6736
|
+
};
|
|
6737
|
+
});
|
|
6738
|
+
|
|
6739
|
+
server.setRequestHandler(ReadResourceRequestSchema, async (req) => {
|
|
6740
|
+
const { uri } = req.params;
|
|
6741
|
+
if (uri !== DATA_MAP_URI) {
|
|
6742
|
+
throw new Error(`Unknown resource: ${uri}`);
|
|
6743
|
+
}
|
|
6744
|
+
return {
|
|
6745
|
+
contents: [
|
|
6746
|
+
{
|
|
6747
|
+
uri: DATA_MAP_URI,
|
|
6748
|
+
mimeType: "text/markdown",
|
|
6749
|
+
text: renderDataMapMarkdown(),
|
|
6750
|
+
},
|
|
6751
|
+
],
|
|
6752
|
+
};
|
|
6753
|
+
});
|
|
6754
|
+
// ── end MCP Resources ──────────────────────────────────────────────────
|
|
6755
|
+
|
|
6693
6756
|
server.setRequestHandler(CallToolRequestSchema, async (req) => {
|
|
6694
6757
|
const { name, arguments: args } = req.params;
|
|
6758
|
+
// Check if the tool exists but is not loaded in the current toolset profile.
|
|
6759
|
+
const knownEntry = TOOLS.find((t) => t.name === name);
|
|
6760
|
+
if (knownEntry && !loadedToolNames.has(name)) {
|
|
6761
|
+
// B3: suggest the union of current sets + the needed set, so the user's
|
|
6762
|
+
// existing profile is not silently dropped.
|
|
6763
|
+
const envelope = toolNotLoadedEnvelope(name, tsResult.sets);
|
|
6764
|
+
return {
|
|
6765
|
+
content: [
|
|
6766
|
+
{
|
|
6767
|
+
type: "text" as const,
|
|
6768
|
+
text: JSON.stringify(envelope, null, 2),
|
|
6769
|
+
},
|
|
6770
|
+
],
|
|
6771
|
+
isError: true,
|
|
6772
|
+
};
|
|
6773
|
+
}
|
|
6695
6774
|
try {
|
|
6696
6775
|
const raw = await runTool(name, args ?? {}, sam);
|
|
6697
6776
|
// A handler may return either its raw domain object OR a MetaBundle
|
|
@@ -6736,9 +6815,11 @@ async function main() {
|
|
|
6736
6815
|
|
|
6737
6816
|
const transport = new StdioServerTransport();
|
|
6738
6817
|
await server.connect(transport);
|
|
6739
|
-
|
|
6740
|
-
|
|
6741
|
-
|
|
6818
|
+
const loadedCount = loadedToolNames.size;
|
|
6819
|
+
const profileNote = isAllTools
|
|
6820
|
+
? `${loadedCount} tools`
|
|
6821
|
+
: `${loadedCount}/${TOOLS.length} tools, toolsets: ${tsResult.sets.join(",")}`;
|
|
6822
|
+
console.error(`[mcp-sam-gov] v${SERVER_VERSION} listening on stdio (${profileNote}).`);
|
|
6742
6823
|
// Fire-and-forget, opt-out, fail-silent update notice (STDERR only, never stdout).
|
|
6743
6824
|
// Deliberately NOT awaited: it must never delay or affect the server (update-check.ts).
|
|
6744
6825
|
void checkForUpdate(SERVER_VERSION);
|