sourcelock 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1159 @@
1
+ """Provider identity / enrollment route.
2
+
3
+ Two authorities, one semantic ladder, deliberately kept apart:
4
+
5
+ * **NPPES NPI Registry API v2.1** (`npiregistry.cms.hhs.gov`) -- does this NPI
6
+ exist, and what identity does NPPES hold for it? Keyless, refreshed daily,
7
+ errors arrive as HTTP 200 bodies with an ``Errors`` array.
8
+ * **PECOS Public Provider Enrollment (PPEF)** on data.cms.gov -- was this
9
+ provider approved to bill Medicare FFS as of the quarterly snapshot's extract
10
+ date? Served through the stable "latest" alias UUID.
11
+
12
+ The ladder these tools must never collapse (CMS says the first rung verbatim:
13
+ "Issuance of an NPI does not ensure or validate that the Health Care Provider
14
+ is Licensed or Credentialed."):
15
+
16
+ NPI exists != Medicare-enrolled != credentialed / licensed / any-payer
17
+
18
+ ``provider.evidence_ladder`` therefore returns the facts SEPARATELY, each with
19
+ its own provenance, and refuses on principle to merge them into a score.
20
+
21
+ A PECOS answer is never one artifact. The rows come from the Data API; the
22
+ snapshot id and extract date that DATE those rows come from the dataset-resources
23
+ metadata, which is a separate document. So every PECOS route reports both -- url,
24
+ status, sha256, retrieval time, one entry per artifact -- and re-reads the
25
+ metadata after the rows: if the "latest" alias rolled to a new quarter mid-read,
26
+ the call fails rather than stamping rows with the wrong snapshot.
27
+
28
+ Deactivation semantics (verified 2026-08-01, full parse of the real report):
29
+ the monthly ``NPPES_Deactivated_NPI_Report_<MMDDYY>_V2.zip`` XLSX is the only
30
+ cumulative list of currently-deactivated NPIs and carries exactly two columns
31
+ (NPI, deactivation date, both text, MM/DD/YYYY); the main dissemination file
32
+ strips a deactivated NPI's row to those same two fields and never populates
33
+ the reason-code column. The Registry API does not serve deactivated NPIs at
34
+ all -- which is why a lookup miss warns about the report instead of concluding
35
+ the NPI never existed. Bulk-file mirroring itself is out of v1 scope; the
36
+ canaries watch the listing, not the gigabyte zips.
37
+ """
38
+
39
+ from __future__ import annotations
40
+
41
+ import hashlib
42
+ import json
43
+ import os
44
+ import re
45
+ from dataclasses import dataclass
46
+ from datetime import date
47
+ from typing import Any
48
+
49
+ from pydantic import BaseModel, ConfigDict, Field, field_validator
50
+
51
+ from ..http import FetchResult, SourceUnreachable, fetch, fetch_json
52
+ from ..interfaces import Canary, CanaryObservation, SourceAdapter, ToolResult, ToolSpec
53
+ from ..npi import NPI_PATTERN, validate_npi
54
+ from ..receipts import build_receipt
55
+ from ..schemas import CanaryStatus, SourceContract
56
+
57
+ SOURCE_ID = "provider"
58
+ #: 2 -- PECOS results carry per-artifact provenance and an explicit
59
+ #: completeness flag; enrollment_count is null rather than a partial number when
60
+ #: paging stopped early.
61
+ TRANSFORM_VERSION = "2"
62
+
63
+ # --------------------------------------------------------------------------
64
+ # Endpoints. Environment overrides exist so drift can be rehearsed against
65
+ # file:// fixtures (`file://` ignores query params, so the same code path
66
+ # serves live HTTP and local rehearsal).
67
+ # --------------------------------------------------------------------------
68
+
69
+ NPI_API_URL = "https://npiregistry.cms.hhs.gov/api/"
70
+ NPPES_FILES_URL = "https://download.cms.gov/nppes/NPI_Files.html"
71
+ PECOS_LATEST_ALIAS_UUID = "2457ea29-fc82-48b0-86ec-3b0755de7515"
72
+ PECOS_DATA_URL = f"https://data.cms.gov/data-api/v1/dataset/{PECOS_LATEST_ALIAS_UUID}/data"
73
+ PECOS_STATS_URL = f"{PECOS_DATA_URL}/stats"
74
+ PECOS_RESOURCES_URL = (
75
+ f"https://data.cms.gov/data-api/v1/dataset-resources/{PECOS_LATEST_ALIAS_UUID}"
76
+ )
77
+
78
+ _ENV_PREFIX = "HC_SOURCE_PROVIDER_"
79
+
80
+
81
+ def _url(name: str, default: str) -> str:
82
+ return os.environ.get(f"{_ENV_PREFIX}{name}", default)
83
+
84
+
85
+ def _npi_api_url() -> str:
86
+ return _url("NPI_API_URL", NPI_API_URL)
87
+
88
+
89
+ def _nppes_files_url() -> str:
90
+ return _url("NPPES_FILES_URL", NPPES_FILES_URL)
91
+
92
+
93
+ def _pecos_data_url() -> str:
94
+ return _url("PECOS_DATA_URL", PECOS_DATA_URL)
95
+
96
+
97
+ def _pecos_stats_url() -> str:
98
+ return _url("PECOS_STATS_URL", PECOS_STATS_URL)
99
+
100
+
101
+ def _pecos_resources_url() -> str:
102
+ return _url("PECOS_RESOURCES_URL", PECOS_RESOURCES_URL)
103
+
104
+
105
+ def _today() -> date:
106
+ """Cadence clock. Module-level so tests can freeze it."""
107
+ return date.today()
108
+
109
+
110
+ def _artifact_url(url: str) -> str:
111
+ """The URL of an artifact, without the query string.
112
+
113
+ Every request this route makes carries the caller's NPI in the query
114
+ (``number=`` on the Registry, ``filter[NPI]=`` on the Data API). Receipt
115
+ warnings and exception text are persisted, human-read surfaces, and the
116
+ no-echo rule applies to them exactly as it applies to a rejection message.
117
+ The sha256 still binds the exact bytes; the URL only names the artifact they
118
+ came from.
119
+ """
120
+ return url.split("?", 1)[0]
121
+
122
+
123
+ def _provenance(fetch_result: FetchResult, source: str) -> dict[str, Any]:
124
+ """One artifact's provenance line: which source, which bytes, when, what upstream said.
125
+
126
+ An answer assembled from several artifacts gets one of these per artifact.
127
+ ``retrieved_at`` is the fetch's own stamp, never the clock at receipt time.
128
+ """
129
+ return {
130
+ "source": source,
131
+ "url": _artifact_url(fetch_result.url),
132
+ "raw_sha256": fetch_result.sha256,
133
+ "upstream_status": fetch_result.status,
134
+ "retrieved_at": fetch_result.retrieved_at.isoformat(),
135
+ }
136
+
137
+
138
+ def _provenance_lines(artifacts: list[dict[str, Any]]) -> list[str]:
139
+ """Render artifacts as receipt warnings.
140
+
141
+ ``Receipt`` has one url, one status and one sha256, so a multi-artifact
142
+ answer cannot express its provenance in the typed fields. ``data`` carries
143
+ the structured list -- but a miss returns ``data=None``, and a miss is
144
+ exactly when someone asks which artifacts were read. These lines are the
145
+ record that survives it.
146
+ """
147
+ return [
148
+ "PROVENANCE_ARTIFACT: source={source} url={url} status={upstream_status} "
149
+ "sha256={raw_sha256} retrieved_at={retrieved_at}".format(**a)
150
+ for a in artifacts
151
+ ]
152
+
153
+
154
+ #: A boringly stable individual NPI used as the API-contract anchor.
155
+ ANCHOR_NPI = "1003000126"
156
+
157
+ NPI_API_VERSION = "2.1"
158
+
159
+ CONTRACT = SourceContract(
160
+ source_id=SOURCE_ID,
161
+ authority_url="https://npiregistry.cms.hhs.gov/api/?version=2.1",
162
+ fallback_url=None,
163
+ license_notes=(
164
+ "US Government works, public domain (https://www.usa.gov/government-works): NPPES "
165
+ "dissemination data and the PECOS Public Provider Enrollment files are freely "
166
+ "redistributable. One caveat: the NUCC Healthcare Provider Taxonomy DESCRIPTIONS the "
167
+ "Registry API returns are NUCC-licensed -- store and redistribute taxonomy CODES; "
168
+ "treat descriptions as display-time convenience and do not vendor the NUCC code-set "
169
+ "CSV. No CPT/AMA content exists on this route."
170
+ ),
171
+ cadence=(
172
+ "NPI Registry API: daily refresh from NPPES; NPPES bulk files: weekly incrementals + "
173
+ "monthly full + monthly deactivation report (V2 grammar since 2026-03-03); "
174
+ "PECOS PPEF: quarterly (R/P3M)"
175
+ ),
176
+ effective_date_semantics=(
177
+ "provider.lookup_npi: effective_from is the NPI's NPPES enumeration date and "
178
+ "effective_to is open while the record is active -- the API refreshes daily and has "
179
+ "no release stamp. provider.pecos_enrollments and provider.evidence_ladder: "
180
+ "effective_from is the PPEF quarterly snapshot's extract date (from the dated "
181
+ "filename, e.g. 2026.07.17); enrollment facts speak as of that extract date only."
182
+ ),
183
+ invariants=[
184
+ "NPI Registry API v2.1 answers a valid active NPI with result_count/results and no "
185
+ "top-level Errors key (errors always arrive as HTTP 200 bodies)",
186
+ "download.cms.gov/nppes lists monthly full, weekly incremental, and monthly "
187
+ "deactivation files in the _V2 naming grammar (V1 retired 2026-03-03)",
188
+ "the PECOS PPEF latest-alias dataset serves >2.5M enrollment rows and all five "
189
+ "relational components (Enrollment, Additional NPIs, Reassignment, Practice "
190
+ "Location, Secondary Specialty)",
191
+ "a deactivated NPI is served only by the cumulative monthly Deactivated NPI Report "
192
+ "(two columns: NPI, deactivation date), never by the Registry API",
193
+ ],
194
+ )
195
+
196
+ # --------------------------------------------------------------------------
197
+ # Non-claims. These are the part reviewers read; each states one rung of the
198
+ # ladder the result must never be stretched across.
199
+ # --------------------------------------------------------------------------
200
+
201
+ _NC_NPI_NOT_CREDENTIALED = (
202
+ "NPI_PRESENT_IS_NOT_CREDENTIALED: an NPI record proves identity registration only. CMS "
203
+ "states verbatim that issuance of an NPI does not ensure or validate that the provider "
204
+ "is licensed or credentialed; NPPES license numbers are self-reported and unvalidated."
205
+ )
206
+ _NC_NPI_NOT_ENROLLED = (
207
+ "DOES_NOT_PROVE_MEDICARE_ENROLLMENT: an NPPES record says nothing about PECOS enrollment, "
208
+ "Medicare billing privileges, or participation in any payer's network."
209
+ )
210
+ _NC_NPI_CURRENCY = (
211
+ "DOES_NOT_PROVE_CURRENCY: served by the NPI Registry API, which refreshes daily from "
212
+ "NPPES with no release stamp; deactivations land in the monthly Deactivated NPI Report "
213
+ "before this answer necessarily reflects them."
214
+ )
215
+ _NC_PECOS_NOT_ALL_PAYER = (
216
+ "PECOS_ACTIVE_IS_NOT_ALL_PAYER_ACTIVE: a PPEF row means approved to bill Medicare "
217
+ "fee-for-service (or an 855O on file) as of the quarterly extract date -- it says "
218
+ "nothing about Medicaid, commercial payers, network participation, or today."
219
+ )
220
+ _NC_PECOS_NOT_CREDENTIALED = (
221
+ "DOES_NOT_PROVE_CREDENTIALING: no public CMS dataset asserts credentialing, privileging, "
222
+ "or licensure in good standing; Medicare enrollment is not that."
223
+ )
224
+ _NC_PECOS_MAIN_FILE_ONLY = (
225
+ "MAIN_FILE_ONLY: absence from the PPEF main enrollment extract is not proof of "
226
+ "non-enrollment -- multi-NPI enrollments can carry an NPI only in the Additional NPIs "
227
+ "sub-file, which this API filter does not search."
228
+ )
229
+ _NC_PECOS_COUNT_INCOMPLETE = (
230
+ "ENROLLMENT_COUNT_IS_NOT_COMPLETE: paging stopped at this route's page bound while the "
231
+ "Data API was still returning full pages, so the enrollments listed are a prefix of what "
232
+ "this NPI has, not all of it. enrollment_count is null for exactly that reason; "
233
+ "enrollment_count_at_least is what was actually read."
234
+ )
235
+ _NC_NOT_A_SCORE = (
236
+ "NOT_A_SCORE: NPI_EXISTS and PECOS_ENROLLED are independent facts with independent "
237
+ "vintages. SourceLock reports them separately and never merges them into a combined "
238
+ "verdict; any downstream aggregation is the consumer's claim, not this receipt's."
239
+ )
240
+
241
+ NPI_NON_CLAIMS = [_NC_NPI_NOT_CREDENTIALED, _NC_NPI_NOT_ENROLLED, _NC_NPI_CURRENCY]
242
+ PECOS_NON_CLAIMS = [_NC_PECOS_NOT_ALL_PAYER, _NC_PECOS_NOT_CREDENTIALED, _NC_PECOS_MAIN_FILE_ONLY]
243
+ LADDER_NON_CLAIMS = [
244
+ _NC_NOT_A_SCORE,
245
+ _NC_NPI_NOT_CREDENTIALED,
246
+ _NC_PECOS_NOT_ALL_PAYER,
247
+ _NC_NPI_NOT_ENROLLED,
248
+ _NC_PECOS_MAIN_FILE_ONLY,
249
+ ]
250
+
251
+
252
+ # --------------------------------------------------------------------------
253
+ # Typed public parameters
254
+ # --------------------------------------------------------------------------
255
+
256
+
257
+ class NpiParams(BaseModel):
258
+ model_config = ConfigDict(extra="forbid")
259
+
260
+ npi: str = Field(
261
+ pattern=NPI_PATTERN,
262
+ description="10-digit National Provider Identifier (check digit is validated).",
263
+ )
264
+
265
+ @field_validator("npi")
266
+ @classmethod
267
+ def _check_digit(cls, v: str) -> str:
268
+ # hc_source.npi is the one validator. This route used to carry its own
269
+ # copy while leie accepted any ten digits, so a transposed NPI was
270
+ # refused here and screened confidently there.
271
+ return validate_npi(v)
272
+
273
+
274
+ # --------------------------------------------------------------------------
275
+ # NPPES NPI Registry API
276
+ # --------------------------------------------------------------------------
277
+
278
+
279
+ def _fetch_npi(npi: str) -> tuple[FetchResult, dict[str, Any]]:
280
+ result, payload = fetch_json(
281
+ _npi_api_url(), params={"version": NPI_API_VERSION, "number": npi}
282
+ )
283
+ if not isinstance(payload, dict):
284
+ raise SourceUnreachable(
285
+ _artifact_url(result.url), "NPI Registry response was not a JSON object"
286
+ )
287
+ errors = payload.get("Errors")
288
+ if errors:
289
+ codes = ",".join(str(e.get("number", "?")) for e in errors)
290
+ raise SourceUnreachable(
291
+ _artifact_url(result.url),
292
+ f"NPI Registry returned API error(s) {codes} with HTTP 200 -- the v{NPI_API_VERSION} "
293
+ "contract has changed (see https://npiregistry.cms.hhs.gov/api-page)",
294
+ status=result.status,
295
+ )
296
+ return result, payload
297
+
298
+
299
+ def _as_date(value: Any) -> date | None:
300
+ if not isinstance(value, str):
301
+ return None
302
+ try:
303
+ return date.fromisoformat(value)
304
+ except ValueError:
305
+ return None
306
+
307
+
308
+ def _map_npi_record(record: dict[str, Any]) -> dict[str, Any]:
309
+ basic = record.get("basic", {})
310
+ return {
311
+ "npi": record.get("number"),
312
+ "enumeration_type": record.get("enumeration_type"),
313
+ "status": basic.get("status"),
314
+ "enumeration_date": basic.get("enumeration_date"),
315
+ "last_updated": basic.get("last_updated"),
316
+ "name": {
317
+ "first_name": basic.get("first_name"),
318
+ "last_name": basic.get("last_name"),
319
+ "credential": basic.get("credential"),
320
+ "organization_name": basic.get("organization_name"),
321
+ },
322
+ "taxonomies": [
323
+ {
324
+ "code": t.get("code"),
325
+ "desc": t.get("desc"),
326
+ "primary": t.get("primary"),
327
+ "state": t.get("state"),
328
+ "license": t.get("license"),
329
+ }
330
+ for t in record.get("taxonomies", [])
331
+ ],
332
+ "addresses": [
333
+ {
334
+ "purpose": a.get("address_purpose"),
335
+ "city": a.get("city"),
336
+ "state": a.get("state"),
337
+ "postal_code": a.get("postal_code"),
338
+ }
339
+ for a in record.get("addresses", [])
340
+ ],
341
+ }
342
+
343
+
344
+ _MISS_WARNING = (
345
+ "The requested NPI was not found by the NPI Registry API (v2.1). The API does not serve "
346
+ "deactivated NPIs: before concluding this NPI never existed, check the monthly NPPES "
347
+ "Deactivated NPI Report (cumulative XLSX of every currently-deactivated NPI with its "
348
+ "deactivation date) at https://download.cms.gov/nppes/NPI_Files.html."
349
+ )
350
+
351
+
352
+ def _lookup_npi(params: NpiParams) -> ToolResult:
353
+ result, payload = _fetch_npi(params.npi)
354
+ warnings: list[str] = []
355
+ data: dict[str, Any] | None = None
356
+ effective_from: date | None = None
357
+
358
+ if payload.get("result_count", 0) >= 1 and payload.get("results"):
359
+ record = payload["results"][0]
360
+ data = _map_npi_record(record)
361
+ effective_from = _as_date(data.get("enumeration_date"))
362
+ if data.get("status") != "A":
363
+ warnings.append(
364
+ "The requested NPI is not in active status "
365
+ f"({data.get('status')!r}) in the NPI Registry."
366
+ )
367
+ else:
368
+ warnings.append(_MISS_WARNING)
369
+
370
+ return ToolResult(
371
+ data=data,
372
+ receipt=build_receipt(
373
+ contract=CONTRACT,
374
+ route="provider.lookup_npi",
375
+ fetch=result,
376
+ source_version=f"npi-registry-api-{NPI_API_VERSION}",
377
+ transform_version=TRANSFORM_VERSION,
378
+ effective_from=effective_from,
379
+ warnings=warnings,
380
+ non_claims=NPI_NON_CLAIMS,
381
+ ),
382
+ )
383
+
384
+
385
+ # --------------------------------------------------------------------------
386
+ # PECOS PPEF (data.cms.gov Data API, latest-alias UUID)
387
+ # --------------------------------------------------------------------------
388
+
389
+ _PPEF_EXTRACT_RE = re.compile(r"PPEF_Enrollment_Extract_(\d{4})\.(\d{2})\.(\d{2})\.csv")
390
+
391
+ #: The five relational components a healthy PPEF release publishes.
392
+ _PPEF_COMPONENTS = {
393
+ "enrollment": "PPEF_Enrollment_Extract_",
394
+ "additional_npis": "PPEF_Additional_NPIs_",
395
+ "reassignment": "PPEF_Reassignment_Extract_",
396
+ "practice_location": "PPEF_Practice_Location_Extract_",
397
+ "secondary_specialty": "PPEF_Secondary_Specialty_Extract_",
398
+ }
399
+
400
+
401
+ def _ppef_snapshot(resources: Any) -> tuple[str, date | None, list[str]]:
402
+ """Extract (snapshot id, extract date, missing components) from dataset-resources."""
403
+ entries = resources.get("data", []) if isinstance(resources, dict) else []
404
+ urls = [e.get("downloadURL", "") for e in entries if isinstance(e, dict)]
405
+
406
+ snapshot, extract_date = "unknown", None
407
+ for u in urls:
408
+ m = _PPEF_EXTRACT_RE.search(u)
409
+ if m:
410
+ snapshot = ".".join(m.groups())
411
+ extract_date = date(int(m.group(1)), int(m.group(2)), int(m.group(3)))
412
+ break
413
+
414
+ missing = sorted(
415
+ name
416
+ for name, pattern in _PPEF_COMPONENTS.items()
417
+ if not any(pattern in u for u in urls)
418
+ )
419
+ return snapshot, extract_date, missing
420
+
421
+
422
+ def _fetch_ppef_version() -> tuple[FetchResult, str, date | None, list[str]]:
423
+ result, resources = fetch_json(_pecos_resources_url())
424
+ snapshot, extract_date, missing = _ppef_snapshot(resources)
425
+ return result, snapshot, extract_date, missing
426
+
427
+
428
+ #: Rows per Data API page. What this route has always asked for, kept well under
429
+ #: the API's own size cap.
430
+ _PPEF_PAGE_SIZE = 200
431
+
432
+ #: Hard stop on paging: 25 pages is 5,000 rows, two orders of magnitude past the
433
+ #: busiest real NPI. The bound is not a guess at how many enrollments a provider
434
+ #: has -- it is the thing that stops an upstream which keeps handing back full
435
+ #: pages from spinning this process forever. Reaching it is reported as an
436
+ #: incomplete read, never as a count.
437
+ _PPEF_MAX_PAGES = 25
438
+
439
+
440
+ def _fetch_ppef_rows(npi: str) -> tuple[list[dict[str, Any]], list[FetchResult], bool]:
441
+ """Read every PPEF row for one NPI. Returns (rows, page fetches, complete).
442
+
443
+ This route used to request one page of 200 and present ``len(rows)`` as the
444
+ enrollment count. A provider with more than 200 enrollments got a wrong
445
+ number that read as a complete one, which is the specific failure this
446
+ product exists not to produce. So: page to exhaustion, and when the page
447
+ bound stops us first, say so instead of returning a number.
448
+ """
449
+ rows: list[dict[str, Any]] = []
450
+ fetches: list[FetchResult] = []
451
+ seen: set[str] = set()
452
+
453
+ for page in range(_PPEF_MAX_PAGES):
454
+ result, batch = fetch_json(
455
+ _pecos_data_url(),
456
+ params={
457
+ "filter[NPI]": npi,
458
+ "size": _PPEF_PAGE_SIZE,
459
+ "offset": page * _PPEF_PAGE_SIZE,
460
+ },
461
+ )
462
+ if not isinstance(batch, list):
463
+ raise SourceUnreachable(
464
+ _artifact_url(result.url),
465
+ "PECOS Data API did not return the expected bare JSON array",
466
+ )
467
+ if result.sha256 in seen:
468
+ # An API that ignores `offset` answers every page with page 1, and
469
+ # paging would then multiply one page into a confident wrong count.
470
+ # Identical bytes at a different offset means we cannot page this
471
+ # source, so we refuse rather than count.
472
+ raise SourceUnreachable(
473
+ _artifact_url(result.url),
474
+ "PECOS Data API returned identical bytes for two different offsets, so it "
475
+ "is not honouring the offset parameter and its rows cannot be paged",
476
+ status=result.status,
477
+ )
478
+ seen.add(result.sha256)
479
+ fetches.append(result)
480
+ rows.extend(r for r in batch if isinstance(r, dict))
481
+ if len(batch) < _PPEF_PAGE_SIZE:
482
+ return rows, fetches, True
483
+
484
+ return rows, fetches, False
485
+
486
+
487
+ def _map_enrollment(row: dict[str, Any]) -> dict[str, Any]:
488
+ return {
489
+ "enrollment_id": row.get("ENRLMT_ID"),
490
+ "pac_id": row.get("PECOS_ASCT_CNTL_ID"),
491
+ "provider_type_code": row.get("PROVIDER_TYPE_CD"),
492
+ "provider_type": row.get("PROVIDER_TYPE_DESC"),
493
+ "state": row.get("STATE_CD"),
494
+ "multiple_npi_flag": row.get("MULTIPLE_NPI_FLAG"),
495
+ "first_name": row.get("FIRST_NAME"),
496
+ "last_name": row.get("LAST_NAME"),
497
+ "org_name": row.get("ORG_NAME"),
498
+ }
499
+
500
+
501
+ # The NPI is NOT interpolated. It is caller-supplied text, this string is
502
+ # persisted into the evidence receipt, and the no-echo rule has no carve-out
503
+ # for identifiers that happen to be public. The snapshot is ours to name.
504
+ _PECOS_MISS_WARNING = (
505
+ "The requested NPI has no row in the PPEF main enrollment extract ({snapshot}). That "
506
+ "means no "
507
+ "approved Medicare FFS enrollment was recorded under this NPI at the extract date -- but "
508
+ "the NPI could still ride a multi-NPI enrollment listed only in the Additional NPIs "
509
+ "sub-file, and enrollment can postdate the snapshot. Absence here is not proof of "
510
+ "non-enrollment."
511
+ )
512
+
513
+ _PECOS_PROVENANCE_WARNING = (
514
+ "MULTI_ARTIFACT_PROVENANCE: {n} upstream artifact(s) produced this answer. The receipt's "
515
+ "raw_sha256 and upstream_status describe only the first PPEF data page; source_version "
516
+ "and effective_from come from the dataset-resources metadata, which is a different "
517
+ "document read at a different moment. Every artifact is listed below as a "
518
+ "PROVENANCE_ARTIFACT line, and in data.provenance when data is not null. The metadata was "
519
+ "re-read after the rows and still reported snapshot {snapshot}, so the rows and the "
520
+ "version claimed here come from one release."
521
+ )
522
+
523
+ _PECOS_TRUNCATION_WARNING = (
524
+ "ENROLLMENT_COUNT_TRUNCATED: the PPEF Data API was still returning full pages after "
525
+ "{pages} pages of {size} rows, so paging stopped at this route's bound and the enrollment "
526
+ "list is a prefix. enrollment_count is null; enrollment_count_at_least is {read}. Do not "
527
+ "read this result as a complete enrollment count."
528
+ )
529
+
530
+ _PECOS_ROLLOVER = (
531
+ "the PPEF latest alias reported snapshot {before} before the row read and {after} after "
532
+ "it, so a new quarterly snapshot landed mid-call. Refusing rather than dating rows to a "
533
+ "release they did not come from; re-run to read the new snapshot whole"
534
+ )
535
+
536
+
537
+ @dataclass(frozen=True)
538
+ class _PpefRead:
539
+ """One PECOS answer plus every artifact that had to be read to produce it."""
540
+
541
+ rows: list[dict[str, Any]]
542
+ #: The first data page. What the receipt's single raw_sha256 attests to.
543
+ rows_fetch: FetchResult
544
+ snapshot: str
545
+ extract_date: date | None
546
+ missing: list[str]
547
+ #: False when the page bound stopped the read before the rows ran out.
548
+ complete: bool
549
+ artifacts: list[dict[str, Any]]
550
+
551
+
552
+ def _read_ppef(npi: str) -> _PpefRead:
553
+ """Read the PPEF rows for one NPI and bind them to the snapshot that dates them.
554
+
555
+ The version and the data are two different fetches of two different
556
+ documents through a "latest" alias that moves quarterly. Reading the
557
+ metadata, then the rows, then the metadata again is what makes the receipt's
558
+ ``source_version`` a claim about the bytes it hashed rather than a claim
559
+ about whatever the alias happened to point at first.
560
+ """
561
+ before, snapshot, extract_date, missing = _fetch_ppef_version()
562
+ rows, page_fetches, complete = _fetch_ppef_rows(npi)
563
+ after, snapshot_after, _extract_after, _missing_after = _fetch_ppef_version()
564
+
565
+ if snapshot_after != snapshot:
566
+ # SourceUnreachable because that is the honest classification: the
567
+ # source was not read coherently, so this run does not know what it
568
+ # says. Every caller already handles it.
569
+ raise SourceUnreachable(
570
+ _artifact_url(after.url),
571
+ _PECOS_ROLLOVER.format(before=snapshot, after=snapshot_after),
572
+ status=after.status,
573
+ )
574
+
575
+ artifacts = [_provenance(before, f"pecos-ppef-{snapshot}:dataset-resources")]
576
+ artifacts += [
577
+ _provenance(f, f"pecos-ppef-{snapshot}:data-page-{i}")
578
+ for i, f in enumerate(page_fetches, start=1)
579
+ ]
580
+ artifacts.append(_provenance(after, f"pecos-ppef-{snapshot}:dataset-resources-recheck"))
581
+
582
+ return _PpefRead(
583
+ rows=rows,
584
+ rows_fetch=page_fetches[0],
585
+ snapshot=snapshot,
586
+ extract_date=extract_date,
587
+ missing=missing,
588
+ complete=complete,
589
+ artifacts=artifacts,
590
+ )
591
+
592
+
593
+ def _enrollment_count_fields(read: _PpefRead) -> dict[str, Any]:
594
+ """The count, or an explicit refusal to state one.
595
+
596
+ ``enrollment_count`` is null rather than a plausible integer when the read
597
+ was truncated: a consumer that reads only that field gets nothing it can
598
+ mistake for a total, and ``enrollment_count_at_least`` says what was
599
+ actually seen.
600
+ """
601
+ if read.complete:
602
+ return {"enrollments_are_complete": True, "enrollment_count": len(read.rows)}
603
+ return {
604
+ "enrollments_are_complete": False,
605
+ "enrollment_count": None,
606
+ "enrollment_count_at_least": len(read.rows),
607
+ }
608
+
609
+
610
+ def _pecos_warnings(read: _PpefRead) -> list[str]:
611
+ """Provenance, component inventory, and truncation -- in that order."""
612
+ warnings = [
613
+ _PECOS_PROVENANCE_WARNING.format(n=len(read.artifacts), snapshot=read.snapshot),
614
+ *_provenance_lines(read.artifacts),
615
+ ]
616
+ if read.missing:
617
+ warnings.append(
618
+ f"PPEF release is missing expected component(s): {', '.join(read.missing)}."
619
+ )
620
+ if not read.complete:
621
+ warnings.append(
622
+ _PECOS_TRUNCATION_WARNING.format(
623
+ pages=_PPEF_MAX_PAGES, size=_PPEF_PAGE_SIZE, read=len(read.rows)
624
+ )
625
+ )
626
+ return warnings
627
+
628
+
629
+ def _pecos_enrollments(params: NpiParams) -> ToolResult:
630
+ read = _read_ppef(params.npi)
631
+ warnings = _pecos_warnings(read)
632
+ non_claims = list(PECOS_NON_CLAIMS)
633
+ if not read.complete:
634
+ non_claims.append(_NC_PECOS_COUNT_INCOMPLETE)
635
+
636
+ if read.rows:
637
+ data: dict[str, Any] | None = {
638
+ "npi": params.npi,
639
+ "snapshot": read.snapshot,
640
+ **_enrollment_count_fields(read),
641
+ "enrollments": [_map_enrollment(r) for r in read.rows],
642
+ "provenance": read.artifacts,
643
+ }
644
+ else:
645
+ data = None
646
+ warnings.append(
647
+ _PECOS_MISS_WARNING.format(snapshot=read.snapshot)
648
+ )
649
+
650
+ return ToolResult(
651
+ data=data,
652
+ receipt=build_receipt(
653
+ contract=CONTRACT,
654
+ route="provider.pecos_enrollments",
655
+ fetch=read.rows_fetch,
656
+ source_version=f"pecos-ppef-{read.snapshot}",
657
+ transform_version=TRANSFORM_VERSION,
658
+ effective_from=read.extract_date,
659
+ warnings=warnings,
660
+ non_claims=non_claims,
661
+ ),
662
+ )
663
+
664
+
665
+ # --------------------------------------------------------------------------
666
+ # The evidence ladder: distinct facts, never a merged verdict
667
+ # --------------------------------------------------------------------------
668
+
669
+
670
+ def _evidence_ladder(params: NpiParams) -> ToolResult:
671
+ npi_fetch, npi_payload = _fetch_npi(params.npi)
672
+ read = _read_ppef(params.npi)
673
+
674
+ npi_exists = bool(npi_payload.get("result_count", 0) >= 1 and npi_payload.get("results"))
675
+ npi_detail: dict[str, Any]
676
+ if npi_exists:
677
+ basic = npi_payload["results"][0].get("basic", {})
678
+ npi_detail = {
679
+ "status": basic.get("status"),
680
+ "enumeration_type": npi_payload["results"][0].get("enumeration_type"),
681
+ "enumeration_date": basic.get("enumeration_date"),
682
+ }
683
+ else:
684
+ npi_detail = {
685
+ "note": (
686
+ "not served by the NPI Registry API -- never issued, or deactivated "
687
+ "(deactivated NPIs appear only in the monthly Deactivated NPI Report)"
688
+ )
689
+ }
690
+
691
+ pecos_enrolled = bool(read.rows)
692
+ pecos_detail = {
693
+ "snapshot": read.snapshot,
694
+ **_enrollment_count_fields(read),
695
+ "enrollment_ids": [r.get("ENRLMT_ID") for r in read.rows],
696
+ }
697
+
698
+ npi_artifacts = [_provenance(npi_fetch, f"npi-registry-api-{NPI_API_VERSION}")]
699
+
700
+ data = {
701
+ "npi": params.npi,
702
+ "facts": {
703
+ "NPI_EXISTS": {
704
+ "value": npi_exists,
705
+ "detail": npi_detail,
706
+ # A list even when there is one artifact: a fact's provenance is
707
+ # everything that produced it, and PECOS_ENROLLED genuinely takes
708
+ # several. One shape for both keeps a consumer from learning the
709
+ # single-artifact form and breaking on the other.
710
+ "provenance": npi_artifacts,
711
+ },
712
+ "PECOS_ENROLLED": {
713
+ "value": pecos_enrolled,
714
+ "detail": pecos_detail,
715
+ "provenance": read.artifacts,
716
+ },
717
+ },
718
+ "ladder_note": (
719
+ "Distinct facts, deliberately not merged: NPI existence (NPPES, daily refresh) "
720
+ "and Medicare FFS enrollment (PECOS PPEF, quarterly snapshot) are different "
721
+ "assertions with different vintages."
722
+ ),
723
+ }
724
+
725
+ warnings = [
726
+ "COMPOSITE_EVIDENCE: this receipt's raw_sha256/upstream_status cover the first PECOS "
727
+ "PPEF data page only. Every artifact behind either fact -- the NPPES response, the "
728
+ "PPEF dataset-resources metadata, each data page -- is listed below as a "
729
+ "PROVENANCE_ARTIFACT line and inside data.facts.*.provenance. The two sources have "
730
+ "different vintages.",
731
+ *_provenance_lines([*npi_artifacts, *read.artifacts]),
732
+ ]
733
+ if read.missing:
734
+ warnings.append(
735
+ f"PPEF release is missing expected component(s): {', '.join(read.missing)}."
736
+ )
737
+ non_claims = list(LADDER_NON_CLAIMS)
738
+ if not read.complete:
739
+ warnings.append(
740
+ _PECOS_TRUNCATION_WARNING.format(
741
+ pages=_PPEF_MAX_PAGES, size=_PPEF_PAGE_SIZE, read=len(read.rows)
742
+ )
743
+ )
744
+ non_claims.append(_NC_PECOS_COUNT_INCOMPLETE)
745
+
746
+ return ToolResult(
747
+ data=data,
748
+ receipt=build_receipt(
749
+ contract=CONTRACT,
750
+ route="provider.evidence_ladder",
751
+ fetch=read.rows_fetch,
752
+ source_version=(
753
+ f"npi-registry-api-{NPI_API_VERSION}+pecos-ppef-{read.snapshot}"
754
+ ),
755
+ transform_version=TRANSFORM_VERSION,
756
+ effective_from=read.extract_date,
757
+ warnings=warnings,
758
+ non_claims=non_claims,
759
+ ),
760
+ )
761
+
762
+
763
+ # --------------------------------------------------------------------------
764
+ # Canary 1: NPI Registry API contract
765
+ # --------------------------------------------------------------------------
766
+
767
+
768
+ def _npi_schema_hash(payload: dict[str, Any]) -> str:
769
+ """Hash the response SHAPE only: key sets, never values."""
770
+ results = payload.get("results") or [{}]
771
+ record = results[0]
772
+ shape = {
773
+ "top": sorted(payload.keys()),
774
+ "result": sorted(record.keys()),
775
+ "basic": sorted(record.get("basic", {}).keys()),
776
+ "taxonomy": sorted((record.get("taxonomies") or [{}])[0].keys()),
777
+ "address": sorted((record.get("addresses") or [{}])[0].keys()),
778
+ }
779
+ return hashlib.sha256(
780
+ json.dumps(shape, sort_keys=True, separators=(",", ":")).encode()
781
+ ).hexdigest()
782
+
783
+
784
+ def _observe_npi_api() -> CanaryObservation:
785
+ result, payload = fetch_json(
786
+ _npi_api_url(), params={"version": NPI_API_VERSION, "number": ANCHOR_NPI}
787
+ )
788
+ if not isinstance(payload, dict):
789
+ raise SourceUnreachable(result.url, "NPI Registry response was not a JSON object")
790
+
791
+ errors = payload.get("Errors")
792
+ if errors:
793
+ codes = ",".join(sorted(str(e.get("number", "?")) for e in errors))
794
+ return CanaryObservation(
795
+ value=f"api-error:{codes}",
796
+ upstream_status=result.status,
797
+ note="registry answered HTTP 200 with an Errors body",
798
+ )
799
+
800
+ if payload.get("result_count", 0) != 1 or not payload.get("results"):
801
+ return CanaryObservation(
802
+ value=f"v{NPI_API_VERSION}:{ANCHOR_NPI}:absent",
803
+ upstream_status=result.status,
804
+ note="anchor NPI not returned",
805
+ )
806
+
807
+ record = payload["results"][0]
808
+ basic = record.get("basic", {})
809
+ return CanaryObservation(
810
+ value=(
811
+ f"v{NPI_API_VERSION}:{ANCHOR_NPI}:{record.get('enumeration_type')}:"
812
+ f"{basic.get('status')}"
813
+ ),
814
+ schema_hash=_npi_schema_hash(payload),
815
+ upstream_status=result.status,
816
+ note="anchor NPI record and response shape",
817
+ )
818
+
819
+
820
+ def _npi_api_remediation(status: CanaryStatus, observed: str | None, expected: str | None) -> str:
821
+ if status is CanaryStatus.SCHEMA_CHANGED:
822
+ return (
823
+ "The NPI Registry API kept answering but its response shape moved (new or renamed "
824
+ "keys). Diff the live response for NPI "
825
+ f"{ANCHOR_NPI} against tests/fixtures/provider/lookup_npi_ok.json, update "
826
+ "_map_npi_record, bump TRANSFORM_VERSION in hc_source/adapters/provider.py, then "
827
+ "re-pin with `hc-source lock init`."
828
+ )
829
+ if status is CanaryStatus.UNREACHABLE:
830
+ return (
831
+ "The NPI Registry API could not be read. Check "
832
+ "https://npiregistry.cms.hhs.gov/api-page for outage or relocation notices, then "
833
+ "run `hc-source doctor` again."
834
+ )
835
+ if observed and observed.startswith("api-error:"):
836
+ return (
837
+ "The NPI Registry answered HTTP 200 with an Errors body -- this is exactly what "
838
+ f"version retirement looks like (v1.0 and v2.0 died the same way; v{NPI_API_VERSION} "
839
+ "was current as of 2026-08). Read https://npiregistry.cms.hhs.gov/api-page for the "
840
+ "current version, update NPI_API_VERSION in hc_source/adapters/provider.py, verify "
841
+ "the response mapping, then re-pin with `hc-source lock init`."
842
+ )
843
+ if observed and observed.endswith(":absent"):
844
+ return (
845
+ f"The anchor NPI {ANCHOR_NPI} is no longer served by the Registry API (likely "
846
+ "deactivated -- check the monthly Deactivated NPI Report). Pick a new boringly "
847
+ "stable anchor NPI, update ANCHOR_NPI and the recorded fixtures, then re-pin with "
848
+ "`hc-source lock init`."
849
+ )
850
+ return (
851
+ "The NPI Registry API contract moved. Compare observed vs expected, consult "
852
+ "https://npiregistry.cms.hhs.gov/api-page, then re-pin with `hc-source lock init`."
853
+ )
854
+
855
+
856
+ # --------------------------------------------------------------------------
857
+ # Canary 2: NPPES bulk V2 file listing + cadence
858
+ # --------------------------------------------------------------------------
859
+
860
+ _MONTHLY_RE = re.compile(r"NPPES_Data_Dissemination_([A-Z][a-z]+)_(\d{4})(_V\d+)?\.zip")
861
+ _WEEKLY_RE = re.compile(r"NPPES_Data_Dissemination_(\d{6})_(\d{6})_Weekly(_V\d+)?\.zip")
862
+ _DEACT_RE = re.compile(r"NPPES_Deactivated_NPI_Report_(\d{6})(_V\d+)?\.zip")
863
+
864
+ _MONTHS = {
865
+ name: i
866
+ for i, name in enumerate(
867
+ [
868
+ "January", "February", "March", "April", "May", "June",
869
+ "July", "August", "September", "October", "November", "December",
870
+ ],
871
+ start=1,
872
+ )
873
+ }
874
+
875
+ #: A monthly older than this many days means publication stalled (the file is
876
+ #: dated mid-month and the next lands ~a month later; 62 gives one month of
877
+ #: slack before alarming).
878
+ _MONTHLY_STALE_DAYS = 62
879
+
880
+
881
+ def _observe_nppes_files() -> CanaryObservation:
882
+ result = fetch(_nppes_files_url())
883
+ html = result.content.decode("utf-8", errors="replace")
884
+
885
+ monthlies: list[tuple[int, int, str]] = []
886
+ for m in _MONTHLY_RE.finditer(html):
887
+ month = _MONTHS.get(m.group(1))
888
+ if month:
889
+ monthlies.append((int(m.group(2)), month, m.group(0)))
890
+ weeklies = {m.group(0) for m in _WEEKLY_RE.finditer(html)}
891
+ deactivations = {m.group(0) for m in _DEACT_RE.finditer(html)}
892
+
893
+ if not monthlies:
894
+ value = "no-monthly-dissemination-file"
895
+ else:
896
+ year, month, filename = max(monthlies)
897
+ value = filename
898
+ flags: list[str] = []
899
+ today = _today()
900
+ age_days = (today - date(year, month, 1)).days
901
+ if age_days > _MONTHLY_STALE_DAYS + 31: # measured from month start
902
+ flags.append("stale-monthly")
903
+ if not weeklies:
904
+ flags.append("no-weekly-files")
905
+ if not deactivations:
906
+ flags.append("no-deactivation-report")
907
+ if flags:
908
+ value = "|".join([value, *sorted(flags)])
909
+
910
+ return CanaryObservation(
911
+ value=value,
912
+ upstream_status=result.status,
913
+ note=(
914
+ f"{len(set(f for *_ , f in monthlies))} monthly, {len(weeklies)} weekly, "
915
+ f"{len(deactivations)} deactivation file(s) listed"
916
+ ),
917
+ )
918
+
919
+
920
+ _V1_RETIREMENT_REMEDIATION = (
921
+ "NPPES retired the V1 bulk files on 2026-03-03; only *_V2.zip names are published now "
922
+ "(V2 extends the First Name and Legal Business Name field lengths). The listing this "
923
+ "canary just read carries a V1-style name with no _V2 suffix, which means the page has "
924
+ "regressed, a stale mirror is being served, or something upstream renamed the grammar "
925
+ "again. Fix: open https://download.cms.gov/nppes/NPI_Files.html, read the 'Important "
926
+ "Information' block (where CMS announced the V1 retirement), confirm the current version "
927
+ "suffix, make sure nothing in your pipeline still generates V1 filenames "
928
+ "(NPPES_Data_Dissemination_<Month>_<YYYY>.zip silently 404s), then re-pin with "
929
+ "`hc-source lock init`."
930
+ )
931
+
932
+
933
+ def _nppes_files_remediation(
934
+ status: CanaryStatus, observed: str | None, expected: str | None
935
+ ) -> str:
936
+ if status is CanaryStatus.UNREACHABLE:
937
+ return (
938
+ "download.cms.gov/nppes/NPI_Files.html could not be read or no longer lists a "
939
+ "monthly dissemination file. Locate the current NPPES dissemination page from "
940
+ "https://www.cms.gov/medicare/regulations-guidance/administrative-simplification/"
941
+ "data-dissemination, update NPPES_FILES_URL if it moved, then re-pin with "
942
+ "`hc-source lock init`."
943
+ )
944
+ head = (observed or "").split("|", 1)[0]
945
+ monthly = _MONTHLY_RE.fullmatch(head)
946
+ if monthly and not monthly.group(3):
947
+ return _V1_RETIREMENT_REMEDIATION
948
+ if monthly and monthly.group(3) and monthly.group(3) != "_V2":
949
+ return (
950
+ f"NPPES bumped the bulk-file version suffix to {monthly.group(3)[1:]} (precedent: "
951
+ "V1 retired 2026-03-03, V2 took over). Read the 'Important Information' block on "
952
+ "https://download.cms.gov/nppes/NPI_Files.html for what changed, update the "
953
+ "filename grammar and column map in hc_source/adapters/provider.py (bump "
954
+ "TRANSFORM_VERSION if the mapping moved), then re-pin with `hc-source lock init`."
955
+ )
956
+ if observed and "stale-monthly" in observed:
957
+ return (
958
+ "The newest NPPES monthly file is more than ~3 months old -- publication stalled "
959
+ "or the page stopped updating. Verify on "
960
+ "https://download.cms.gov/nppes/NPI_Files.html; if CMS paused publication, keep "
961
+ "the current pin and re-run doctor after the next release; otherwise re-pin with "
962
+ "`hc-source lock init`."
963
+ )
964
+ if observed and ("no-weekly-files" in observed or "no-deactivation-report" in observed):
965
+ return (
966
+ "The NPPES listing lost its weekly incrementals and/or the monthly Deactivated "
967
+ "NPI Report. Deactivation truth lives ONLY in that report (the main file strips "
968
+ "deactivated NPIs to two fields), so treat this as a real coverage gap: check "
969
+ "https://download.cms.gov/nppes/NPI_Files.html, then re-pin with "
970
+ "`hc-source lock init` once the listing is whole."
971
+ )
972
+ return (
973
+ "NPPES published a new monthly dissemination file. Review the release (column map "
974
+ "unchanged? deactivation report present?), then re-pin with `hc-source lock init`."
975
+ )
976
+
977
+
978
+ # --------------------------------------------------------------------------
979
+ # Canary 3: PECOS PPEF release identity + row-count + component inventory
980
+ # --------------------------------------------------------------------------
981
+
982
+ _PPEF_ROW_FLOOR = 2_500_000
983
+
984
+
985
+ def _observe_pecos_ppef() -> CanaryObservation:
986
+ stats_fetch, stats = fetch_json(_pecos_stats_url())
987
+ total = int(stats.get("total_rows", 0)) if isinstance(stats, dict) else 0
988
+
989
+ _, snapshot, _extract_date, missing = _fetch_ppef_version()
990
+
991
+ rows_fetch, rows = fetch_json(_pecos_data_url(), params={"size": 1})
992
+ if not isinstance(rows, list) or not rows or not isinstance(rows[0], dict):
993
+ raise SourceUnreachable(
994
+ rows_fetch.url, "PECOS Data API did not return a non-empty JSON array"
995
+ )
996
+
997
+ value = f"ppef-{snapshot}:rows={total}"
998
+ flags: list[str] = []
999
+ if total < _PPEF_ROW_FLOOR:
1000
+ flags.append("row-count-low")
1001
+ if missing:
1002
+ flags.append(f"missing-components={','.join(missing)}")
1003
+ if flags:
1004
+ value = "|".join([value, *sorted(flags)])
1005
+
1006
+ schema_hash = hashlib.sha256(
1007
+ json.dumps(sorted(rows[0].keys()), separators=(",", ":")).encode()
1008
+ ).hexdigest()
1009
+
1010
+ return CanaryObservation(
1011
+ value=value,
1012
+ schema_hash=schema_hash,
1013
+ upstream_status=stats_fetch.status,
1014
+ note="latest-alias snapshot id, row count, and column set",
1015
+ )
1016
+
1017
+
1018
+ def _pecos_remediation(status: CanaryStatus, observed: str | None, expected: str | None) -> str:
1019
+ if status is CanaryStatus.SCHEMA_CHANGED:
1020
+ return (
1021
+ "The PPEF enrollment extract kept its snapshot id but its column set moved. "
1022
+ "Fetch one row from the latest-alias Data API, diff the keys against "
1023
+ "tests/fixtures/provider/pecos_size1.json, update the column map and bump "
1024
+ "TRANSFORM_VERSION in hc_source/adapters/provider.py, then re-pin with "
1025
+ "`hc-source lock init`."
1026
+ )
1027
+ if status is CanaryStatus.UNREACHABLE:
1028
+ return (
1029
+ "The PECOS Public Provider Enrollment latest-alias dataset "
1030
+ f"({PECOS_LATEST_ALIAS_UUID}) could not be read -- it may have been re-keyed. "
1031
+ "Re-download https://data.cms.gov/data.json, locate the dataset titled "
1032
+ "'Medicare Fee-For-Service Public Provider Enrollment' (the double space is "
1033
+ "verbatim), take the first API distribution's UUID as the new latest alias, "
1034
+ "update PECOS_LATEST_ALIAS_UUID, then re-pin with `hc-source lock init`."
1035
+ )
1036
+ if observed and "row-count-low" in observed:
1037
+ return (
1038
+ "PPEF row count collapsed below 2.5M (2,978,925 as of 2026-08) -- likely a "
1039
+ "partial load upstream. Do NOT re-pin onto a truncated snapshot: re-check the "
1040
+ "row count in a day via `hc-source doctor`, and only re-pin with "
1041
+ "`hc-source lock init` once it recovers or CMS documents the change."
1042
+ )
1043
+ if observed and "missing-components" in observed:
1044
+ return (
1045
+ "A PPEF relational component (of: Enrollment, Additional NPIs, Reassignment, "
1046
+ "Practice Location, Secondary Specialty) vanished from dataset-resources. "
1047
+ "Diff the file inventory at "
1048
+ f"https://data.cms.gov/data-api/v1/dataset-resources/{PECOS_LATEST_ALIAS_UUID} "
1049
+ "against the five PPEF_* patterns, update _PPEF_COMPONENTS if CMS renamed one, "
1050
+ "then re-pin with `hc-source lock init`."
1051
+ )
1052
+ return (
1053
+ "PECOS published a new quarterly PPEF snapshot. Review the new extract date and row "
1054
+ "count, re-record fixtures if the shape moved, then re-pin with "
1055
+ "`hc-source lock init`."
1056
+ )
1057
+
1058
+
1059
+ # --------------------------------------------------------------------------
1060
+ # Adapter
1061
+ # --------------------------------------------------------------------------
1062
+
1063
+
1064
+ class ProviderAdapter(SourceAdapter):
1065
+ source_id = SOURCE_ID
1066
+ contract = CONTRACT
1067
+
1068
+ # These canaries read a live CMS service, so a single 502 is not news --
1069
+ # it is Tuesday. Two retries with exponential backoff, applied only to
1070
+ # transient failures (see hc_source.doctor._observe). A gate that goes red
1071
+ # on somebody else's bad afternoon gets uninstalled.
1072
+ canary_retries = 2
1073
+
1074
+ def canaries(self) -> list[Canary]:
1075
+ return [
1076
+ Canary(
1077
+ canary_id="provider.npi_api_contract",
1078
+ source_id=SOURCE_ID,
1079
+ description=(
1080
+ "NPI Registry API v2.1 answers the anchor NPI with the pinned record "
1081
+ "identity and response shape (errors arrive as HTTP 200 bodies)."
1082
+ ),
1083
+ observe=_observe_npi_api,
1084
+ remediation=(
1085
+ "The NPI Registry API contract moved. Check "
1086
+ "https://npiregistry.cms.hhs.gov/api-page, then re-pin with "
1087
+ "`hc-source lock init`."
1088
+ ),
1089
+ remediation_for=_npi_api_remediation,
1090
+ ),
1091
+ Canary(
1092
+ canary_id="provider.nppes_v2_files",
1093
+ source_id=SOURCE_ID,
1094
+ description=(
1095
+ "download.cms.gov/nppes lists the V2 monthly full file (newest wins; two "
1096
+ "may coexist at a month boundary), weekly incrementals, and the monthly "
1097
+ "Deactivated NPI Report, at a live cadence."
1098
+ ),
1099
+ observe=_observe_nppes_files,
1100
+ remediation=(
1101
+ "The NPPES bulk file listing moved (naming, cadence, or contents). Read "
1102
+ "https://download.cms.gov/nppes/NPI_Files.html and its 'Important "
1103
+ "Information' block, then re-pin with `hc-source lock init`."
1104
+ ),
1105
+ remediation_for=_nppes_files_remediation,
1106
+ ),
1107
+ Canary(
1108
+ canary_id="provider.pecos_ppef",
1109
+ source_id=SOURCE_ID,
1110
+ description=(
1111
+ "PECOS PPEF latest alias serves the pinned quarterly snapshot: extract "
1112
+ "date, row count above floor, five relational components, stable columns."
1113
+ ),
1114
+ observe=_observe_pecos_ppef,
1115
+ remediation=(
1116
+ "The PECOS Public Provider Enrollment dataset changed on data.cms.gov. "
1117
+ "Re-locate it in data.json (title has a verbatim double space), verify "
1118
+ "UUID and components, then re-pin with `hc-source lock init`."
1119
+ ),
1120
+ remediation_for=_pecos_remediation,
1121
+ ),
1122
+ ]
1123
+
1124
+ def tools(self) -> list[ToolSpec]:
1125
+ return [
1126
+ ToolSpec(
1127
+ name="provider.lookup_npi",
1128
+ description=(
1129
+ "Look up one NPI in the NPPES NPI Registry API v2.1 (daily refresh): "
1130
+ "identity, status, taxonomies, practice locations."
1131
+ ),
1132
+ params_model=NpiParams,
1133
+ handler=_lookup_npi,
1134
+ tags=("lookup",),
1135
+ ),
1136
+ ToolSpec(
1137
+ name="provider.pecos_enrollments",
1138
+ description=(
1139
+ "List a provider's Medicare FFS enrollment records from the PECOS PPEF "
1140
+ "quarterly snapshot (latest alias) for one NPI."
1141
+ ),
1142
+ params_model=NpiParams,
1143
+ handler=_pecos_enrollments,
1144
+ tags=("lookup",),
1145
+ ),
1146
+ ToolSpec(
1147
+ name="provider.evidence_ladder",
1148
+ description=(
1149
+ "Report NPI_EXISTS (NPPES) and PECOS_ENROLLED (PPEF snapshot) as "
1150
+ "separate facts with separate provenance -- never a merged score."
1151
+ ),
1152
+ params_model=NpiParams,
1153
+ handler=_evidence_ladder,
1154
+ tags=("evidence",),
1155
+ ),
1156
+ ]
1157
+
1158
+
1159
+ ADAPTER = ProviderAdapter()