mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,798 @@
1
+ """Strict metadata-only parser for the official Data.gov Catalog API v4 search response."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import re
7
+ from dataclasses import dataclass
8
+ from typing import Any
9
+ from urllib.parse import parse_qs, urlencode, urlsplit
10
+
11
+ from mostlyright.data_harness.acquisition.http import RetrievalLimits
12
+ from mostlyright.data_harness.acquisition.url_policy import MAX_URL_LENGTH
13
+ from mostlyright.data_harness.canonical import (
14
+ CanonicalJSONError,
15
+ canonical_json_bytes,
16
+ canonical_sha256,
17
+ sha256_bytes,
18
+ )
19
+ from mostlyright.data_harness.sources.catalog.harvest.protocol import (
20
+ DATAGOV_V4_HARVESTER_ID,
21
+ DATAGOV_V4_HARVESTER_VERSION,
22
+ CatalogHarvestError,
23
+ HarvestCursor,
24
+ HarvesterDescriptor,
25
+ HarvestPage,
26
+ HarvestResponseEvidence,
27
+ validate_datagov_cursor_text,
28
+ )
29
+ from mostlyright.data_harness.sources.contracts import EvidenceReference, _CanonicalContract
30
+
31
+ DATAGOV_V4_ENDPOINT = "https://api.gsa.gov/technology/datagov/v4/search"
32
+ DATAGOV_V4_RECORD_VERSION = "harness-datagov-v4-normalized-record.v2"
33
+ DATAGOV_V4_MAX_RESPONSE_BYTES = 16 * 1024 * 1024
34
+ DATAGOV_V4_MAX_RECORD_BYTES = 2 * 1024 * 1024
35
+ DATAGOV_V4_MAX_IDENTIFIER_CHARS = 512
36
+ DATAGOV_V4_MAX_CURSOR = 2_048
37
+ DATAGOV_V4_MAX_JSON_DEPTH = 256
38
+ DATAGOV_V4_MAX_NUMBER_TOKEN = 128
39
+ DATAGOV_V4_LIMITS = RetrievalLimits(
40
+ max_response_bytes=DATAGOV_V4_MAX_RESPONSE_BYTES,
41
+ max_aggregate_response_bytes=DATAGOV_V4_MAX_RESPONSE_BYTES,
42
+ max_redirects=0,
43
+ max_requests=1,
44
+ allowed_media_types=("application/json",),
45
+ )
46
+
47
+ _TOP_LEVEL_FIELDS = frozenset(
48
+ {
49
+ "dcat",
50
+ "description",
51
+ "distribution_titles",
52
+ "harvest_record",
53
+ "harvest_record_raw",
54
+ "harvest_record_transformed",
55
+ "has_download",
56
+ "has_spatial",
57
+ "identifier",
58
+ "keyword",
59
+ "last_harvested_date",
60
+ "organization",
61
+ "popularity",
62
+ "publisher",
63
+ "slug",
64
+ "spatial_centroid",
65
+ "spatial_shape",
66
+ "theme",
67
+ "title",
68
+ }
69
+ )
70
+ # Catalog-ingestion observations the provider rewrites on every re-harvest even when no metadata
71
+ # changed. They stay in raw and normalized evidence but never enter the semantic digest, because a
72
+ # digest that moved with a nightly harvest burst would make two adjacent equal pass maps
73
+ # unsatisfiable. `_score`/`_sort` are per-response ranking values and are not allowlisted at all.
74
+ DATAGOV_V4_VOLATILE_FIELDS = frozenset({"last_harvested_date", "popularity"})
75
+ _VOLATILE_PATHS = frozenset(
76
+ path for name in DATAGOV_V4_VOLATILE_FIELDS for path in (f"$.{name}", f"$.{name}[*]")
77
+ )
78
+ _DCAT_FIELDS = frozenset(
79
+ {
80
+ "@type",
81
+ "accessLevel",
82
+ "accrualPeriodicity",
83
+ "agencyDataSeriesURL",
84
+ "agencyProgramURL",
85
+ "analysisUnit",
86
+ "bureauCode",
87
+ "categoryDesignation",
88
+ "collectionInstrument",
89
+ "contactPoint",
90
+ "dataQuality",
91
+ "describedBy",
92
+ "describedByType",
93
+ "description",
94
+ "distribution",
95
+ "identifier",
96
+ "isPartOf",
97
+ "issued",
98
+ "keyword",
99
+ "landingPage",
100
+ "language",
101
+ "license",
102
+ "modified",
103
+ "phone",
104
+ "programCode",
105
+ "publisher",
106
+ "references",
107
+ "rights",
108
+ "spatial",
109
+ "temporal",
110
+ "theme",
111
+ "title",
112
+ }
113
+ )
114
+ _ORGANIZATION_FIELDS = frozenset(
115
+ {"id", "name", "slug", "organization_type", "aliases", "logo", "description"}
116
+ )
117
+ _DISTRIBUTION_FIELDS = frozenset(
118
+ {
119
+ "@type",
120
+ "accessURL",
121
+ "describedBy",
122
+ "describedByType",
123
+ "description",
124
+ "downloadURL",
125
+ "format",
126
+ "mediaType",
127
+ "title",
128
+ }
129
+ )
130
+ _PROVIDER_PATH = re.compile(r'^\$(?:\.[A-Za-z_][A-Za-z0-9_-]*|\[\*\]|\["@type"\])+$')
131
+ _PLAIN_PROVIDER_KEY = re.compile(r"^[A-Za-z_][A-Za-z0-9_-]*$")
132
+ _JSON_NUMBER = re.compile(r"-?(?:0|[1-9][0-9]*)(?:\.[0-9]+)?(?:[eE][+-]?[0-9]+)?")
133
+
134
+
135
+ class _ForeignJsonError(ValueError):
136
+ pass
137
+
138
+
139
+ @dataclass(frozen=True)
140
+ class _ForeignNumber:
141
+ lexeme: str
142
+
143
+
144
+ @dataclass(frozen=True)
145
+ class DatagovProjectedField(_CanonicalContract):
146
+ path: str
147
+ value: Any
148
+ encoding: str = "canonical_json"
149
+
150
+ def __post_init__(self) -> None:
151
+ if (
152
+ not isinstance(self.path, str)
153
+ or not _PROVIDER_PATH.fullmatch(self.path)
154
+ or not _allowed_provider_path(self.path, self.value, self.encoding)
155
+ ):
156
+ raise CatalogHarvestError("DATAGOV_PATH", "record.fields.path", "path is invalid")
157
+ if self.encoding not in {"canonical_json", "typed_json_ast.v1"}:
158
+ raise CatalogHarvestError("DATAGOV_VALUE", "record.fields.encoding", "is invalid")
159
+ try:
160
+ canonical_json_bytes(self.value)
161
+ except Exception as error:
162
+ raise CatalogHarvestError(
163
+ "DATAGOV_VALUE", "record.fields.value", "value is not strict JSON"
164
+ ) from error
165
+ if self.encoding == "typed_json_ast.v1":
166
+ _validate_typed_json_ast(self.value)
167
+
168
+ def to_dict(self) -> dict[str, Any]:
169
+ result = {"path": self.path, "value": self.value}
170
+ if self.encoding != "canonical_json":
171
+ result["encoding"] = self.encoding
172
+ return result
173
+
174
+
175
+ @dataclass(frozen=True)
176
+ class DatagovV4Record(_CanonicalContract):
177
+ record_id: str
178
+ fields: tuple[DatagovProjectedField, ...]
179
+ raw_response_sha256: str
180
+ observations: tuple[DatagovProjectedField, ...] = ()
181
+ protocol: str = "datagov_v4"
182
+ schema_version: str = DATAGOV_V4_RECORD_VERSION
183
+
184
+ def __post_init__(self) -> None:
185
+ if self.protocol != "datagov_v4" or self.schema_version != DATAGOV_V4_RECORD_VERSION:
186
+ raise CatalogHarvestError("VERSION", "record", "unsupported v4 record")
187
+ if not isinstance(self.record_id, str) or not self.record_id:
188
+ raise CatalogHarvestError(
189
+ "DATAGOV_IDENTIFIER", "record.record_id", "identifier must be nonempty"
190
+ )
191
+ try:
192
+ self.record_id.encode("utf-8", errors="strict")
193
+ except UnicodeEncodeError:
194
+ raise CatalogHarvestError(
195
+ "DATAGOV_IDENTIFIER", "record.record_id", "identifier must be valid Unicode"
196
+ ) from None
197
+ if len(self.record_id) > DATAGOV_V4_MAX_IDENTIFIER_CHARS:
198
+ raise CatalogHarvestError(
199
+ "DATAGOV_IDENTIFIER",
200
+ "record.record_id",
201
+ "identifier exceeds the shared provider identity bound",
202
+ )
203
+ for character in self.record_id:
204
+ codepoint = ord(character)
205
+ if (
206
+ codepoint < 0x20
207
+ or 0x7F <= codepoint <= 0x9F
208
+ or (character.isspace() and character != " ")
209
+ ):
210
+ raise CatalogHarvestError(
211
+ "DATAGOV_IDENTIFIER",
212
+ "record.record_id",
213
+ "identifier contains unsafe provider identity text",
214
+ )
215
+ if not isinstance(self.fields, tuple) or not self.fields:
216
+ raise CatalogHarvestError("DATAGOV_FIELDS", "record.fields", "fields are required")
217
+ if any(not isinstance(field, DatagovProjectedField) for field in self.fields):
218
+ raise CatalogHarvestError(
219
+ "DATAGOV_FIELDS", "record.fields", "fields must use the strict contract"
220
+ )
221
+ paths = tuple(field.path for field in self.fields)
222
+ if paths != tuple(sorted(paths)) or len(paths) != len(set(paths)):
223
+ raise CatalogHarvestError(
224
+ "DATAGOV_FIELDS", "record.fields", "field paths must be sorted and unique"
225
+ )
226
+ if any(path in _VOLATILE_PATHS for path in paths):
227
+ raise CatalogHarvestError(
228
+ "DATAGOV_FIELDS",
229
+ "record.fields",
230
+ "volatile observation fields may not carry a semantic field path",
231
+ )
232
+ if not isinstance(self.observations, tuple) or any(
233
+ not isinstance(field, DatagovProjectedField) for field in self.observations
234
+ ):
235
+ raise CatalogHarvestError(
236
+ "DATAGOV_OBSERVATIONS",
237
+ "record.observations",
238
+ "observations must use the strict contract",
239
+ )
240
+ observed_paths = tuple(field.path for field in self.observations)
241
+ if (
242
+ observed_paths != tuple(sorted(observed_paths))
243
+ or len(observed_paths) != len(set(observed_paths))
244
+ or any(path not in _VOLATILE_PATHS for path in observed_paths)
245
+ ):
246
+ raise CatalogHarvestError(
247
+ "DATAGOV_OBSERVATIONS",
248
+ "record.observations",
249
+ "observation paths must be sorted, unique, and volatile",
250
+ )
251
+ identifier_fields = tuple(field for field in self.fields if field.path == "$.identifier")
252
+ if (
253
+ len(identifier_fields) != 1
254
+ or identifier_fields[0].encoding != "canonical_json"
255
+ or identifier_fields[0].value != self.record_id
256
+ ):
257
+ raise CatalogHarvestError(
258
+ "DATAGOV_IDENTIFIER",
259
+ "record.fields",
260
+ "exactly one canonical $.identifier must equal record_id",
261
+ )
262
+ if len(self.raw_response_sha256) != 64 or any(
263
+ character not in "0123456789abcdef" for character in self.raw_response_sha256
264
+ ):
265
+ raise CatalogHarvestError("DIGEST", "record.raw_response_sha256", "must be SHA-256")
266
+ try:
267
+ semantic = self._semantic_dict()
268
+ observed = self._observed_dict()
269
+ normalized_size = len(canonical_json_bytes(observed))
270
+ normalized_record = dict(observed)
271
+ normalized_record["raw_response_sha256"] = self.raw_response_sha256
272
+ normalized_record["normalized_sha256"] = canonical_sha256(semantic)
273
+ canonical_json_bytes(
274
+ {
275
+ "schema_version": "harness-catalog-fill-shard.v1",
276
+ "kind": "normalized",
277
+ "payload": {
278
+ "schema_version": "harness-datagov-v4-normalized-record-segment.v1",
279
+ "raw_response_sha256": self.raw_response_sha256,
280
+ "first_record": 0,
281
+ "records": [normalized_record],
282
+ },
283
+ }
284
+ )
285
+ except CanonicalJSONError:
286
+ raise CatalogHarvestError(
287
+ "DATAGOV_RECORD_LIMIT",
288
+ "record",
289
+ "normalized record exceeds its canonical member contract",
290
+ ) from None
291
+ if normalized_size > DATAGOV_V4_MAX_RECORD_BYTES:
292
+ raise CatalogHarvestError(
293
+ "DATAGOV_RECORD_LIMIT", "record", "normalized record exceeds its byte bound"
294
+ )
295
+
296
+ def _semantic_dict(self) -> dict[str, Any]:
297
+ """Exactly the normalized allowlisted metadata fields the pass map converges on."""
298
+
299
+ return {
300
+ "schema_version": self.schema_version,
301
+ "protocol": self.protocol,
302
+ "record_id": self.record_id,
303
+ "fields": [field.to_dict() for field in self.fields],
304
+ }
305
+
306
+ def _observed_dict(self) -> dict[str, Any]:
307
+ """The semantic record plus retained, non-semantic volatile observation values."""
308
+
309
+ result = self._semantic_dict()
310
+ if self.observations:
311
+ result["observations"] = [field.to_dict() for field in self.observations]
312
+ return result
313
+
314
+ @property
315
+ def normalized_sha256(self) -> str:
316
+ return canonical_sha256(self._semantic_dict())
317
+
318
+ @property
319
+ def digest(self) -> str:
320
+ return self.normalized_sha256
321
+
322
+ def to_dict(self) -> dict[str, Any]:
323
+ result = self._observed_dict()
324
+ result["raw_response_sha256"] = self.raw_response_sha256
325
+ result["normalized_sha256"] = self.normalized_sha256
326
+ return result
327
+
328
+
329
+ class DatagovV4Harvester:
330
+ descriptor = HarvesterDescriptor(
331
+ protocol="datagov_v4",
332
+ harvester_id=DATAGOV_V4_HARVESTER_ID,
333
+ harvester_version=DATAGOV_V4_HARVESTER_VERSION,
334
+ )
335
+
336
+ def parse(
337
+ self,
338
+ payload: bytes,
339
+ *,
340
+ uri: str,
341
+ observed_at: str,
342
+ evidence: EvidenceReference,
343
+ ) -> tuple[DatagovV4Record, ...]:
344
+ return self.parse_page(
345
+ payload,
346
+ uri=uri,
347
+ observed_at=observed_at,
348
+ evidence=evidence,
349
+ response_evidence=None,
350
+ cursor=None,
351
+ ).records
352
+
353
+ def parse_page(
354
+ self,
355
+ payload: bytes,
356
+ *,
357
+ uri: str,
358
+ observed_at: str,
359
+ evidence: EvidenceReference,
360
+ response_evidence: HarvestResponseEvidence | None,
361
+ cursor: HarvestCursor | None,
362
+ ) -> HarvestPage:
363
+ if not isinstance(payload, bytes) or len(payload) > DATAGOV_V4_MAX_RESPONSE_BYTES:
364
+ raise CatalogHarvestError(
365
+ "DATAGOV_RESPONSE_LIMIT", "harvest.response", "v4 response exceeds 16 MiB"
366
+ )
367
+ if cursor is not None and cursor.protocol != "datagov_v4":
368
+ raise CatalogHarvestError("HARVEST_CURSOR", "cursor", "protocol mismatch")
369
+ try:
370
+ _preflight_json(payload)
371
+ document = json.loads(
372
+ payload.decode("utf-8", errors="strict"),
373
+ object_pairs_hook=_unique_object,
374
+ parse_float=_bounded_foreign_number,
375
+ parse_int=_bounded_foreign_integer,
376
+ parse_constant=_reject_nonfinite,
377
+ )
378
+ except (
379
+ UnicodeDecodeError,
380
+ json.JSONDecodeError,
381
+ _ForeignJsonError,
382
+ RecursionError,
383
+ ValueError,
384
+ OverflowError,
385
+ ):
386
+ raise CatalogHarvestError(
387
+ "DATAGOV_JSON", "harvest.response", "v4 response must be strict UTF-8 JSON"
388
+ ) from None
389
+ if not isinstance(document, dict) or set(document) - {"results", "sort", "after"}:
390
+ raise CatalogHarvestError(
391
+ "DATAGOV_ENVELOPE", "harvest.response", "v4 response envelope is invalid"
392
+ )
393
+ if document.get("sort") != _requested_sort(uri) or not isinstance(
394
+ document.get("results"), list
395
+ ):
396
+ raise CatalogHarvestError(
397
+ "DATAGOV_ENVELOPE", "harvest.response", "v4 sort/results are invalid"
398
+ )
399
+ requested_page_size = _requested_page_size(uri)
400
+ if len(document["results"]) > requested_page_size:
401
+ raise CatalogHarvestError(
402
+ "DATAGOV_PAGE_SIZE",
403
+ "harvest.response.results",
404
+ "v4 response contains more results than the exact request admitted",
405
+ )
406
+ after = document.get("after")
407
+ if after is not None and (
408
+ not isinstance(after, str) or not after or len(after) > DATAGOV_V4_MAX_CURSOR
409
+ ):
410
+ raise CatalogHarvestError(
411
+ "HARVEST_CURSOR", "harvest.response.after", "v4 cursor is invalid"
412
+ )
413
+ if after is not None:
414
+ _require_safe_cursor(after)
415
+ raw_sha256 = sha256_bytes(payload)
416
+ records: list[DatagovV4Record] = []
417
+ skipped: list[str] = []
418
+ for index, item in enumerate(document["results"]):
419
+ if not isinstance(item, dict):
420
+ skipped.append(f"index:{index}")
421
+ continue
422
+ identifier = item.get("identifier")
423
+ if not isinstance(identifier, str) or not identifier:
424
+ skipped.append(f"index:{index}")
425
+ continue
426
+ try:
427
+ fields, observations = _project_record(item)
428
+ records.append(
429
+ DatagovV4Record(
430
+ record_id=identifier,
431
+ fields=fields,
432
+ observations=observations,
433
+ raw_response_sha256=raw_sha256,
434
+ )
435
+ )
436
+ except CatalogHarvestError:
437
+ skipped.append(f"sha256:{sha256_bytes(identifier.encode('utf-8', 'replace'))}")
438
+ next_cursor = (
439
+ None
440
+ if after is None
441
+ else HarvestCursor(protocol="datagov_v4", kind="after", value=after)
442
+ )
443
+ return HarvestPage(
444
+ protocol="datagov_v4",
445
+ records=tuple(records),
446
+ next_cursor=next_cursor,
447
+ provider_count=None,
448
+ count_basis="not_reported",
449
+ skipped_record_ids=tuple(sorted(skipped)),
450
+ evidence=evidence,
451
+ response_evidence=response_evidence,
452
+ )
453
+
454
+
455
+ def _project_record(
456
+ item: dict[str, Any],
457
+ ) -> tuple[tuple[DatagovProjectedField, ...], tuple[DatagovProjectedField, ...]]:
458
+ """Project the allowlisted union, then split semantic fields from volatile observations."""
459
+
460
+ projected: list[DatagovProjectedField] = []
461
+ base = "$"
462
+ for name in sorted(set(item) & _TOP_LEVEL_FIELDS):
463
+ value = item[name]
464
+ path = f"{base}.{name}"
465
+ if name == "dcat":
466
+ if not isinstance(value, dict):
467
+ raise CatalogHarvestError("DATAGOV_VALUE", path, "dcat must be a closed object")
468
+ for child in sorted(set(value) & _DCAT_FIELDS):
469
+ _flatten_allowed(
470
+ projected,
471
+ _provider_child_path(path, child),
472
+ value[child],
473
+ child_allowlist=_DISTRIBUTION_FIELDS if child == "distribution" else None,
474
+ allow_structured_value=child != "distribution",
475
+ )
476
+ elif name == "organization":
477
+ if not isinstance(value, dict):
478
+ raise CatalogHarvestError(
479
+ "DATAGOV_VALUE", path, "organization must be a closed object"
480
+ )
481
+ for child in sorted(set(value) & _ORGANIZATION_FIELDS):
482
+ _flatten_allowed(
483
+ projected,
484
+ _provider_child_path(path, child),
485
+ value[child],
486
+ allow_structured_value=True,
487
+ )
488
+ else:
489
+ _flatten_allowed(projected, path, value, allow_structured_value=True)
490
+ ordered = sorted(projected, key=lambda field: field.path)
491
+ return (
492
+ tuple(field for field in ordered if field.path not in _VOLATILE_PATHS),
493
+ tuple(field for field in ordered if field.path in _VOLATILE_PATHS),
494
+ )
495
+
496
+
497
+ def _flatten_allowed(
498
+ target: list[DatagovProjectedField],
499
+ path: str,
500
+ value: Any,
501
+ *,
502
+ child_allowlist: frozenset[str] | None = None,
503
+ allow_structured_value: bool = False,
504
+ ) -> None:
505
+ if isinstance(value, dict):
506
+ if child_allowlist is None:
507
+ if not allow_structured_value:
508
+ raise CatalogHarvestError(
509
+ "DATAGOV_VALUE", path, "nested objects require a closed field schema"
510
+ )
511
+ target.append(_projected_field(path, value))
512
+ return
513
+ keys = set(value) & child_allowlist
514
+ if not keys:
515
+ target.append(DatagovProjectedField(path, {}))
516
+ return
517
+ for key in sorted(keys):
518
+ _flatten_allowed(target, _provider_child_path(path, key), value[key])
519
+ return
520
+ if isinstance(value, list):
521
+ if not value:
522
+ target.append(DatagovProjectedField(f"{path}[*]", []))
523
+ return
524
+ if child_allowlist is not None:
525
+ if not all(isinstance(item, dict) for item in value):
526
+ raise CatalogHarvestError(
527
+ "DATAGOV_VALUE", path, "closed object arrays may not mix member types"
528
+ )
529
+ keys = set().union(*(set(item) for item in value)) & child_allowlist
530
+ for key in sorted(keys):
531
+ values = [item[key] for item in value if key in item]
532
+ stable_value: Any = values[0] if len(values) == 1 else values
533
+ target.append(
534
+ _projected_field(_provider_child_path(f"{path}[*]", key), stable_value)
535
+ )
536
+ return
537
+ if not allow_structured_value and any(isinstance(item, (dict, list)) for item in value):
538
+ raise CatalogHarvestError(
539
+ "DATAGOV_VALUE", path, "nested arrays or objects require a closed field schema"
540
+ )
541
+ target.append(_projected_field(f"{path}[*]", value))
542
+ return
543
+ target.append(_projected_field(path, value))
544
+
545
+
546
+ def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
547
+ result: dict[str, Any] = {}
548
+ for key, value in pairs:
549
+ if key in result:
550
+ raise _ForeignJsonError("duplicate object key")
551
+ result[key] = value
552
+ return result
553
+
554
+
555
+ def _provider_child_path(path: str, key: str) -> str:
556
+ if key == "@type":
557
+ return f'{path}["@type"]'
558
+ if not _PLAIN_PROVIDER_KEY.fullmatch(key):
559
+ raise _ForeignJsonError("allowlisted provider key has no canonical path spelling")
560
+ return f"{path}.{key}"
561
+
562
+
563
+ def _allowed_provider_path(path: str, value: Any, encoding: str) -> bool:
564
+ if path == "$.dcat.distribution":
565
+ return encoding == "canonical_json" and value == {}
566
+ if path == "$.dcat.distribution[*]":
567
+ return encoding == "canonical_json" and value == []
568
+ for name in _TOP_LEVEL_FIELDS - {"dcat", "organization"}:
569
+ if path in {f"$.{name}", f"$.{name}[*]"}:
570
+ return True
571
+ for name in _ORGANIZATION_FIELDS:
572
+ if path in {f"$.organization.{name}", f"$.organization.{name}[*]"}:
573
+ return True
574
+ for name in _DCAT_FIELDS - {"distribution"}:
575
+ base = _provider_child_path("$.dcat", name)
576
+ if path in {base, f"{base}[*]"}:
577
+ return True
578
+ for name in _DISTRIBUTION_FIELDS:
579
+ if path == _provider_child_path("$.dcat.distribution[*]", name):
580
+ return True
581
+ return False
582
+
583
+
584
+ def _validate_typed_json_ast(value: Any) -> None:
585
+ pending: list[tuple[Any, int]] = [(value, 1)]
586
+ visited = 0
587
+ while pending:
588
+ current, depth = pending.pop()
589
+ visited += 1
590
+ if visited > 1_000_000 or depth > DATAGOV_V4_MAX_JSON_DEPTH:
591
+ raise CatalogHarvestError(
592
+ "DATAGOV_VALUE", "record.fields.value", "typed JSON AST exceeds its bound"
593
+ )
594
+ if not isinstance(current, dict) or not isinstance(current.get("type"), str):
595
+ raise CatalogHarvestError(
596
+ "DATAGOV_VALUE", "record.fields.value", "typed JSON AST node is invalid"
597
+ )
598
+ kind = current["type"]
599
+ if kind == "null":
600
+ valid = set(current) == {"type"}
601
+ elif kind == "boolean":
602
+ valid = set(current) == {"type", "value"} and type(current["value"]) is bool
603
+ elif kind == "integer":
604
+ valid = set(current) == {"type", "value"} and type(current["value"]) is int
605
+ elif kind == "number":
606
+ lexeme = current.get("lexeme")
607
+ valid = (
608
+ set(current) == {"type", "lexeme"}
609
+ and isinstance(lexeme, str)
610
+ and len(lexeme) <= DATAGOV_V4_MAX_NUMBER_TOKEN
611
+ and _JSON_NUMBER.fullmatch(lexeme) is not None
612
+ )
613
+ elif kind == "string":
614
+ valid = set(current) == {"type", "value"} and isinstance(current["value"], str)
615
+ elif kind == "array":
616
+ items = current.get("items")
617
+ valid = set(current) == {"type", "items"} and isinstance(items, list)
618
+ if valid:
619
+ pending.extend((item, depth + 1) for item in reversed(items))
620
+ elif kind == "object":
621
+ entries = current.get("entries")
622
+ valid = set(current) == {"type", "entries"} and isinstance(entries, list)
623
+ if valid:
624
+ keys: list[str] = []
625
+ for entry in entries:
626
+ if (
627
+ not isinstance(entry, dict)
628
+ or set(entry) != {"key", "value"}
629
+ or not isinstance(entry["key"], str)
630
+ ):
631
+ valid = False
632
+ break
633
+ keys.append(entry["key"])
634
+ pending.append((entry["value"], depth + 1))
635
+ valid = valid and keys == sorted(set(keys))
636
+ else:
637
+ valid = False
638
+ if not valid:
639
+ raise CatalogHarvestError(
640
+ "DATAGOV_VALUE", "record.fields.value", "typed JSON AST node is invalid"
641
+ )
642
+
643
+
644
+ def _bounded_foreign_integer(value: str) -> int | _ForeignNumber:
645
+ if len(value) > DATAGOV_V4_MAX_NUMBER_TOKEN:
646
+ raise _ForeignJsonError("numeric token exceeds bound")
647
+ parsed = int(value)
648
+ return parsed if -(1 << 63) <= parsed <= (1 << 63) - 1 else _ForeignNumber(value)
649
+
650
+
651
+ def _reject_nonfinite(_value: str) -> None:
652
+ raise _ForeignJsonError("non-finite number")
653
+
654
+
655
+ def _bounded_foreign_number(value: str) -> _ForeignNumber:
656
+ if len(value) > DATAGOV_V4_MAX_NUMBER_TOKEN:
657
+ raise _ForeignJsonError("numeric token exceeds bound")
658
+ return _ForeignNumber(value)
659
+
660
+
661
+ def _projected_field(path: str, value: Any) -> DatagovProjectedField:
662
+ try:
663
+ if not _contains_foreign_number(value):
664
+ return DatagovProjectedField(path, value)
665
+ return DatagovProjectedField(path, _typed_json_ast(value), encoding="typed_json_ast.v1")
666
+ except (_ForeignJsonError, RecursionError):
667
+ raise CatalogHarvestError(
668
+ "DATAGOV_VALUE", path, "provider value exceeds the bounded traversal contract"
669
+ ) from None
670
+
671
+
672
+ def _contains_foreign_number(value: Any) -> bool:
673
+ pending = [value]
674
+ visited = 0
675
+ while pending:
676
+ current = pending.pop()
677
+ visited += 1
678
+ if visited > 1_000_000:
679
+ raise _ForeignJsonError("foreign JSON value exceeds traversal bound")
680
+ if isinstance(current, _ForeignNumber):
681
+ return True
682
+ if isinstance(current, list):
683
+ pending.extend(current)
684
+ elif isinstance(current, dict):
685
+ pending.extend(current.values())
686
+ return False
687
+
688
+
689
+ def _typed_json_ast(value: Any) -> dict[str, Any]:
690
+ if value is None:
691
+ return {"type": "null"}
692
+ if type(value) is bool:
693
+ return {"type": "boolean", "value": value}
694
+ if type(value) is int:
695
+ return {"type": "integer", "value": value}
696
+ if isinstance(value, _ForeignNumber):
697
+ return {"type": "number", "lexeme": value.lexeme}
698
+ if isinstance(value, str):
699
+ return {"type": "string", "value": value}
700
+ if isinstance(value, list):
701
+ return {"type": "array", "items": [_typed_json_ast(item) for item in value]}
702
+ if isinstance(value, dict):
703
+ return {
704
+ "type": "object",
705
+ "entries": [
706
+ {"key": key, "value": _typed_json_ast(value[key])} for key in sorted(value)
707
+ ],
708
+ }
709
+ raise _ForeignJsonError("foreign JSON value type is invalid")
710
+
711
+
712
+ def _preflight_json(payload: bytes) -> None:
713
+ """Bound nesting and numeric lexemes before CPython's decoder touches them."""
714
+
715
+ depth = 0
716
+ in_string = False
717
+ escaped = False
718
+ index = 0
719
+ while index < len(payload):
720
+ byte = payload[index]
721
+ if in_string:
722
+ if escaped:
723
+ escaped = False
724
+ elif byte == 0x5C:
725
+ escaped = True
726
+ elif byte == 0x22:
727
+ in_string = False
728
+ index += 1
729
+ continue
730
+ if byte == 0x22:
731
+ in_string = True
732
+ elif byte in {0x5B, 0x7B}:
733
+ depth += 1
734
+ if depth > DATAGOV_V4_MAX_JSON_DEPTH:
735
+ raise _ForeignJsonError("JSON nesting exceeds bound")
736
+ elif byte in {0x5D, 0x7D}:
737
+ depth -= 1
738
+ if depth < 0:
739
+ raise _ForeignJsonError("JSON nesting is invalid")
740
+ elif byte == 0x2D or 0x30 <= byte <= 0x39:
741
+ end = index + 1
742
+ while end < len(payload) and payload[end] in b"0123456789eE+.-":
743
+ end += 1
744
+ if end - index > DATAGOV_V4_MAX_NUMBER_TOKEN:
745
+ raise _ForeignJsonError("numeric token exceeds bound")
746
+ index = end
747
+ continue
748
+ index += 1
749
+
750
+
751
+ #: The sorts the v4 cursor contract admits, longest first so the URL bound stays the strictest.
752
+ DATAGOV_V4_SORT_ORDERS = ("last_harvested_date", "relevance")
753
+
754
+
755
+ def _requested_sort(uri: str) -> str:
756
+ query = parse_qs(urlsplit(uri).query, keep_blank_values=True)
757
+ values = query.get("sort")
758
+ if values is None:
759
+ return DATAGOV_V4_SORT_ORDERS[0]
760
+ if len(values) != 1 or values[0] not in DATAGOV_V4_SORT_ORDERS:
761
+ raise CatalogHarvestError(
762
+ "DATAGOV_SORT", "harvest.request.sort", "sort is not one this contract can page"
763
+ )
764
+ return values[0]
765
+
766
+
767
+ def _requested_page_size(uri: str) -> int:
768
+ query = parse_qs(urlsplit(uri).query, keep_blank_values=True)
769
+ values = query.get("per_page")
770
+ if values is None:
771
+ return 1_000
772
+ if len(values) != 1 or not re.fullmatch(r"[1-9][0-9]{0,3}", values[0]):
773
+ raise CatalogHarvestError(
774
+ "DATAGOV_PAGE_SIZE", "harvest.request.per_page", "per_page is invalid"
775
+ )
776
+ page_size = int(values[0])
777
+ if page_size > 1_000:
778
+ raise CatalogHarvestError(
779
+ "DATAGOV_PAGE_SIZE", "harvest.request.per_page", "per_page exceeds 1000"
780
+ )
781
+ return page_size
782
+
783
+
784
+ def _require_safe_cursor(value: str) -> None:
785
+ validate_datagov_cursor_text(value, "harvest.response.after")
786
+ continuation = f"{DATAGOV_V4_ENDPOINT}?" + urlencode(
787
+ (
788
+ ("sort", max(DATAGOV_V4_SORT_ORDERS, key=len)),
789
+ ("per_page", "1000"),
790
+ ("after", value),
791
+ )
792
+ )
793
+ if len(continuation) > MAX_URL_LENGTH:
794
+ raise CatalogHarvestError(
795
+ "HARVEST_CURSOR",
796
+ "harvest.response.after",
797
+ "Data.gov cursor exceeds the canonical continuation URL boundary",
798
+ )