mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,445 @@
1
+ """The SDMX harvester: read one 2.1 dataflow structure message into public-source records.
2
+
3
+ XML is the one genuinely new attack surface in the harvest package, and it is hardened in three
4
+ ordered steps.
5
+
6
+ **First, a byte-level pre-check, before any parser object exists.** A payload declaring
7
+ ``<!DOCTYPE`` or ``<!ENTITY`` is refused outright. That single check closes XXE, external-entity
8
+ SSRF and entity-expansion (the billion-laughs family) together, because all three require a
9
+ document type or an entity declaration. Refusing rather than sanitising is right here: a
10
+ legitimate SDMX structure message has no reason to declare either, so there is nothing to lose.
11
+
12
+ The check walks the whole prolog rather than scanning a fixed-width prefix of it, and it walks it
13
+ in the encoding the parser will read. Both halves are load-bearing, and each was found missing in
14
+ turn.
15
+
16
+ A prefix scan looks equivalent to a walk and is not: comments, processing instructions and
17
+ whitespace may all legally precede the root element at any length, so padding past the window
18
+ would carry a declaration through to the parser. The walk is bounded by ``MAX_PROLOG_BYTES`` and
19
+ fails closed, because walking an unbounded prolog is itself attacker-controlled work.
20
+
21
+ A UTF-8-only scan looks equivalent to an encoding-aware one and is not: expat auto-detects UTF-16
22
+ and UTF-32 from a byte-order mark or from the byte spelling of the mandatory leading ``<``, so a
23
+ declaration written in either was invisible to a search for the ASCII pattern and reached the
24
+ parser unseen. The payload is therefore normalised to UTF-8 before the walk, by the parser's own
25
+ detection rules. This was not theoretical: a UTF-16 payload well inside the byte cap defined an
26
+ entity that expanded to sixteen million characters, which the element budget cannot see because
27
+ one element with an enormous body is still one element.
28
+
29
+ Nothing behind the pre-check would catch either evasion -- ``ElementTree`` expands a defined
30
+ internal entity, and the bundled expat's amplification limit is a build-dependent accident rather
31
+ than a control this module may lean on.
32
+
33
+ **Second, an element budget enforced during a pull parse.** ``MAX_XML_ELEMENTS`` stops work in
34
+ progress. A one-shot ``fromstring`` would build the entire tree before anyone could count it,
35
+ which turns a size-capped response into an unbounded allocation.
36
+
37
+ **Third, typed failure.** A raw ``ParseError`` escaping a trust boundary is an untyped failure at
38
+ exactly the place a caller needs to distinguish "this endpoint is broken" from "this endpoint is
39
+ hostile", so parser errors are wrapped.
40
+
41
+ No XML dependency is added. The three checks above are what a hardening library would do for this
42
+ document shape, and pulling one in would change the governed dependency closure for no gain.
43
+
44
+ **Query shape.** ``dataflow/{agency}/{id}/latest`` is what this harvester expects. ``all/latest``
45
+ is expected to exceed the byte cap on a large provider -- the observed Eurostat response is
46
+ 37,166,239 bytes against a 4 MiB cap. That is a usage constraint, not a defect: an agent asking a
47
+ statistical office for its entire structure catalogue in one request is asking the wrong question.
48
+ The endpoint path itself is per-provider configuration, so this module never constructs one.
49
+
50
+ **Payload formats.** A structure message describes dataflows, not payloads, so every record here
51
+ carries ``data_formats=()``. That is honest and it is also a scope limit worth stating plainly: an
52
+ SDMX record cannot be composed into a catalog entry on the record's own evidence, because an entry
53
+ must declare at least one format the harness can read. The author supplies them at the composition
54
+ seam -- ``compose_catalog_entry(..., data_formats=("csv",))`` -- and the harvested record remains
55
+ the cited evidence for everything else. Guessing a format here would be this module claiming a
56
+ capability the structure message never stated.
57
+
58
+ **Licences.** An SDMX structure message carries none, so every record has ``license_id=None``.
59
+ That is honest, and it maps to ``unclear`` rights, which the facts gate escalates to a human
60
+ rather than guessing at -- the correct outcome for a source whose terms are genuinely not
61
+ machine-readable.
62
+ """
63
+
64
+ from __future__ import annotations
65
+
66
+ import re
67
+ from xml.etree.ElementTree import ParseError, XMLPullParser
68
+
69
+ from mostlyright.data_harness.sources.catalog.harvest.ckan import MAX_HARVEST_RECORDS
70
+ from mostlyright.data_harness.sources.catalog.harvest.protocol import (
71
+ MAX_RECORD_PUBLISHER,
72
+ MAX_RECORD_TEXT,
73
+ MAX_RECORD_TITLE,
74
+ CatalogHarvestError,
75
+ HarvestCursor,
76
+ HarvestedRecord,
77
+ HarvesterDescriptor,
78
+ HarvestPage,
79
+ HarvestResponseEvidence,
80
+ bounded_record_text,
81
+ harvest_limits,
82
+ usable_https_uri,
83
+ usable_record_id,
84
+ )
85
+ from mostlyright.data_harness.sources.contracts import EvidenceReference
86
+
87
+ SDMX_LIMITS = harvest_limits(
88
+ "application/vnd.sdmx.structure+xml",
89
+ max_response_bytes=4 * 1024 * 1024,
90
+ )
91
+
92
+ MAX_XML_ELEMENTS = 200_000
93
+
94
+ # How much prolog the declaration pre-check will walk before refusing. A declaration must precede
95
+ # the root element, but the prolog itself is unbounded -- comments, processing instructions and
96
+ # whitespace may all be arbitrarily long -- so the walk is bounded and fails closed. A real SDMX
97
+ # structure message carries a prolog of a few dozen bytes.
98
+ MAX_PROLOG_BYTES = 64 * 1024
99
+
100
+ # Feed size for the bounded pull parse.
101
+ FEED_CHUNK_BYTES = 64 * 1024
102
+
103
+ SDMX_STRUCTURE_NS = "http://www.sdmx.org/resources/sdmxml/schemas/v2_1/structure"
104
+ SDMX_MESSAGE_NS = "http://www.sdmx.org/resources/sdmxml/schemas/v2_1/message"
105
+ SDMX_COMMON_NS = "http://www.sdmx.org/resources/sdmxml/schemas/v2_1/common"
106
+ XML_NS = "http://www.w3.org/XML/1998/namespace"
107
+
108
+ _DATAFLOW_TAG = f"{{{SDMX_STRUCTURE_NS}}}Dataflow"
109
+ _NAME_TAG = f"{{{SDMX_COMMON_NS}}}Name"
110
+ _DESCRIPTION_TAG = f"{{{SDMX_COMMON_NS}}}Description"
111
+ _LANG_ATTR = f"{{{XML_NS}}}lang"
112
+
113
+ _DECLARATION = re.compile(rb"(?i)<!\s*(doctype|entity)")
114
+
115
+ # How a conforming XML parser works out the byte order of a document before it has read the
116
+ # encoding declaration -- XML 1.0 appendix F, and what expat implements. Longest patterns first,
117
+ # because a UTF-32LE mark begins with a UTF-16LE one. The four unmarked patterns are the byte
118
+ # spellings of the mandatory leading `<`.
119
+ _AUTODETECTED_ENCODINGS = (
120
+ (b"\x00\x00\xfe\xff", "utf-32"),
121
+ (b"\xff\xfe\x00\x00", "utf-32"),
122
+ (b"\x00\x00\x00\x3c", "utf-32-be"),
123
+ (b"\x3c\x00\x00\x00", "utf-32-le"),
124
+ (b"\xfe\xff", "utf-16"),
125
+ (b"\xff\xfe", "utf-16"),
126
+ (b"\x00\x3c", "utf-16-be"),
127
+ (b"\x3c\x00", "utf-16-le"),
128
+ )
129
+
130
+
131
+ class SdmxHarvester:
132
+ """Read an SDMX 2.1 dataflow structure message. Takes bytes; never fetches."""
133
+
134
+ _DESCRIPTOR = HarvesterDescriptor(
135
+ protocol="sdmx",
136
+ harvester_id="sdmx.dataflow",
137
+ harvester_version="1.0.0",
138
+ )
139
+
140
+ @property
141
+ def descriptor(self) -> HarvesterDescriptor:
142
+ return self._DESCRIPTOR
143
+
144
+ def parse(
145
+ self,
146
+ payload: bytes,
147
+ *,
148
+ uri: str,
149
+ observed_at: str,
150
+ evidence: EvidenceReference,
151
+ ) -> tuple[HarvestedRecord, ...]:
152
+ page = self.parse_page(
153
+ payload,
154
+ uri=uri,
155
+ observed_at=observed_at,
156
+ evidence=evidence,
157
+ response_evidence=None,
158
+ cursor=None,
159
+ )
160
+ if not page.records:
161
+ raise CatalogHarvestError(
162
+ "HARVEST_SHAPE", "sdmx.Structures.Dataflows", "message carries no dataflow"
163
+ )
164
+ return page.records
165
+
166
+ def parse_page(
167
+ self,
168
+ payload: bytes,
169
+ *,
170
+ uri: str,
171
+ observed_at: str,
172
+ evidence: EvidenceReference,
173
+ response_evidence: HarvestResponseEvidence | None,
174
+ cursor: HarvestCursor | None,
175
+ ) -> HarvestPage:
176
+ if cursor is not None:
177
+ raise CatalogHarvestError(
178
+ "HARVEST_CURSOR", "sdmx.cursor", "SDMX dataflow harvest is one-shot"
179
+ )
180
+ if not isinstance(payload, bytes):
181
+ raise CatalogHarvestError("TYPE", "sdmx.payload", "payload must be bytes")
182
+ require_no_xml_declarations(payload)
183
+ root = _parse_bounded(payload)
184
+ dataflows = [element for element in root.iter() if element.tag == _DATAFLOW_TAG]
185
+ if len(dataflows) > MAX_HARVEST_RECORDS:
186
+ raise CatalogHarvestError(
187
+ "HARVEST_RECORD_LIMIT",
188
+ "sdmx.Structures.Dataflows",
189
+ f"a single message may not carry more than {MAX_HARVEST_RECORDS} dataflows",
190
+ )
191
+ records: list[HarvestedRecord] = []
192
+ skipped: list[str] = []
193
+ for position, dataflow in enumerate(dataflows):
194
+ record = _record(
195
+ dataflow,
196
+ position=position,
197
+ uri=uri,
198
+ evidence=evidence,
199
+ skipped=skipped,
200
+ )
201
+ if record is not None:
202
+ records.append(record)
203
+ return HarvestPage(
204
+ protocol="sdmx",
205
+ records=tuple(records),
206
+ next_cursor=None,
207
+ provider_count=len(dataflows),
208
+ count_basis="one_shot_response",
209
+ skipped_record_ids=tuple(sorted(skipped)),
210
+ evidence=evidence,
211
+ response_evidence=response_evidence,
212
+ )
213
+
214
+
215
+ def require_no_xml_declarations(payload: bytes) -> None:
216
+ """Refuse a payload declaring a document type or an entity, before any parser exists.
217
+
218
+ The whole prolog is walked, not a fixed-width prefix of it. A prefix scan is an evadable
219
+ control: comments, processing instructions and whitespace are all legal before the root element
220
+ and all three can be padded to any length, so a declaration hidden behind enough padding would
221
+ reach the parser -- and ElementTree does expand a defined internal entity, so the parser is not
222
+ a fallback control. The walk consumes exactly the constructs that may legally precede the root
223
+ element and stops at the first thing that is not one of them.
224
+
225
+ The walk runs over UTF-8 bytes, whichever byte order the sender chose. expat auto-detects
226
+ UTF-16 and UTF-32 from a byte-order mark or from the first character's byte pattern, so a
227
+ declaration spelled in either was invisible to a scan for the ASCII pattern and reached the
228
+ parser unseen -- a walk that only reads one encoding is an evadable control for the same
229
+ reason a prefix scan is. Normalising rather than refusing keeps the refusal in one place and
230
+ keeps the reported code the same whatever the encoding.
231
+ """
232
+
233
+ payload = _utf8_normalised(payload)
234
+ cursor = _skip_bom_and_space(payload, 0)
235
+ while cursor < len(payload):
236
+ if cursor > MAX_PROLOG_BYTES:
237
+ raise _declaration_refused(
238
+ f"the prolog exceeds {MAX_PROLOG_BYTES} bytes before any element begins, so it "
239
+ "cannot be cleared of declarations within its budget"
240
+ )
241
+ window = payload[cursor : cursor + 32]
242
+ match = _DECLARATION.match(window)
243
+ if match is not None:
244
+ raise _declaration_refused(
245
+ f"the document declares a {match.group(1).decode('ascii').lower()}; document type "
246
+ "and entity declarations are refused outright, because an SDMX structure message "
247
+ "has no reason to carry one"
248
+ )
249
+ if window.startswith(b"<!--"):
250
+ end = payload.find(b"-->", cursor + 4)
251
+ if end < 0:
252
+ # Unterminated: nothing can follow it, so no declaration can be hiding behind it.
253
+ return
254
+ cursor = _skip_bom_and_space(payload, end + 3)
255
+ continue
256
+ if window.startswith(b"<?"):
257
+ end = payload.find(b"?>", cursor + 2)
258
+ if end < 0:
259
+ return
260
+ cursor = _skip_bom_and_space(payload, end + 2)
261
+ continue
262
+ # Anything else is the root element or malformed input; either way the prolog is over and
263
+ # no declaration may legally appear past this point.
264
+ return
265
+
266
+
267
+ def _utf8_normalised(payload: bytes) -> bytes:
268
+ """Return the payload as UTF-8 bytes, whatever byte order it arrived in.
269
+
270
+ The detection rules are the parser's own -- XML 1.0 appendix F, which is what expat implements:
271
+ a byte-order mark if there is one, otherwise the byte pattern of the mandatory leading ``<``.
272
+ Reading it any other way would leave a gap between what this pre-check inspects and what the
273
+ parser will act on, which is precisely the defect this closes.
274
+
275
+ A payload that announces one of these encodings and then is not valid in it is refused rather
276
+ than passed along, because at that point nothing can say what the parser would make of it.
277
+ """
278
+
279
+ encoding = _autodetected_encoding(payload)
280
+ if encoding is None:
281
+ return payload
282
+ try:
283
+ text = payload.decode(encoding)
284
+ except UnicodeDecodeError:
285
+ raise CatalogHarvestError(
286
+ "HARVEST_XML_ENCODING",
287
+ "sdmx.payload",
288
+ f"the payload announces {encoding} and is not valid in it, so what the parser would "
289
+ "make of it cannot be established before parsing it",
290
+ ) from None
291
+ return text.encode("utf-8")
292
+
293
+
294
+ def _autodetected_encoding(payload: bytes) -> str | None:
295
+ """The encoding a conforming XML parser would detect from the first four bytes.
296
+
297
+ ``None`` means UTF-8 or another ASCII-compatible single-byte encoding, for which the ASCII
298
+ declaration pattern is already directly readable in the bytes. The byte-order-mark codecs are
299
+ named without an endianness suffix on purpose: those forms consume the mark, so the decoded
300
+ text starts at the first real character.
301
+ """
302
+
303
+ prefix = payload[:4]
304
+ for pattern, encoding in _AUTODETECTED_ENCODINGS:
305
+ if prefix.startswith(pattern):
306
+ return encoding
307
+ return None
308
+
309
+
310
+ def _skip_bom_and_space(payload: bytes, cursor: int) -> int:
311
+ if cursor == 0 and payload.startswith(b"\xef\xbb\xbf"):
312
+ cursor = 3
313
+ while cursor < len(payload) and payload[cursor : cursor + 1].isspace():
314
+ cursor += 1
315
+ return cursor
316
+
317
+
318
+ def _declaration_refused(detail: str) -> CatalogHarvestError:
319
+ return CatalogHarvestError("HARVEST_XML_DOCTYPE", "sdmx.payload", detail)
320
+
321
+
322
+ def _parse_bounded(payload: bytes): # type: ignore[no-untyped-def]
323
+ # Fed in chunks and drained between them, so the budget stops work in progress rather than
324
+ # after the whole tree has already been built.
325
+ parser = XMLPullParser(events=("start",))
326
+ root = None
327
+ elements = 0
328
+ try:
329
+ for offset in range(0, len(payload), FEED_CHUNK_BYTES):
330
+ parser.feed(payload[offset : offset + FEED_CHUNK_BYTES])
331
+ for _event, element in parser.read_events():
332
+ elements += 1
333
+ if root is None:
334
+ root = element
335
+ if elements > MAX_XML_ELEMENTS:
336
+ raise CatalogHarvestError(
337
+ "HARVEST_XML_ELEMENTS",
338
+ "sdmx.payload",
339
+ f"the document exceeds {MAX_XML_ELEMENTS} elements",
340
+ )
341
+ parser.close()
342
+ for _event, element in parser.read_events():
343
+ elements += 1
344
+ if root is None:
345
+ root = element
346
+ except ParseError as error:
347
+ raise CatalogHarvestError(
348
+ "HARVEST_SHAPE",
349
+ "sdmx.payload",
350
+ f"the response is not well-formed XML: {error}",
351
+ ) from None
352
+ if root is None:
353
+ raise CatalogHarvestError(
354
+ "HARVEST_SHAPE",
355
+ "sdmx.payload",
356
+ "the response carries no XML element",
357
+ )
358
+ return root
359
+
360
+
361
+ def _record(
362
+ dataflow, # type: ignore[no-untyped-def]
363
+ *,
364
+ position: int,
365
+ uri: str,
366
+ evidence: EvidenceReference,
367
+ skipped: list[str],
368
+ ) -> HarvestedRecord | None:
369
+ flow_id = _attribute(dataflow, "id")
370
+ agency = _attribute(dataflow, "agencyID")
371
+ version = _attribute(dataflow, "version") or "1.0"
372
+ if flow_id is None or agency is None:
373
+ skipped.append(f"<dataflow {position} without id or agencyID>")
374
+ return None
375
+ # The dataflow's natural coordinate is an identifier, so it degrades this record rather than
376
+ # being cut: an agency, id or version outside the record grammar would raise out of `parse` and
377
+ # take every other dataflow in the message with it.
378
+ record_id = usable_record_id(f"{agency}:{flow_id}({version})")
379
+ if record_id is None:
380
+ skipped.append(f"<dataflow {position} whose coordinate is not a usable record id>")
381
+ return None
382
+ landing = usable_https_uri(_structure_uri(uri, agency, flow_id, version))
383
+ if landing is None:
384
+ skipped.append(record_id)
385
+ return None
386
+ name = _localised(dataflow, _NAME_TAG) or flow_id
387
+ description = _localised(dataflow, _DESCRIPTION_TAG) or (
388
+ f"SDMX dataflow {agency}:{flow_id}({version}) published as a 2.1 structure message."
389
+ )
390
+ return HarvestedRecord(
391
+ protocol="sdmx",
392
+ record_id=record_id,
393
+ title=bounded_record_text(name, maximum=MAX_RECORD_TITLE),
394
+ description=bounded_record_text(description, maximum=MAX_RECORD_TEXT),
395
+ publisher=bounded_record_text(agency, maximum=MAX_RECORD_PUBLISHER),
396
+ landing_uri=landing,
397
+ data_formats=(),
398
+ evidence=evidence,
399
+ # A structure message states no licence. Saying so honestly maps to unclear rights, which
400
+ # escalates rather than admits.
401
+ license_id=None,
402
+ license_uri=None,
403
+ declared_updated_at=None,
404
+ )
405
+
406
+
407
+ def _structure_uri(uri: str, agency: str, flow_id: str, version: str) -> str | None:
408
+ if not uri.startswith("https://"):
409
+ return None
410
+ base = uri.split("?", 1)[0].rstrip("/")
411
+ marker = "/dataflow"
412
+ if marker in base:
413
+ base = base[: base.index(marker) + len(marker)]
414
+ return f"{base}/{agency}/{flow_id}/{version}"
415
+ return base
416
+
417
+
418
+ def _attribute(element, name: str) -> str | None: # type: ignore[no-untyped-def]
419
+ value = element.get(name)
420
+ if not isinstance(value, str):
421
+ return None
422
+ text = value.strip()
423
+ return text or None
424
+
425
+
426
+ def _localised(element, tag: str) -> str | None: # type: ignore[no-untyped-def]
427
+ """The English text of this child, or the first declared one, deterministically.
428
+
429
+ Falling back to "whichever came out of the iterator" would make the harvested title depend on
430
+ document order in a way nobody documented, so the fallback is explicit and ordered.
431
+ """
432
+
433
+ candidates = [child for child in element if child.tag == tag]
434
+ if not candidates:
435
+ return None
436
+ for child in candidates:
437
+ if (child.get(_LANG_ATTR) or "").lower().startswith("en"):
438
+ text = (child.text or "").strip()
439
+ if text:
440
+ return text
441
+ for child in candidates:
442
+ text = (child.text or "").strip()
443
+ if text:
444
+ return text
445
+ return None