mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,1000 @@
1
+ """Look at a file, or a public https address, before writing a Recipe against it.
2
+
3
+ `peek` is the first thing a person tries and the first thing a driving agent runs. It answers one
4
+ question -- what is in there -- and it answers it without building anything, approving anything, or
5
+ writing a single byte to disk.
6
+
7
+ Ordinary tabular bytes go through :func:`parse_tabular_bytes_for_display`. Reader-pinned bytes go
8
+ through the exact Toolbox coordinate's warm-up and decode inside the same clean room as a Build;
9
+ for a URL, retrieval and decode are one operation so container bytes never return to this process.
10
+ That is deliberate and load-bearing: the size budgets, archive and executable refusal, strict
11
+ UTF-8 rule, and row and cell limits come along for free, with no second unconfined parser. There is
12
+ no ``open()``, no ``csv`` import, and no ``json`` import in this module, and
13
+ `tests/test_ux_peek.py` reads this file's own source to keep it that way. Local bytes are fetched
14
+ through :mod:`ux.plain_file`, the one open-then-fstat rule every command that takes a path from a
15
+ person shares, so peek has no reading rule of its own either.
16
+
17
+ The one difference is the display rendering. A Build's Reader refuses a value the canonical rule
18
+ cannot hold exactly -- a date, a time, a duration, a decimal, a not-a-number -- because what a
19
+ Build reads gets sealed. A peek seals nothing and writes nothing, so it renders those values into
20
+ their exact text instead of refusing them, and reports the column as the kind of column it really
21
+ is. That is why peek opens a Build's own ``table.parquet``, and every timestamped source the
22
+ product exists to work with, rather than turning them away.
23
+
24
+ What peek reports is deliberately small: the columns, the type observed in each, how many values
25
+ are missing, how many are distinct, the lowest and highest, and the first few rows. Nothing here
26
+ infers a type, coerces a value, or refuses a messy cell. Peek exists to look at data nobody has
27
+ cleaned yet, so a peek that refuses messy data would not be a peek.
28
+
29
+ Column types are reported in the product's own vocabulary -- ``string``, ``int64``, ``float64``,
30
+ ``boolean``, ``date``, ``timestamp_utc`` -- which is the set a Recipe accepts and the set
31
+ ``mr-data show`` and ``mr-data inventory`` print. That is not an inference: it is the observed kind
32
+ under the name the rest of the product gives it, so a type read out of a peek is a type ``author``
33
+ takes. A column that holds no single kind, or a kind none of the six covers, is never dressed up as
34
+ one of them; see :func:`plain_column_type`.
35
+
36
+ ``timestamp_utc`` is the narrowest of the six and is not a synonym for "a timestamp": it is bound
37
+ to ``timestamp[us, tz=UTC]``, and a Recipe refuses the type unless the timezone it declares is UTC.
38
+ So a column whose timestamps carry no zone, or carry one that is not UTC, is reported as the kind
39
+ it is rather than as that one -- the sample rows printed two lines below say the same thing, and
40
+ the two may not disagree.
41
+ """
42
+
43
+ from __future__ import annotations
44
+
45
+ import math
46
+ import os
47
+ import tempfile
48
+ from collections.abc import Mapping
49
+ from dataclasses import dataclass, replace
50
+ from datetime import datetime, timedelta
51
+ from pathlib import Path
52
+ from typing import Any
53
+ from urllib.parse import urlsplit
54
+
55
+ from mostlyright.data_harness.acquisition.http import (
56
+ PinnedHttpsRetriever,
57
+ PinnedTransport,
58
+ RetrievalLimits,
59
+ StdlibPinnedTransport,
60
+ reader_retrieval_limits,
61
+ )
62
+ from mostlyright.data_harness.acquisition.parsing import (
63
+ _FORMAT_MEDIA_TYPES,
64
+ _FORMAT_SUFFIXES,
65
+ ParsedTable,
66
+ ParseLimits,
67
+ column_types,
68
+ parse_tabular_bytes_for_display,
69
+ )
70
+ from mostlyright.data_harness.acquisition.sandbox import CrawlerSandbox, SandboxResult
71
+ from mostlyright.data_harness.acquisition.url_policy import (
72
+ AcquisitionSecurityError,
73
+ EgressPolicy,
74
+ Resolver,
75
+ SystemResolver,
76
+ _normalize_hostname,
77
+ )
78
+ from mostlyright.data_harness.canonical import sha256_bytes
79
+ from mostlyright.data_harness.readers.contracts import ReaderError, ReaderPin, identifier, semver
80
+ from mostlyright.data_harness.readers.registry import TOOLBOX
81
+ from mostlyright.data_harness.readers.samples import DecodedFacts, ReaderSample, warm_up
82
+ from mostlyright.data_harness.ux.path_kind import UNKNOWN_KIND, name_of_kind, presence_at
83
+ from mostlyright.data_harness.ux.plain_file import (
84
+ DANGLING,
85
+ MISSING,
86
+ PlainFileRefusal,
87
+ open_plain_file,
88
+ read_bounded,
89
+ )
90
+
91
+ # How many rows come back by default, and the ceiling on asking for more. A peek is a look, not an
92
+ # export: anything larger is a Build.
93
+ DEFAULT_SAMPLE_ROWS = 5
94
+ MAX_SAMPLE_ROWS = 100
95
+
96
+ # Derived from the Reader's own allowlist rather than re-typed here, so the list of formats peek
97
+ # offers cannot drift from the list of formats the harness can actually open.
98
+ PEEK_FORMATS: tuple[str, ...] = tuple(sorted(_FORMAT_SUFFIXES))
99
+ _SUFFIX_FORMATS: dict[str, str] = {
100
+ suffix: data_format
101
+ for data_format, suffixes in _FORMAT_SUFFIXES.items()
102
+ for suffix in sorted(suffixes)
103
+ }
104
+ _MEDIA_FORMATS: dict[str, str] = {
105
+ media_type: data_format
106
+ for data_format, media_types in _FORMAT_MEDIA_TYPES.items()
107
+ for media_type in sorted(media_types)
108
+ }
109
+ _READABLE = ", ".join(PEEK_FORMATS)
110
+ _JSON_READER_COORDINATE = ("json.tabular", "1.0.0")
111
+ _INHERITED_MEMORY_CGROUP_ROOT_FD = 3
112
+ _EXTERNAL_NETWORK_POLICY_ATTESTATION_ENV = "MOSTLYRIGHT_EXTERNAL_NETWORK_POLICY_ATTESTATION"
113
+
114
+
115
+ def _reader_sandbox(staging_root: Path) -> CrawlerSandbox:
116
+ memory_fd = None
117
+ if os.sys.platform == "linux":
118
+ try:
119
+ os.fstat(_INHERITED_MEMORY_CGROUP_ROOT_FD)
120
+ inherited = os.get_inheritable(_INHERITED_MEMORY_CGROUP_ROOT_FD)
121
+ except OSError:
122
+ pass
123
+ else:
124
+ if inherited:
125
+ memory_fd = _INHERITED_MEMORY_CGROUP_ROOT_FD
126
+ return CrawlerSandbox(
127
+ staging_root=staging_root,
128
+ external_network_policy_attestation=os.environ.get(
129
+ _EXTERNAL_NETWORK_POLICY_ATTESTATION_ENV
130
+ ),
131
+ memory_cgroup_root_fd=memory_fd,
132
+ )
133
+
134
+
135
+ def _reader_pin(
136
+ coordinate: tuple[str, str] | None,
137
+ options: Mapping[str, Any],
138
+ ) -> tuple[Any, dict[str, Any], str]:
139
+ family = TOOLBOX.resolve(*(coordinate or _JSON_READER_COORDINATE))
140
+ admitted = dict(family.validate_options(options))
141
+ resolved = ReaderPin(family.family_id, family.family_version, admitted)
142
+ return (
143
+ family,
144
+ {
145
+ "family_id": family.family_id,
146
+ "family_version": family.family_version,
147
+ "decode_options": admitted,
148
+ },
149
+ resolved.options_digest,
150
+ )
151
+
152
+
153
+ def _normalise_reader_request(
154
+ coordinate: tuple[str, str] | None,
155
+ options: Mapping[str, Any] | None,
156
+ data_format: str | None,
157
+ ) -> tuple[tuple[str, str] | None, Mapping[str, Any] | None]:
158
+ """Choose Reader mode once; an exact coordinate can never fall through to direct parsing."""
159
+
160
+ reader_requested = coordinate is not None or options is not None
161
+ if not reader_requested:
162
+ return None, None
163
+ if data_format is not None:
164
+ raise ReaderError(
165
+ "READER_OPTIONS",
166
+ "peek.format",
167
+ "data_format is for direct tabular preview and cannot be combined with a Reader",
168
+ )
169
+ if options is not None and not isinstance(options, Mapping):
170
+ raise ReaderError("READER_OPTIONS", "peek.reader_options", "must be an object")
171
+ if coordinate is None:
172
+ resolved_coordinate = _JSON_READER_COORDINATE
173
+ else:
174
+ if not isinstance(coordinate, tuple) or len(coordinate) != 2:
175
+ raise ReaderError(
176
+ "READER_CONTRACT",
177
+ "peek.reader",
178
+ "must be an exact (family, version) Reader coordinate",
179
+ )
180
+ resolved_coordinate = (
181
+ identifier(coordinate[0], "peek.reader.family"),
182
+ semver(coordinate[1], "peek.reader.version"),
183
+ )
184
+ return resolved_coordinate, options or {}
185
+
186
+
187
+ def _require_reader_result(
188
+ result: SandboxResult,
189
+ *,
190
+ operation: str,
191
+ family: Any,
192
+ pin: Mapping[str, Any],
193
+ options_digest: str,
194
+ retrieval: bool,
195
+ ) -> None:
196
+ """Bind a sandbox answer to the exact decode the coordinator requested."""
197
+
198
+ expected_identity = (
199
+ family.family_id,
200
+ family.family_version,
201
+ options_digest,
202
+ )
203
+ returned_identity = (
204
+ result.decode_family_id,
205
+ result.decode_family_version,
206
+ result.decode_options_digest,
207
+ )
208
+ if (
209
+ result.operation != operation
210
+ or returned_identity != expected_identity
211
+ or result.decode_flags is None
212
+ or result.content is None
213
+ or result.parsed is None
214
+ or pin.get("family_id") != family.family_id
215
+ or pin.get("family_version") != family.family_version
216
+ ):
217
+ raise AcquisitionSecurityError(
218
+ "SANDBOX_RESULT",
219
+ "Reader preview result does not bind the exact requested Reader decode",
220
+ )
221
+ if retrieval and (
222
+ result.fetched_content_sha256 is None
223
+ or not result.final_url
224
+ or result.transport_evidence_digest is None
225
+ ):
226
+ raise AcquisitionSecurityError(
227
+ "SANDBOX_RESULT",
228
+ "Reader URL preview result omits required retrieval evidence",
229
+ )
230
+
231
+
232
+ def _warm_up_reader(sandbox: CrawlerSandbox, family: Any) -> None:
233
+ """Prove the exact Reader coordinate on its packaged sample through this clean room."""
234
+
235
+ def decode(sample: ReaderSample) -> DecodedFacts:
236
+ sample_pin = ReaderPin(
237
+ sample.family_id,
238
+ sample.family_version,
239
+ dict(sample.decode_options),
240
+ )
241
+ result = sandbox.decode_and_parse(
242
+ request_id=f"peek.warmup.{sample.family_id}.{sample.family_version}".lower(),
243
+ content=sample.content,
244
+ reader_pin={
245
+ "family_id": sample.family_id,
246
+ "family_version": sample.family_version,
247
+ "decode_options": dict(sample.decode_options),
248
+ },
249
+ output_format=family.output_format,
250
+ limits=ParseLimits(),
251
+ )
252
+ _require_reader_result(
253
+ result,
254
+ operation="decode_and_parse",
255
+ family=family,
256
+ pin={
257
+ "family_id": sample.family_id,
258
+ "family_version": sample.family_version,
259
+ "decode_options": dict(sample.decode_options),
260
+ },
261
+ options_digest=sample_pin.options_digest,
262
+ retrieval=False,
263
+ )
264
+ assert result.content is not None
265
+ assert result.parsed is not None
266
+ assert result.decode_flags is not None
267
+ return DecodedFacts(
268
+ sha256_bytes(result.content),
269
+ len(result.parsed.rows),
270
+ tuple(result.parsed.columns),
271
+ tuple(result.decode_flags),
272
+ )
273
+
274
+ try:
275
+ warm_up(family.family_id, family.family_version, decode=decode)
276
+ except ReaderError as error:
277
+ # A preview must distinguish "this certified Reader is unavailable on this host" from
278
+ # "the Reader or source is invalid". Warm-up intentionally wraps runtime failures for
279
+ # Build execution, but peek is the capability-inspection surface: preserve the exact
280
+ # typed platform refusal there so discovery does not misclassify the source as unfit.
281
+ cause = error.__cause__
282
+ if isinstance(cause, AcquisitionSecurityError) and cause.code in {
283
+ "SANDBOX_MEMORY_BOUNDARY",
284
+ "SANDBOX_OS_BOUNDARY",
285
+ }:
286
+ raise cause from error
287
+ raise
288
+
289
+
290
+ def _peek_reader_bytes(
291
+ content: bytes,
292
+ *,
293
+ family: Any,
294
+ pin: Mapping[str, Any],
295
+ options_digest: str,
296
+ limits: ParseLimits,
297
+ sandbox: CrawlerSandbox,
298
+ ) -> ParsedTable:
299
+ _warm_up_reader(sandbox, family)
300
+ result = sandbox.decode_and_parse(
301
+ request_id=f"peek.{family.family_id}.{family.family_version}",
302
+ content=content,
303
+ reader_pin=pin,
304
+ output_format=family.output_format,
305
+ limits=limits,
306
+ )
307
+ _require_reader_result(
308
+ result,
309
+ operation="decode_and_parse",
310
+ family=family,
311
+ pin=pin,
312
+ options_digest=options_digest,
313
+ retrieval=False,
314
+ )
315
+ assert result.parsed is not None
316
+ return result.parsed
317
+
318
+
319
+ # ------------------------------------------------------------------------------------------------
320
+ # What a column's type is called
321
+ # ------------------------------------------------------------------------------------------------
322
+ #
323
+ # The Reader reports what it saw in Python's own words -- `str`, `int`, `float`, `int|str`. Those
324
+ # are not words this product uses anywhere else: `mr-data show` prints `string` and `float64`,
325
+ # `mr-data inventory` publishes the same six, and a Recipe accepts only those six. peek is the
326
+ # look-before-you-author step, so an agent that carries a peek answer straight into `author` was
327
+ # being handed a value the Recipe contract rejects, with a refusal that names neither the offending
328
+ # value nor a legal one.
329
+ #
330
+ # So the observed kind is rendered into the product's own column-type vocabulary here. Nothing is
331
+ # inferred or coerced: this is the same fact under the name the rest of the product gives it. A
332
+ # column that holds no single kind, or a kind no Recipe column type covers, is never dressed up as
333
+ # one -- it gets a named marker instead, and the markers are written out below rather than being
334
+ # whatever text happened to come back.
335
+ OBSERVED_LOGICAL_TYPES: dict[str, str] = {
336
+ "str": "string",
337
+ "int": "int64",
338
+ "float": "float64",
339
+ "bool": "boolean",
340
+ "date": "date",
341
+ "datetime": "timestamp_utc",
342
+ }
343
+
344
+ # `timestamp_utc` is not this product's word for "a timestamp". It is bound to
345
+ # `timestamp[us, tz=UTC]` (`preparation/contracts.PHYSICAL_TYPES`), a Recipe refuses the type
346
+ # unless its declared timezone is UTC (`recipe.py`), and the contract reader accepts only UTC
347
+ # instants. A column of naive timestamps, or of timestamps at some other offset, is none of those
348
+ # -- so reporting `timestamp_utc` for every datetime column asserted UTC about values the sample
349
+ # rows two lines above show are not, and handed an agent a type `author` then refuses.
350
+ #
351
+ # The Reader names the kind after the Python type it rendered, and every ``datetime.datetime``
352
+ # renders under one name, so the zone is read back off the rendered text here. These two markers
353
+ # are observed kinds like any other: no Recipe column type covers them, so
354
+ # :func:`plain_column_type` reports them as exactly that, and a mixed column lists them beside the
355
+ # kinds it holds.
356
+ TIMESTAMP_NO_ZONE = "timestamp with no time zone"
357
+ TIMESTAMP_OTHER_ZONE = "timestamp at an offset other than UTC"
358
+
359
+ _OBSERVED_DATETIME = "datetime"
360
+
361
+ # The three answers that are not a column type, spelled out. `null` inside a column that also holds
362
+ # one kind is not one of them: how many values are missing is already its own reported number, so a
363
+ # column of whole numbers with gaps in it is a whole-number column.
364
+ NO_ROWS = "no rows to read"
365
+ ALL_MISSING = "missing in every row"
366
+ MIXED_PREFIX = "mixed: "
367
+ NOT_A_COLUMN_TYPE_PREFIX = "no Recipe column type for this: "
368
+
369
+ # The literal the Reader uses for a column it saw no rows of, and for an absent value.
370
+ _OBSERVED_EMPTY = "empty"
371
+ _OBSERVED_NULL = "null"
372
+
373
+
374
+ def plain_column_type(observed: str) -> str:
375
+ """The Reader's observed kind, in the vocabulary the rest of the product uses.
376
+
377
+ ``observed`` is what :func:`acquisition.parsing.column_types` returns: one Python type name, or
378
+ several joined by ``|`` for a column holding more than one kind, or ``empty``.
379
+ """
380
+
381
+ if observed == _OBSERVED_EMPTY:
382
+ return NO_ROWS
383
+ kinds = [kind for kind in observed.split("|") if kind != _OBSERVED_NULL]
384
+ if not kinds:
385
+ return ALL_MISSING
386
+ named = sorted(
387
+ OBSERVED_LOGICAL_TYPES[kind] if kind in OBSERVED_LOGICAL_TYPES else kind for kind in kinds
388
+ )
389
+ if len(named) > 1:
390
+ return MIXED_PREFIX + " and ".join(named)
391
+ if kinds[0] not in OBSERVED_LOGICAL_TYPES:
392
+ return NOT_A_COLUMN_TYPE_PREFIX + named[0]
393
+ return named[0]
394
+
395
+
396
+ @dataclass(frozen=True)
397
+ class PeekColumn:
398
+ """One column as it is, before anyone has cleaned it."""
399
+
400
+ name: str
401
+ type: str
402
+ null_count: int
403
+ distinct_count: int
404
+ minimum: str | None
405
+ maximum: str | None
406
+
407
+ def to_dict(self) -> dict[str, Any]:
408
+ return {
409
+ "name": self.name,
410
+ "type": self.type,
411
+ "null_count": self.null_count,
412
+ "distinct_count": self.distinct_count,
413
+ "minimum": self.minimum,
414
+ "maximum": self.maximum,
415
+ }
416
+
417
+
418
+ @dataclass(frozen=True)
419
+ class PeekResult:
420
+ """What one look found: where it looked, what the columns are, and the first rows."""
421
+
422
+ source: str
423
+ origin: str
424
+ data_format: str
425
+ input_sha256: str
426
+ schema_digest: str
427
+ row_count: int
428
+ columns: tuple[PeekColumn, ...]
429
+ sample: tuple[tuple[Any, ...], ...]
430
+ media_type: str | None = None
431
+ final_url: str | None = None
432
+ content_sha256: str | None = None
433
+ fetched_from: str | None = None
434
+ note: str | None = None
435
+ reader_coordinate: str | None = None
436
+ reader_options_digest: str | None = None
437
+
438
+ def to_dict(self) -> dict[str, Any]:
439
+ """The one payload both renderings are built from.
440
+
441
+ No key here is taken from the data. A column is called ``column 1`` and carries its real
442
+ name as a value, and a sample row is the row's values in column order. That is deliberate:
443
+ the plain rendering gives a payload key its plain label, so a column named ``max_temp_c``
444
+ used as a key would be shown to a person as ``Max temp c`` -- a name the file does not
445
+ have. Facts about the data go in values, where nothing rewrites them.
446
+ """
447
+
448
+ payload: dict[str, Any] = {
449
+ "status": "peeked",
450
+ "source": self.source,
451
+ "origin": self.origin,
452
+ "data_format": self.data_format,
453
+ "input_sha256": self.input_sha256,
454
+ "schema_digest": self.schema_digest,
455
+ "row_count": self.row_count,
456
+ "columns": [column.name for column in self.columns],
457
+ "schema": self._schema_payload(),
458
+ "sample": self._sample_payload(),
459
+ }
460
+ if self.origin == "url":
461
+ payload["url"] = self.final_url
462
+ payload["fetched_from"] = self.fetched_from
463
+ payload["content_sha256"] = self.content_sha256
464
+ if self.media_type is not None:
465
+ payload["media_type"] = self.media_type
466
+ payload["saved_to_disk"] = False
467
+ if self.note is not None:
468
+ payload["note"] = self.note
469
+ if self.reader_coordinate is not None:
470
+ payload["reader_coordinate"] = self.reader_coordinate
471
+ payload["reader_options_digest"] = self.reader_options_digest
472
+ payload["reader_execution_status"] = "available"
473
+ return payload
474
+
475
+ def _schema_payload(self) -> dict[str, dict[str, Any]]:
476
+ width = len(str(max(len(self.columns), 1)))
477
+ return {
478
+ f"column {index:0{width}d}": column.to_dict()
479
+ for index, column in enumerate(self.columns, start=1)
480
+ }
481
+
482
+ def _sample_payload(self) -> dict[str, list[Any]]:
483
+ """Each row as its values in column order, the way a table has always been written."""
484
+
485
+ width = len(str(max(len(self.sample), 1)))
486
+ return {
487
+ f"row {index:0{width}d}": [_json_safe(value) for value in row]
488
+ for index, row in enumerate(self.sample, start=1)
489
+ }
490
+
491
+
492
+ def peek_bytes(
493
+ content: bytes,
494
+ *,
495
+ filename: str,
496
+ data_format: str | None = None,
497
+ sample_rows: int = DEFAULT_SAMPLE_ROWS,
498
+ limits: ParseLimits | None = None,
499
+ source: str | None = None,
500
+ reader_coordinate: tuple[str, str] | None = None,
501
+ reader_options: Mapping[str, Any] | None = None,
502
+ reader_sandbox: CrawlerSandbox | None = None,
503
+ ) -> PeekResult:
504
+ """Look at exact bytes. The only byte reader in this module, by design."""
505
+
506
+ reader_coordinate, reader_options = _normalise_reader_request(
507
+ reader_coordinate, reader_options, data_format
508
+ )
509
+ rows_wanted = _bounded_sample_rows(sample_rows)
510
+ original_sha256 = sha256_bytes(content)
511
+ resolved_reader_coordinate = None
512
+ reader_options_digest = None
513
+ if reader_options is not None:
514
+ family, pin, reader_options_digest = _reader_pin(reader_coordinate, reader_options)
515
+ resolved_reader_coordinate = f"{family.family_id}@{family.family_version}"
516
+ if reader_sandbox is not None:
517
+ table = _peek_reader_bytes(
518
+ content,
519
+ family=family,
520
+ pin=pin,
521
+ options_digest=reader_options_digest,
522
+ limits=limits or ParseLimits(),
523
+ sandbox=reader_sandbox,
524
+ )
525
+ else:
526
+ with tempfile.TemporaryDirectory(prefix="mr-peek-reader-") as temporary:
527
+ table = _peek_reader_bytes(
528
+ content,
529
+ family=family,
530
+ pin=pin,
531
+ options_digest=reader_options_digest,
532
+ limits=limits or ParseLimits(),
533
+ sandbox=_reader_sandbox(Path(temporary).resolve()),
534
+ )
535
+ else:
536
+ resolved_format, media_type = _format_for(filename, data_format)
537
+ table = parse_tabular_bytes_for_display(
538
+ content,
539
+ data_format=resolved_format,
540
+ media_type=media_type,
541
+ filename=filename,
542
+ limits=limits or ParseLimits(),
543
+ )
544
+ return PeekResult(
545
+ source=source if source is not None else filename,
546
+ origin="file",
547
+ data_format=table.data_format,
548
+ input_sha256=original_sha256,
549
+ schema_digest=table.schema_digest,
550
+ row_count=len(table.rows),
551
+ columns=_summarise(table),
552
+ sample=table.rows[:rows_wanted],
553
+ reader_coordinate=resolved_reader_coordinate if reader_options is not None else None,
554
+ reader_options_digest=reader_options_digest,
555
+ )
556
+
557
+
558
+ def peek_path(
559
+ path: Path | str,
560
+ *,
561
+ data_format: str | None = None,
562
+ sample_rows: int = DEFAULT_SAMPLE_ROWS,
563
+ limits: ParseLimits | None = None,
564
+ reader_coordinate: tuple[str, str] | None = None,
565
+ reader_options: Mapping[str, Any] | None = None,
566
+ reader_sandbox: CrawlerSandbox | None = None,
567
+ ) -> PeekResult:
568
+ """Look at one file on this machine.
569
+
570
+ The size is checked before the bytes are loaded, so a file too large to open is refused rather
571
+ than read into memory first -- and it is checked on the open descriptor, through
572
+ :mod:`ux.plain_file`, which is the rule the rest of this tree opens anything with. Asking the
573
+ path instead got two answers wrong at once. ``st_size`` is 0 for a pipe and a character device,
574
+ so the cap governed nothing there, and ``Path.is_file`` is false for both, so a peek at either
575
+ one was refused with the sentence written for a folder -- a statement about what is at the path
576
+ that is simply not true, and advice ("check the spelling") aimed at a caller who typed the path
577
+ they meant. The bound also has to govern the read that follows it, so a file that grows between
578
+ the two is caught by the read rather than trusted from a moment earlier.
579
+ """
580
+
581
+ reader_coordinate, reader_options = _normalise_reader_request(
582
+ reader_coordinate, reader_options, data_format
583
+ )
584
+ target = Path(path)
585
+ rows_wanted = _bounded_sample_rows(sample_rows)
586
+ budgets = limits or ParseLimits()
587
+ try:
588
+ descriptor, size = open_plain_file(target)
589
+ except PlainFileRefusal as refusal:
590
+ # Two codes rather than two wordings of one, because the two findings contradict each
591
+ # other: `PEEK_TARGET` says nothing is there and tells the caller to check the spelling,
592
+ # which is advice about a mistake somebody pointing at a real pipe did not make.
593
+ # `OUTPUT_PARENT_ABSENT` and `OUTPUT_PARENT_INVALID` are two codes for the same reason.
594
+ # Raised here rather than returned from a helper: the sweep that proves every typed code
595
+ # names its fix reads `raise` statements out of the live source.
596
+ if refusal.reason == DANGLING:
597
+ # A dangling link is present even though opening its target raises
598
+ # `FileNotFoundError`, so it must not be translated as an absent input. The noun comes
599
+ # from the `st_mode` retained from the non-following lookup, keeping the sentence and
600
+ # finding bound to one observation.
601
+ raise AcquisitionSecurityError(
602
+ "PEEK_TARGET_DANGLING",
603
+ f"there is {name_of_kind(refusal.mode) or UNKNOWN_KIND} at {refusal.path}, "
604
+ "and it leads nowhere",
605
+ ) from None
606
+ if refusal.reason == MISSING:
607
+ raise AcquisitionSecurityError(
608
+ "PEEK_TARGET",
609
+ f"there is {presence_at(target) or UNKNOWN_KIND} at {target} to look at",
610
+ ) from None
611
+ raise AcquisitionSecurityError(
612
+ "PEEK_TARGET_NOT_A_FILE",
613
+ f"{target} is not a file: a peek reads a file, not a folder, a pipe, or a device",
614
+ ) from None
615
+ try:
616
+ # Resolve the format before the read as well, so an unreadable file is named as unreadable
617
+ # rather than loaded and then refused. Opening the descriptor has loaded nothing yet.
618
+ if reader_options is None:
619
+ _format_for(target.name, data_format)
620
+ if size > budgets.max_input_bytes:
621
+ raise AcquisitionSecurityError(
622
+ "PEEK_SIZE",
623
+ f"{target} holds {size} bytes, above the {budgets.max_input_bytes} byte "
624
+ "limit on anything this harness opens",
625
+ )
626
+ content = read_bounded(descriptor, max_bytes=budgets.max_input_bytes)
627
+ finally:
628
+ os.close(descriptor)
629
+ if len(content) > budgets.max_input_bytes:
630
+ raise AcquisitionSecurityError(
631
+ "PEEK_SIZE",
632
+ f"{target} holds more than {budgets.max_input_bytes} bytes, above the limit on "
633
+ "anything this harness opens",
634
+ )
635
+ return peek_bytes(
636
+ content,
637
+ filename=target.name,
638
+ data_format=data_format,
639
+ sample_rows=rows_wanted,
640
+ limits=budgets,
641
+ source=str(target),
642
+ reader_coordinate=reader_coordinate,
643
+ reader_options=reader_options,
644
+ reader_sandbox=reader_sandbox,
645
+ )
646
+
647
+
648
+ def peek_url(
649
+ url: str,
650
+ *,
651
+ allow_redirect_hosts: tuple[str, ...] = (),
652
+ data_format: str | None = None,
653
+ sample_rows: int = DEFAULT_SAMPLE_ROWS,
654
+ limits: ParseLimits | None = None,
655
+ transport: PinnedTransport | None = None,
656
+ resolver: Resolver | None = None,
657
+ reader_coordinate: tuple[str, str] | None = None,
658
+ reader_options: Mapping[str, Any] | None = None,
659
+ reader_sandbox: CrawlerSandbox | None = None,
660
+ ) -> PeekResult:
661
+ """Fetch one public https address and look at what came back, saving nothing.
662
+
663
+ The fetch goes through the harness's one Courier: the same address validation, the same
664
+ pinned-peer connection, the same redirect and byte budgets a Recipe's own fetches use. Nothing
665
+ here widens that; the only thing peek supplies is which single host is allowed.
666
+
667
+ ``transport`` and ``resolver`` exist so a test can hand in a recorded exchange instead of
668
+ reaching the network. Neither is reachable from the command line.
669
+ """
670
+
671
+ reader_coordinate, reader_options = _normalise_reader_request(
672
+ reader_coordinate, reader_options, data_format
673
+ )
674
+ rows_wanted = _bounded_sample_rows(sample_rows)
675
+ hostname = _hostname_of(url)
676
+ # The allowlist of one is derived from an explicit argument -- the address the person or their
677
+ # agent typed -- and never from a config file and never from the content of a source. That is
678
+ # what keeps the fail-closed egress model intact: peek narrows it to a single host rather than
679
+ # widening it, and `--allow-redirect-host` names any further host explicitly. Every entry
680
+ # is written in the egress rule's own spelling, the named host and the redirect hosts alike:
681
+ # an entry spelled any other way is either refused as non-canonical or silently never matched.
682
+ allowed_hostnames = (
683
+ hostname,
684
+ *(_normalize_hostname(host) for host in allow_redirect_hosts),
685
+ )
686
+ if reader_options is not None:
687
+ family, pin, reader_options_digest = _reader_pin(reader_coordinate, reader_options)
688
+ budgets = limits or ParseLimits()
689
+
690
+ def retrieve(sandbox: CrawlerSandbox) -> PeekResult:
691
+ _warm_up_reader(sandbox, family)
692
+ result = sandbox.retrieve_decode_and_parse(
693
+ request_id=f"peek.url.{family.family_id}.{family.family_version}",
694
+ url=url,
695
+ allowed_hostnames=allowed_hostnames,
696
+ reader_pin=pin,
697
+ output_format=family.output_format,
698
+ parse_limits=budgets,
699
+ retrieval_limits=reader_retrieval_limits(
700
+ RetrievalLimits(),
701
+ family_id=family.family_id,
702
+ family_version=family.family_version,
703
+ ),
704
+ )
705
+ _require_reader_result(
706
+ result,
707
+ operation="retrieve_decode_and_parse",
708
+ family=family,
709
+ pin=pin,
710
+ options_digest=reader_options_digest,
711
+ retrieval=True,
712
+ )
713
+ assert result.parsed is not None
714
+ assert result.fetched_content_sha256 is not None
715
+ return PeekResult(
716
+ source=url,
717
+ origin="url",
718
+ data_format=result.parsed.data_format,
719
+ input_sha256=result.fetched_content_sha256,
720
+ schema_digest=result.parsed.schema_digest,
721
+ row_count=len(result.parsed.rows),
722
+ columns=_summarise(result.parsed),
723
+ sample=result.parsed.rows[:rows_wanted],
724
+ # The combined clean-room result exposes the normalized CSV media type, not the
725
+ # original response header. Do not fabricate source metadata from decode settings.
726
+ media_type=None,
727
+ final_url=result.final_url,
728
+ content_sha256=result.fetched_content_sha256,
729
+ fetched_from=hostname,
730
+ note=(
731
+ f"The fetched media type was enforced against {family.family_id}@"
732
+ f"{family.family_version} inside the "
733
+ "clean room; the original response header is not returned by that operation."
734
+ ),
735
+ reader_coordinate=f"{family.family_id}@{family.family_version}",
736
+ reader_options_digest=reader_options_digest,
737
+ )
738
+
739
+ if reader_sandbox is not None:
740
+ return retrieve(reader_sandbox)
741
+ with tempfile.TemporaryDirectory(prefix="mr-peek-reader-") as temporary:
742
+ return retrieve(_reader_sandbox(Path(temporary).resolve()))
743
+
744
+ policy = EgressPolicy(allowed_hostnames=allowed_hostnames)
745
+ retriever = PinnedHttpsRetriever(
746
+ resolver=resolver if resolver is not None else SystemResolver(),
747
+ egress_policy=policy,
748
+ transport=transport if transport is not None else StdlibPinnedTransport(),
749
+ limits=RetrievalLimits(),
750
+ )
751
+ retrieved = retriever.retrieve(url)
752
+ resolved_format, filename, note = _format_for_response(
753
+ final_url=retrieved.final_url,
754
+ media_type=retrieved.media_type,
755
+ declared=data_format,
756
+ )
757
+ result = peek_bytes(
758
+ retrieved.content,
759
+ filename=filename,
760
+ data_format=resolved_format,
761
+ sample_rows=rows_wanted,
762
+ limits=limits,
763
+ reader_coordinate=reader_coordinate,
764
+ reader_options=reader_options,
765
+ )
766
+ return replace(
767
+ result,
768
+ source=url,
769
+ origin="url",
770
+ media_type=retrieved.media_type,
771
+ final_url=retrieved.final_url,
772
+ content_sha256=retrieved.content_sha256,
773
+ fetched_from=hostname,
774
+ note=note,
775
+ )
776
+
777
+
778
+ def _hostname_of(url: str) -> str:
779
+ """The hostname to build the allowlist from. Validating the address is not this job.
780
+
781
+ ``validate_public_https_url`` is the validator, and it runs on every hop inside the Courier.
782
+ Reading the host out here is only how the allowlist gets its single entry, so anything odd
783
+ about the address is still refused there rather than quietly accepted here.
784
+
785
+ The spelling is the egress rule's own: ``_normalize_hostname`` is what ``EgressPolicy`` measures
786
+ an entry against and what the validator normalizes every hop to, so this allowlist of one is
787
+ written in the same spelling it will later be compared in. Lower-casing the host here instead
788
+ was a second rule that agreed with that one on ordinary addresses and disagreed on two:
789
+ an internationalized address (whose entry was refused as non-canonical before a byte was
790
+ fetched) and a trailing-dot name (whose entry was stored unnormalized and never matched), both
791
+ turned away with a message about an allowlist a `peek` invocation does not have.
792
+ """
793
+
794
+ try:
795
+ hostname = urlsplit(url).hostname
796
+ except ValueError:
797
+ hostname = None
798
+ if not hostname:
799
+ raise AcquisitionSecurityError(
800
+ "PEEK_TARGET",
801
+ f"{url} is not an address with a host in it",
802
+ )
803
+ return _normalize_hostname(hostname)
804
+
805
+
806
+ def _format_for_response(
807
+ *,
808
+ final_url: str,
809
+ media_type: str,
810
+ declared: str | None,
811
+ ) -> tuple[str, str, str | None]:
812
+ """Decide what was served, and name a file the Reader's own suffix check can look at.
813
+
814
+ Order: what the caller said, then what the server said it was serving, then the name at the
815
+ end of the address. The media type only picks which Reader opens the bytes; that Reader then
816
+ re-checks the magic bytes, the encoding, and the budgets, so a content type that lies gets a
817
+ refusal rather than a wrong reading.
818
+ """
819
+
820
+ basename = _basename_of(final_url)
821
+ suffix_format = _SUFFIX_FORMATS.get(_suffix_of(basename))
822
+ data_format = declared or _MEDIA_FORMATS.get(media_type) or suffix_format
823
+ if data_format is None:
824
+ raise AcquisitionSecurityError("PEEK_FORMAT", _unreadable(final_url))
825
+ if data_format not in PEEK_FORMATS:
826
+ raise AcquisitionSecurityError("PEEK_FORMAT", _unreadable(data_format))
827
+ if suffix_format is not None:
828
+ # The address names a file, so the Reader checks that name against the bytes as usual.
829
+ return data_format, basename, None
830
+ stand_in = f"download{_suffix_for(data_format)}"
831
+ reason = (
832
+ f"the server did not name a file, so this was read as {data_format} "
833
+ f"because that is what the media type ({media_type}) says it is"
834
+ )
835
+ return data_format, stand_in, reason
836
+
837
+
838
+ def _basename_of(url: str) -> str:
839
+ return urlsplit(url).path.rsplit("/", 1)[-1]
840
+
841
+
842
+ def _suffix_for(data_format: str) -> str:
843
+ """The suffix the Reader accepts for this format, taken from the Reader's own table."""
844
+
845
+ return min(_FORMAT_SUFFIXES[data_format])
846
+
847
+
848
+ def _summarise(table: ParsedTable) -> tuple[PeekColumn, ...]:
849
+ """One :class:`PeekColumn` per column, over rows the Reader has already bounded."""
850
+
851
+ types = column_types(table)
852
+ columns: list[PeekColumn] = []
853
+ for index, name in enumerate(table.columns):
854
+ values = [row[index] for row in table.rows]
855
+ present = [value for value in values if value is not None]
856
+ minimum, maximum = _extremes(present)
857
+ columns.append(
858
+ PeekColumn(
859
+ name=name,
860
+ type=plain_column_type(_zoned(types[index], present)),
861
+ null_count=len(values) - len(present),
862
+ distinct_count=_distinct(present),
863
+ minimum=minimum,
864
+ maximum=maximum,
865
+ )
866
+ )
867
+ return tuple(columns)
868
+
869
+
870
+ def _zoned(observed: str, values: list[Any]) -> str:
871
+ """``observed`` with its ``datetime`` kind replaced by what the values say about the zone.
872
+
873
+ Nothing is inferred and nothing is refused: the rendered text of a ``datetime.datetime`` is its
874
+ own ``isoformat``, and reading it back with ``datetime.fromisoformat`` -- the inverse of the
875
+ call that wrote it -- answers whether the value carried a zone and whether that zone was UTC.
876
+ A column whose datetimes do not all agree on UTC is reported as the kind it is, which is one no
877
+ Recipe column type covers. Text that will not read back is treated the same way: this says what
878
+ the column is, and ``timestamp_utc`` is a claim, not a default.
879
+ """
880
+
881
+ if _OBSERVED_DATETIME not in observed.split("|"):
882
+ return observed
883
+ for value in values:
884
+ if type(value).__name__ != _OBSERVED_DATETIME:
885
+ continue
886
+ offset = _utc_offset(str(value))
887
+ if offset is None:
888
+ return observed.replace(_OBSERVED_DATETIME, TIMESTAMP_NO_ZONE)
889
+ if offset != timedelta(0):
890
+ return observed.replace(_OBSERVED_DATETIME, TIMESTAMP_OTHER_ZONE)
891
+ return observed
892
+
893
+
894
+ def _utc_offset(text: str) -> timedelta | None:
895
+ """The offset from UTC the rendered timestamp carries, or ``None`` when it carries none."""
896
+
897
+ try:
898
+ return datetime.fromisoformat(text).utcoffset()
899
+ except ValueError:
900
+ return None
901
+
902
+
903
+ def _extremes(values: list[Any]) -> tuple[str | None, str | None]:
904
+ """The lowest and highest value, or nothing at all when they cannot be ordered.
905
+
906
+ A column holding more than one kind of value has no order, and saying so is more honest than
907
+ inventing one. It is never a refusal: unclean data is exactly what peek is for.
908
+ """
909
+
910
+ if not values:
911
+ return None, None
912
+ try:
913
+ ordered = sorted(values)
914
+ except TypeError:
915
+ return None, None
916
+ return str(ordered[0]), str(ordered[-1])
917
+
918
+
919
+ def _distinct(values: list[Any]) -> int:
920
+ """How many different values a column holds, counting only the ones that are there."""
921
+
922
+ try:
923
+ return len(set(values))
924
+ except TypeError:
925
+ return len({repr(value) for value in values})
926
+
927
+
928
+ def _bounded_sample_rows(sample_rows: int) -> int:
929
+ if type(sample_rows) is not int or not 0 <= sample_rows <= MAX_SAMPLE_ROWS:
930
+ raise AcquisitionSecurityError(
931
+ "PEEK_SAMPLE",
932
+ f"the number of rows to show must be a whole number from 0 to {MAX_SAMPLE_ROWS}",
933
+ )
934
+ return sample_rows
935
+
936
+
937
+ def _format_for(filename: str, declared: str | None) -> tuple[str, str]:
938
+ """Pick the format and the media type that agree with the Reader's own allowlists."""
939
+
940
+ if declared is not None:
941
+ if declared not in PEEK_FORMATS:
942
+ raise AcquisitionSecurityError("PEEK_FORMAT", _unreadable(declared))
943
+ return declared, _media_type_for(declared)
944
+ data_format = _SUFFIX_FORMATS.get(_suffix_of(filename))
945
+ if data_format is None:
946
+ raise AcquisitionSecurityError("PEEK_FORMAT", _unreadable(filename))
947
+ return data_format, _media_type_for(data_format)
948
+
949
+
950
+ def _media_type_for(data_format: str) -> str:
951
+ """The media type the Reader expects for this format, taken from the Reader's own table."""
952
+
953
+ return min(_FORMAT_MEDIA_TYPES[data_format])
954
+
955
+
956
+ def _suffix_of(filename: str) -> str:
957
+ dot = filename.rfind(".")
958
+ return filename[dot:].lower() if dot > 0 else ""
959
+
960
+
961
+ def _unreadable(subject: str) -> str:
962
+ return (
963
+ f"{subject} is not one of the formats this build can read ({_READABLE}); "
964
+ "the set of readable formats grows as Readers are certified"
965
+ )
966
+
967
+
968
+ def _json_safe(value: Any) -> Any:
969
+ """Keep a cell as it is when it is already a plain value; otherwise show it as written.
970
+
971
+ A rendered value arrives here as a ``str`` subclass carrying the name of the kind it came from.
972
+ That name is a fact about the column and it is already reported there; on the value itself it
973
+ would only be a type nobody asked about, so the text is handed on as plain text.
974
+ """
975
+
976
+ if isinstance(value, str):
977
+ return str(value)
978
+ if value is None or isinstance(value, bool | int):
979
+ return value
980
+ if isinstance(value, float):
981
+ return value if math.isfinite(value) else str(value)
982
+ return str(value)
983
+
984
+
985
+ __all__ = [
986
+ "ALL_MISSING",
987
+ "DEFAULT_SAMPLE_ROWS",
988
+ "MAX_SAMPLE_ROWS",
989
+ "MIXED_PREFIX",
990
+ "NOT_A_COLUMN_TYPE_PREFIX",
991
+ "NO_ROWS",
992
+ "OBSERVED_LOGICAL_TYPES",
993
+ "PEEK_FORMATS",
994
+ "PeekColumn",
995
+ "PeekResult",
996
+ "peek_bytes",
997
+ "peek_path",
998
+ "peek_url",
999
+ "plain_column_type",
1000
+ ]