mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,1070 @@
1
+ """The read, compare and view commands, answered from Studio instead of from a folder.
2
+
3
+ WHAT THIS MODULE IS FOR. ADR 0021 requires every lane to be reachable through hosted dispatch
4
+ with the same vocabulary and the same receipts, so that *where* a command runs stops being
5
+ something a person has to know. :mod:`mostlyright.data_harness.thin.commands` does that for the
6
+ five commands that submit, follow and fetch a run. These ten are the other half of a working day:
7
+ looking at what you have, checking it, comparing two of them, and reading the notebook.
8
+
9
+ THE ONE TRANSLATION, STATED ONCE. Every command here takes a run identifier where its local twin
10
+ takes a folder. That is the whole difference in the vocabulary, and it is not a new flag or a new
11
+ name -- a hosted build has no folder, and the run identifier is what the submission printed and
12
+ what the dashboard address contains. Everything else is held identical on purpose: the same
13
+ status words, so ``mr-data show --json`` answers with ``build_shown`` in both profiles; the same
14
+ payload keys wherever the fact exists on both sides; and the same numbered-key convention, where
15
+ a fact about the data goes in a value and never in a key.
16
+
17
+ WHERE A FLAG CANNOT MEAN WHAT IT MEANS LOCALLY. It is accepted and reported, never refused. A
18
+ flag that errors is a script that breaks on a machine that installed a different extra, which is
19
+ exactly the variance this phase exists to remove. Each such flag lands in ``flags_without_effect``
20
+ with the sentence saying why -- ``mr-data list --depth 3 --hosted`` runs, lists the workspace's
21
+ runs, and says that folder depth described nothing here.
22
+
23
+ WHAT THESE COMMANDS NEVER DO. They never mutate. Every route below is a GET, so an interrupted
24
+ ``show`` and an interrupted ``verify`` leave a workspace exactly as they found it, and none of
25
+ them can be the reason a run changed state.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import argparse
31
+ from collections.abc import Callable, Mapping, Sequence
32
+ from pathlib import Path
33
+ from typing import Any
34
+ from uuid import UUID
35
+
36
+ from mostlyright.data_harness.thin import THIN_SCHEMA_PREFIX
37
+ from mostlyright.data_harness.thin.commands import (
38
+ artifact_records,
39
+ media_type_essence,
40
+ optional_receipt,
41
+ run_receipt,
42
+ suffix_for,
43
+ )
44
+ from mostlyright.data_harness.thin.download import download_signed_artifact
45
+ from mostlyright.data_harness.thin.runs import StudioRunClient
46
+ from mostlyright.data_harness.thin.session import (
47
+ StudioSession,
48
+ open_studio_session,
49
+ run_dashboard_url,
50
+ )
51
+ from mostlyright.data_harness.thin.stream import TERMINAL_RUN_STATUSES
52
+ from mostlyright.data_harness.thin.transport import ThinLaneError
53
+ from mostlyright.data_harness.ux.credentials import resolve_cloud_credentials
54
+
55
+ SHOW_SCHEMA = f"{THIN_SCHEMA_PREFIX}-build-shown.v1"
56
+ LIST_SCHEMA = f"{THIN_SCHEMA_PREFIX}-builds-listed.v1"
57
+ INSPECT_SCHEMA = f"{THIN_SCHEMA_PREFIX}-build-inspected.v1"
58
+ VERIFY_SCHEMA = f"{THIN_SCHEMA_PREFIX}-build-verified.v1"
59
+ DIFF_SCHEMA = f"{THIN_SCHEMA_PREFIX}-build-comparison.v1"
60
+ FLEET_SCHEMA = f"{THIN_SCHEMA_PREFIX}-fleet.v1"
61
+ NOTEBOOK_SCHEMA = f"{THIN_SCHEMA_PREFIX}-notebook.v1"
62
+ PEEK_SCHEMA = f"{THIN_SCHEMA_PREFIX}-peek.v1"
63
+ DEPLOYMENT_STATUS_SCHEMA = f"{THIN_SCHEMA_PREFIX}-deployment-run-status.v1"
64
+
65
+ #: Where a hosted notebook lands when the caller did not choose. A folder rather than a file,
66
+ #: because one run renders two documents and writing either over the other would be a lie about
67
+ #: which one is there.
68
+ DEFAULT_NOTEBOOK_OUTPUT = "hosted-notebook"
69
+
70
+ # --------------------------------------------------------------------------------------------
71
+ # Words this lane must not invent
72
+ # --------------------------------------------------------------------------------------------
73
+ #
74
+ # Both blocks below restate strings the local lane already owns, because the modules that hold
75
+ # them (`ux.readers`, `fleet`) cannot be imported on a thin install -- they reach the engine. A
76
+ # restated string is a string that can drift, so `tests/test_thin_parity.py` asserts each one is
77
+ # byte-identical to its original, which runs on a full install and can import both.
78
+
79
+ #: ``ux.readers.LISTED_ONLY`` / ``VERIFIED``. A hosted listing is Studio's own reading of its own
80
+ #: runs, so it is never "listed, not verified" in the local sense -- but it is also not a replay,
81
+ #: and saying "verified" about bytes nobody re-hashed here would be the claim `--verify` exists to
82
+ #: avoid making. The listing says what it is: Studio's verdict, read.
83
+ LISTED_ONLY = "listed, not verified"
84
+ VERIFIED = "verified"
85
+ NOT_ALL_VERIFIED = "not all verified"
86
+ NOTHING_TO_VERIFY = "nothing to verify"
87
+
88
+ #: ``fleet._WAITING``, ``fleet._FIXES_REASON``, and the two failure sentences beside them.
89
+ FLEET_WAITING = "waiting for a person to look at it"
90
+ FLEET_FIXES = "reviewed: the review asked for fixes"
91
+ FLEET_FAILED = "stopped: the build failed"
92
+
93
+ #: Studio's Run status -> the plain state word ``fleet`` prints for a local run. Every one of the
94
+ #: thirteen is named: a state with no word here would print a state word this product does not
95
+ #: use, and a fleet view that shows one row in Studio's vocabulary and the rest in ours is worse
96
+ #: than one that shows all thirteen in either.
97
+ FLEET_STATE: dict[str, str] = {
98
+ "queued": "not started",
99
+ "producer_retry_pending": "not started",
100
+ "running": "building",
101
+ "awaiting_candidate_selection": "built",
102
+ "verifying": "built",
103
+ "verifier_retry_pending": "built",
104
+ "releasable": "reviewed",
105
+ "released": "reviewed",
106
+ "repair_required": "stopped",
107
+ "human_direction_required": "stopped",
108
+ "failed": "stopped",
109
+ "cancelled": "stopped",
110
+ "abandoned": "stopped",
111
+ }
112
+
113
+ #: The states that are waiting on a person, and the sentence each one gets. A state absent here
114
+ #: needs nobody: it is either moving on its own or finished.
115
+ FLEET_NEEDS_A_PERSON: dict[str, str] = {
116
+ "releasable": FLEET_WAITING,
117
+ "repair_required": FLEET_FIXES,
118
+ "human_direction_required": FLEET_WAITING,
119
+ "failed": FLEET_FAILED,
120
+ }
121
+
122
+
123
+ # --------------------------------------------------------------------------------------------
124
+ # Shared plumbing
125
+ # --------------------------------------------------------------------------------------------
126
+
127
+
128
+ def _session(args: argparse.Namespace) -> StudioSession:
129
+ return open_studio_session(resolve_cloud_credentials())
130
+
131
+
132
+ def _client(args: argparse.Namespace) -> StudioRunClient:
133
+ return StudioRunClient(_session(args))
134
+
135
+
136
+ def _run_id(value: Any) -> str:
137
+ """Refuse anything that is not a run identifier before it reaches a URL path.
138
+
139
+ Checked before a credential is resolved and a token minted, which is the ordering
140
+ :mod:`thin.commands` already establishes: a typo should cost a sentence, not a round trip.
141
+ The sentence names the substitution, because a folder path is exactly what somebody moving
142
+ from the local lane will type here first.
143
+ """
144
+
145
+ try:
146
+ return str(UUID(str(value)))
147
+ except (ValueError, AttributeError) as error:
148
+ raise ThinLaneError(
149
+ "THIN_REQUEST_INVALID",
150
+ f"{value!r} is not a run identifier; the hosted lane names a run by the identifier "
151
+ f"the submission printed, not by a folder",
152
+ ) from error
153
+
154
+
155
+ def _no_effect(
156
+ args: argparse.Namespace, declared: Sequence[tuple[str, str, str]]
157
+ ) -> dict[str, str]:
158
+ """The flags this invocation passed that the hosted lane cannot honour, and why.
159
+
160
+ ``declared`` is one entry per flag: the ``argparse`` destination, the flag as a person typed
161
+ it, and the sentence. Only flags actually given are reported -- a caller who did not pass
162
+ ``--depth`` is not told that ``--depth`` would have done nothing.
163
+ """
164
+
165
+ reported: dict[str, str] = {}
166
+ for destination, flag, sentence in declared:
167
+ value = getattr(args, destination, None)
168
+ if _was_given(value):
169
+ reported[flag] = sentence
170
+ return reported
171
+
172
+
173
+ def _was_given(value: Any) -> bool:
174
+ """Whether this argument was actually passed, told apart from its default.
175
+
176
+ ⚠ Written out rather than as ``value not in (None, False, ())``, which is the obvious version
177
+ and is wrong: ``0 == False`` in Python, so ``--depth 0`` and ``--port 0`` would be read as
178
+ absent and silently not reported. An argument somebody typed must be answered for whatever
179
+ number they typed.
180
+ """
181
+
182
+ if value is None or value is False:
183
+ return False
184
+ return not (isinstance(value, (str, tuple, list, dict)) and len(value) == 0)
185
+
186
+
187
+ def _released_version(client: StudioRunClient, run: Mapping[str, Any]) -> tuple[str, str] | None:
188
+ """The table version this run released, as ``(table_id, table_version_id)``.
189
+
190
+ A Run carries the table it builds but not the version it produced -- the version is created by
191
+ the release, after the run's own record is written -- so it is found by asking the table for
192
+ its versions and taking the one this run made. ``None`` whenever the run has not released
193
+ anything, which is the ordinary state of most runs and never an error.
194
+ """
195
+
196
+ table_id = run.get("table_id")
197
+ run_id = run.get("run_id")
198
+ if not isinstance(table_id, str) or not isinstance(run_id, str):
199
+ return None
200
+ versions = optional_receipt("table_versions", lambda: client.table_versions(table_id), {})
201
+ if not isinstance(versions, list):
202
+ return None
203
+ for version in versions:
204
+ if isinstance(version, Mapping) and version.get("run_id") == run_id:
205
+ identifier = version.get("table_version_id")
206
+ if isinstance(identifier, str):
207
+ return table_id, identifier
208
+ return None
209
+
210
+
211
+ def _evidence_receipt(evidence: Mapping[str, Any]) -> dict[str, Any]:
212
+ """The evidence coordinates a person reads, exactly as Studio reported them."""
213
+
214
+ return {
215
+ key: evidence[key]
216
+ for key in (
217
+ "candidate_digest",
218
+ "execution_scope_digest",
219
+ "release_policy_digest",
220
+ "validation_policy_digest",
221
+ "verification_report_id",
222
+ "admission",
223
+ )
224
+ if key in evidence
225
+ }
226
+
227
+
228
+ def _artifact_receipt(artifact: Mapping[str, Any]) -> dict[str, Any]:
229
+ return {
230
+ key: artifact[key]
231
+ for key in (
232
+ "artifact_id",
233
+ "kind",
234
+ "purpose",
235
+ "media_type",
236
+ "size_bytes",
237
+ "content_digest",
238
+ "classification",
239
+ )
240
+ if key in artifact
241
+ }
242
+
243
+
244
+ def _numbered(prefix: str, items: Sequence[Any]) -> dict[str, Any]:
245
+ """``{"build 1": ..., "build 2": ...}``, zero-padded so twelve sorts after nine.
246
+
247
+ The convention ``ux.readers`` and ``ux.diffing`` already use, restated here rather than
248
+ imported for the same reason the words above are: those modules reach the engine.
249
+ """
250
+
251
+ width = len(str(max(len(items), 1)))
252
+ return {f"{prefix} {index:0{width}d}": item for index, item in enumerate(items, start=1)}
253
+
254
+
255
+ def _hosted_head(session: StudioSession, run_id: str) -> dict[str, Any]:
256
+ """The three facts every answer in this module opens with."""
257
+
258
+ return {
259
+ "lane": "hosted",
260
+ "run_id": run_id,
261
+ "dashboard_url": run_dashboard_url(session.cloud_url, run_id),
262
+ }
263
+
264
+
265
+ def _require_evidence(client: StudioRunClient, run_id: str, run: Mapping[str, Any]) -> Any:
266
+ """Read one run's sealed evidence, saying what state it is in when there is none yet.
267
+
268
+ A 404 here is the ordinary state of every run that has not produced a result, and reporting it
269
+ as "not found" would tell somebody their run does not exist when what happened is that it is
270
+ still going.
271
+ """
272
+
273
+ try:
274
+ return client.candidate_evidence(run_id)
275
+ except ThinLaneError as refusal:
276
+ if refusal.code != "THIN_NOT_FOUND":
277
+ raise
278
+ state = run.get("status")
279
+ raise ThinLaneError(
280
+ "THIN_NO_SEALED_RESULT",
281
+ f"this run has sealed nothing to read yet; Studio has it as {state!r}"
282
+ + ("" if state in TERMINAL_RUN_STATUSES else ", and it is still going"),
283
+ ) from refusal
284
+
285
+
286
+ # --------------------------------------------------------------------------------------------
287
+ # Reading one run back
288
+ # --------------------------------------------------------------------------------------------
289
+
290
+
291
+ def show(args: argparse.Namespace, *, client: StudioRunClient | None = None) -> dict[str, Any]:
292
+ """``mr-data show``: what one hosted run holds.
293
+
294
+ The local command answers from a Receipt on disk. This answers from the three records Studio
295
+ holds about the same build -- the run, its sealed evidence, and, once it released one, the
296
+ table version and the bounded preview beside it -- and puts them under the keys the local
297
+ payload already uses, so a script that reads ``candidate_digest`` reads it in both profiles.
298
+ """
299
+
300
+ run_id = _run_id(args.run_dir)
301
+ selected = client or _client(args)
302
+ run = selected.get_run(run_id)
303
+ evidence = _require_evidence(selected, run_id, run)
304
+ unavailable: dict[str, str] = {}
305
+ payload: dict[str, Any] = {
306
+ "schema_version": SHOW_SCHEMA,
307
+ "status": "build_shown",
308
+ **_hosted_head(selected.session, run_id),
309
+ "run": run_receipt(run),
310
+ **_evidence_receipt(evidence),
311
+ }
312
+ released = _released_version(selected, run)
313
+ if released is not None:
314
+ table_id, version_id = released
315
+ payload["table_id"] = table_id
316
+ payload["table_version_id"] = version_id
317
+ preview = optional_receipt(
318
+ "preview",
319
+ lambda: selected.table_version_preview(table_id, version_id),
320
+ unavailable,
321
+ )
322
+ if isinstance(preview, Mapping):
323
+ payload["row_count"] = preview.get("total_row_count")
324
+ payload["columns"] = _column_names(preview)
325
+ payload["schema"] = _schema_payload(preview)
326
+ if getattr(args, "members", False):
327
+ payload["members"] = _numbered(
328
+ "member", [_artifact_receipt(item) for item in artifact_records(evidence)]
329
+ )
330
+ payload["card"] = _card(run, evidence, payload)
331
+ if getattr(args, "card", False):
332
+ return {
333
+ "schema_version": SHOW_SCHEMA,
334
+ "status": "build_shown",
335
+ **_hosted_head(selected.session, run_id),
336
+ "card": payload["card"],
337
+ }
338
+ payload["receipts_unavailable"] = unavailable
339
+ return payload
340
+
341
+
342
+ def _card(run: Mapping[str, Any], evidence: Mapping[str, Any], payload: Mapping[str, Any]) -> str:
343
+ """The plain-language summary, built only from facts already in the payload.
344
+
345
+ Nothing here is computed, rounded or inferred. Every clause is present exactly when the fact
346
+ behind it is, so a card that says less is a build Studio said less about.
347
+ """
348
+
349
+ parts = [f"Studio has this run as {run.get('status', 'a state it did not name')}."]
350
+ rows = payload.get("row_count")
351
+ columns = payload.get("columns")
352
+ if isinstance(rows, int) and isinstance(columns, list):
353
+ parts.append(f"The released version holds {rows} rows across {len(columns)} columns.")
354
+ digest = evidence.get("candidate_digest")
355
+ if isinstance(digest, str):
356
+ parts.append(f"Its fingerprint is {digest}.")
357
+ return " ".join(parts)
358
+
359
+
360
+ def _column_names(preview: Mapping[str, Any]) -> list[str]:
361
+ columns = preview.get("columns")
362
+ if not isinstance(columns, list):
363
+ return []
364
+ names = []
365
+ for column in columns:
366
+ if isinstance(column, Mapping) and isinstance(column.get("name"), str):
367
+ names.append(column["name"])
368
+ elif isinstance(column, str):
369
+ names.append(column)
370
+ return names
371
+
372
+
373
+ def _schema_payload(preview: Mapping[str, Any]) -> dict[str, Any]:
374
+ """One numbered entry per column, carrying its real name as a value.
375
+
376
+ The same rule ``ux.readers._schema_payload`` follows: a column called ``max_temp_c`` used as a
377
+ key would be shown to a person as ``Max temp c``, a name the data does not have.
378
+ """
379
+
380
+ columns = preview.get("columns")
381
+ if not isinstance(columns, list):
382
+ return {}
383
+ entries = []
384
+ for column in columns:
385
+ if isinstance(column, Mapping):
386
+ entries.append({"name": column.get("name"), "type": column.get("type")})
387
+ else:
388
+ entries.append({"name": column, "type": None})
389
+ return _numbered("column", entries)
390
+
391
+
392
+ def inspect(args: argparse.Namespace, *, client: StudioRunClient | None = None) -> dict[str, Any]:
393
+ """``mr-data inspect``: the sealed evidence Studio holds for one hosted run.
394
+
395
+ Where ``show`` answers "what is in it", this answers "what was proved about it": the sealed
396
+ evidence, the execution history, the redacted model-call evidence, and every artifact the run
397
+ published. The evidence itself must be there -- there is nothing to inspect without it -- and
398
+ the two records beside it are optional in the way ``status --receipts`` already makes them:
399
+ one that could not be read says WHY under ``receipts_unavailable`` rather than being absent,
400
+ because "there is nothing" and "you may not see it" are different sentences.
401
+ """
402
+
403
+ run_id = _run_id(args.workspace)
404
+ selected = client or _client(args)
405
+ run = selected.get_run(run_id)
406
+ evidence = _require_evidence(selected, run_id, run)
407
+ unavailable: dict[str, str] = {}
408
+ summary = optional_receipt(
409
+ "execution_summary", lambda: selected.execution_summary(run_id), unavailable
410
+ )
411
+ agent = optional_receipt(
412
+ "agent_evidence", lambda: selected.run_agent_evidence(run_id), unavailable
413
+ )
414
+ return {
415
+ "schema_version": INSPECT_SCHEMA,
416
+ "status": "candidate_inspected",
417
+ **_hosted_head(selected.session, run_id),
418
+ "run": run_receipt(run),
419
+ **_evidence_receipt(evidence),
420
+ "execution_summary": summary,
421
+ "agent_evidence": _numbered("call", list(agent)) if isinstance(agent, list) else None,
422
+ "members": _numbered(
423
+ "member", [_artifact_receipt(item) for item in artifact_records(evidence)]
424
+ ),
425
+ "receipts_unavailable": unavailable,
426
+ }
427
+
428
+
429
+ def verify(args: argparse.Namespace, *, client: StudioRunClient | None = None) -> dict[str, Any]:
430
+ """``mr-data verify``: re-check one hosted run against the evidence Studio sealed.
431
+
432
+ ⚠ WHAT "VERIFY" HONESTLY MEANS HERE. The local command replays a build's bytes on this
433
+ computer. A hosted build's bytes are not here, and downloading gigabytes to re-hash them is
434
+ not what somebody typing ``verify`` is asking for. What this reads instead is the independent
435
+ check Studio already ran -- a different worker, a different attempt, the same recipe -- and
436
+ the findings it opened. That is a stronger check than a local replay, not a weaker one, and
437
+ the payload says which of the two it is rather than letting the word carry the difference.
438
+
439
+ The verdict comes from the findings, never from the run's state word: a released run with an
440
+ open blocking finding is not a verified build, and the exit code follows the verdict.
441
+ """
442
+
443
+ run_id = _run_id(args.run_dir)
444
+ selected = client or _client(args)
445
+ run = selected.get_run(run_id)
446
+ evidence = _require_evidence(selected, run_id, run)
447
+ findings = selected.run_findings(run_id)
448
+ blocking = [
449
+ finding
450
+ for finding in findings
451
+ if finding.get("blocks_release") is True and finding.get("status") != "closed"
452
+ ]
453
+ return {
454
+ "schema_version": VERIFY_SCHEMA,
455
+ "status": "fixes_required" if blocking else "candidate_verified",
456
+ **_hosted_head(selected.session, run_id),
457
+ "run": run_receipt(run),
458
+ **_evidence_receipt(evidence),
459
+ "checked_by": "the independent hosted check, read back; the bytes were not replayed here",
460
+ "finding count": len(findings),
461
+ "blocking count": len(blocking),
462
+ "findings": _numbered("finding", [_finding_receipt(item) for item in findings]),
463
+ }
464
+
465
+
466
+ def verify_exit_code(payload: Mapping[str, Any]) -> int:
467
+ """``2`` when a finding blocks release, so a script can gate on the verdict.
468
+
469
+ The same convention the local review verdict follows: an answer is an answer whatever it says,
470
+ but "fixes are required" is not a passing check and must not exit as one.
471
+ """
472
+
473
+ return 2 if payload.get("status") == "fixes_required" else 0
474
+
475
+
476
+ def _finding_receipt(finding: Mapping[str, Any]) -> dict[str, Any]:
477
+ return {
478
+ key: finding[key]
479
+ for key in (
480
+ "finding_id",
481
+ "check_id",
482
+ "category",
483
+ "code",
484
+ "severity",
485
+ "summary",
486
+ "status",
487
+ "blocks_release",
488
+ "opened_at",
489
+ "closed_at",
490
+ )
491
+ if key in finding
492
+ }
493
+
494
+
495
+ # --------------------------------------------------------------------------------------------
496
+ # Reading many runs back
497
+ # --------------------------------------------------------------------------------------------
498
+
499
+
500
+ def listing(args: argparse.Namespace, *, client: StudioRunClient | None = None) -> dict[str, Any]:
501
+ """``mr-data list``: the hosted runs in this workspace.
502
+
503
+ The local command walks a folder; this asks Studio, which scopes the answer to the workspace
504
+ the credential names. ``root`` and ``--depth`` describe a folder and are reported as having
505
+ described nothing; ``--verify`` is reported too, and the reason is worth reading -- there is
506
+ nothing here to replay, because the independent check already ran on the backend and its
507
+ verdict is in each row.
508
+ """
509
+
510
+ selected = client or _client(args)
511
+ runs = selected.list_runs(
512
+ status=getattr(args, "run_status", None), limit=getattr(args, "limit", None)
513
+ )
514
+ entries = [_listing_entry(run) for run in runs]
515
+ return {
516
+ "schema_version": LIST_SCHEMA,
517
+ "status": "builds_listed",
518
+ "lane": "hosted",
519
+ "workspace_id": str(selected.session.workspace_id),
520
+ "verification": _listing_verification(entries),
521
+ "note": _listing_note(entries),
522
+ "builds": _numbered("build", entries),
523
+ "flags_without_effect": _no_effect(
524
+ args,
525
+ (
526
+ ("root", "root", "a hosted listing is the workspace's runs, not a folder's"),
527
+ ("depth", "--depth", "there are no folders here to look down through"),
528
+ (
529
+ "verify",
530
+ "--verify",
531
+ "there is nothing here to replay; the independent hosted check already ran, "
532
+ "and each row carries its verdict",
533
+ ),
534
+ ),
535
+ ),
536
+ }
537
+
538
+
539
+ def _listing_entry(run: Mapping[str, Any]) -> dict[str, Any]:
540
+ return {
541
+ "run_id": run.get("run_id"),
542
+ "run_status": run.get("status"),
543
+ "kind": run.get("kind"),
544
+ "recipe_digest": run.get("recipe_digest"),
545
+ "created_at": run.get("created_at"),
546
+ "completed_at": run.get("completed_at"),
547
+ "verification": VERIFIED if run.get("status") == "released" else LISTED_ONLY,
548
+ }
549
+
550
+
551
+ def _listing_verification(entries: Sequence[Mapping[str, Any]]) -> str:
552
+ """What the listing as a whole may claim, taken from the rows and never from a flag."""
553
+
554
+ if not entries:
555
+ return NOTHING_TO_VERIFY
556
+ if all(entry["verification"] == VERIFIED for entry in entries):
557
+ return VERIFIED
558
+ return NOT_ALL_VERIFIED
559
+
560
+
561
+ def _listing_note(entries: Sequence[Mapping[str, Any]]) -> str:
562
+ if not entries:
563
+ return "No runs in this workspace yet. Run mr-data build --hosted to queue one."
564
+ released = sum(1 for entry in entries if entry["verification"] == VERIFIED)
565
+ return (
566
+ f"{len(entries)} runs, of which {released} released a version the independent hosted "
567
+ "check passed. Run mr-data show on one of their identifiers for the full answer."
568
+ )
569
+
570
+
571
+ def fleet(args: argparse.Namespace, *, client: StudioRunClient | None = None) -> dict[str, Any]:
572
+ """``mr-data fleet``: which hosted runs are waiting for a person.
573
+
574
+ Same rows, same four keys, same words as the local scan -- ``name``, ``state``,
575
+ ``needs_person``, ``reasons`` -- with the run identifier as the name, because that is what a
576
+ hosted run is called. ``--html`` is reported as having no effect and the dashboard address is
577
+ given instead: a page rendered here would be a second, staler answer beside the live one.
578
+ """
579
+
580
+ selected = client or _client(args)
581
+ runs = selected.list_runs(limit=getattr(args, "limit", None))
582
+ rows = [_fleet_row(run) for run in runs]
583
+ return {
584
+ "schema_version": FLEET_SCHEMA,
585
+ "status": "fleet_scanned",
586
+ "lane": "hosted",
587
+ "workspace_id": str(selected.session.workspace_id),
588
+ "waiting": sum(1 for row in rows if row["needs_person"]),
589
+ "runs": rows,
590
+ "flags_without_effect": _no_effect(
591
+ args,
592
+ (
593
+ ("root", "root", "a hosted fleet is the workspace's runs, not a folder's"),
594
+ (
595
+ "html",
596
+ "--html",
597
+ "the live page for these runs is the dashboard; a page rendered here would be "
598
+ "a second answer that is already out of date",
599
+ ),
600
+ ),
601
+ ),
602
+ }
603
+
604
+
605
+ def _fleet_row(run: Mapping[str, Any]) -> dict[str, Any]:
606
+ state = run.get("status")
607
+ reason = FLEET_NEEDS_A_PERSON.get(str(state))
608
+ return {
609
+ "name": run.get("run_id"),
610
+ "state": FLEET_STATE.get(str(state), "state not recorded"),
611
+ "needs_person": reason is not None,
612
+ "reasons": [reason] if reason is not None else [],
613
+ }
614
+
615
+
616
+ def diff(args: argparse.Namespace, *, client: StudioRunClient | None = None) -> dict[str, Any]:
617
+ """``mr-data diff``: compare two hosted runs and say what changed.
618
+
619
+ Finding a difference is an answer, not a failure, so this exits 0 either way and the verdict
620
+ is the status word -- ``builds_identical`` or ``builds_differ`` -- exactly as the local
621
+ command's help promises. Identity is the sealed fingerprint and nothing else: two runs of the
622
+ same recipe at different times are the same data when their fingerprints agree.
623
+ """
624
+
625
+ left_id = _run_id(args.left)
626
+ right_id = _run_id(args.right)
627
+ selected = client or _client(args)
628
+ left = _side(selected, left_id)
629
+ right = _side(selected, right_id)
630
+ changes = _changes(left, right)
631
+ columns_only = bool(getattr(args, "columns_only", False))
632
+ shown = (
633
+ [change for change in changes if change["is column change"]] if columns_only else changes
634
+ )
635
+ identical = left["candidate_digest"] == right["candidate_digest"]
636
+ payload: dict[str, Any] = {
637
+ "schema_version": DIFF_SCHEMA,
638
+ "status": "builds_identical" if identical else "builds_differ",
639
+ "lane": "hosted",
640
+ "summary": _comparison_summary(identical, changes, shown),
641
+ # The full count, always. A narrowed view narrows what is listed, never what is true.
642
+ "change count": len(changes),
643
+ "changes": _numbered("change", shown),
644
+ "first build": left,
645
+ "second build": right,
646
+ }
647
+ if len(shown) != len(changes):
648
+ payload["columns only"] = True
649
+ return payload
650
+
651
+
652
+ def _side(client: StudioRunClient, run_id: str) -> dict[str, Any]:
653
+ """One side of the comparison, refused unless it carries the fingerprint identity rests on.
654
+
655
+ Absent everywhere else in this module -- ``show`` reports a run Studio said less about -- and
656
+ mandatory here, because a comparison whose identity field is missing on both sides would
657
+ answer "the same" about two builds nobody compared.
658
+ """
659
+
660
+ run = client.get_run(run_id)
661
+ evidence = _require_evidence(client, run_id, run)
662
+ if not isinstance(evidence.get("candidate_digest"), str):
663
+ raise ThinLaneError(
664
+ "THIN_NO_FINGERPRINT",
665
+ f"Studio's evidence for {run_id} carries no fingerprint, so this run cannot be "
666
+ "compared with another",
667
+ )
668
+ side: dict[str, Any] = {
669
+ "run_id": run_id,
670
+ "run_status": run.get("status"),
671
+ "recipe_digest": run.get("recipe_digest"),
672
+ "recipe_version": run.get("recipe_version"),
673
+ "candidate_digest": evidence.get("candidate_digest"),
674
+ "row_count": None,
675
+ "columns": [],
676
+ }
677
+ released = _released_version(client, run)
678
+ if released is not None:
679
+ table_id, version_id = released
680
+ preview = optional_receipt(
681
+ "preview", lambda: client.table_version_preview(table_id, version_id), {}
682
+ )
683
+ if isinstance(preview, Mapping):
684
+ side["row_count"] = preview.get("total_row_count")
685
+ side["columns"] = _column_names(preview)
686
+ return side
687
+
688
+
689
+ def _changes(left: Mapping[str, Any], right: Mapping[str, Any]) -> list[dict[str, Any]]:
690
+ """Every difference between two sides, columns first and then the counts.
691
+
692
+ Column order is the second build's, so a listing reads in the order the newer data has.
693
+ """
694
+
695
+ changes: list[dict[str, Any]] = []
696
+ before = list(left["columns"])
697
+ after = list(right["columns"])
698
+ for name in after:
699
+ if name not in before:
700
+ changes.append({"what": "column added", "name": name, "is column change": True})
701
+ for name in before:
702
+ if name not in after:
703
+ changes.append({"what": "column removed", "name": name, "is column change": True})
704
+ if left["row_count"] != right["row_count"]:
705
+ changes.append(
706
+ {
707
+ "what": "row count moved",
708
+ "before": left["row_count"],
709
+ "after": right["row_count"],
710
+ "is column change": False,
711
+ }
712
+ )
713
+ if left["candidate_digest"] != right["candidate_digest"]:
714
+ changes.append(
715
+ {
716
+ "what": "the data is different",
717
+ "before": left["candidate_digest"],
718
+ "after": right["candidate_digest"],
719
+ "is column change": False,
720
+ }
721
+ )
722
+ return changes
723
+
724
+
725
+ def _comparison_summary(
726
+ identical: bool, changes: Sequence[Mapping[str, Any]], shown: Sequence[Mapping[str, Any]]
727
+ ) -> str:
728
+ if identical:
729
+ return "Those two runs hold the same data, in the same shape."
730
+ counted = len(changes)
731
+ sentence = f"Those two runs are not the same: {counted} " + (
732
+ "difference." if counted == 1 else "differences."
733
+ )
734
+ if len(shown) != counted:
735
+ sentence += f" Only the {len(shown)} about columns are listed."
736
+ return sentence
737
+
738
+
739
+ # --------------------------------------------------------------------------------------------
740
+ # The notebook, rendered where the run ran
741
+ # --------------------------------------------------------------------------------------------
742
+
743
+
744
+ def notebook(args: argparse.Namespace, *, client: StudioRunClient | None = None) -> dict[str, Any]:
745
+ """``mr-data notebook``: bring back the notebook Studio rendered for one hosted run.
746
+
747
+ ADR 0021 moves this rendering to the backend, and that is the point rather than a compromise:
748
+ Studio renders the notebook from the run's own append-only event log and from nothing else, so
749
+ the notebook, the live stream and the sealed evidence read the same record and cannot disagree.
750
+ A notebook rendered here from downloaded bytes would be a second answer with no such guarantee.
751
+
752
+ Both documents are fetched by default -- the ``.ipynb`` somebody opens in Jupyter and the
753
+ standalone HTML somebody opens in a browser -- and both are digest-verified before they are
754
+ written, through the same ``O_EXCL`` write ``deploy-dataset`` uses.
755
+ """
756
+
757
+ return _fetch_notebook(args, client, wanted=("ipynb", "html"), open_html=False)
758
+
759
+
760
+ def view(args: argparse.Namespace, *, client: StudioRunClient | None = None) -> dict[str, Any]:
761
+ """``mr-data view``: open the notebook Studio rendered for one hosted run.
762
+
763
+ The local command starts a viewer server on this computer and points a browser at it. There is
764
+ nothing here to serve, so this brings back the standalone HTML rendering -- whose charts are
765
+ inline, with no external asset, precisely so it stands alone -- and opens it when ``--open``
766
+ is given, exactly as the local command's flag means. Every argument about the local server is
767
+ reported as having no effect.
768
+ """
769
+
770
+ payload = _fetch_notebook(
771
+ args, client, wanted=("html",), open_html=bool(getattr(args, "open", False))
772
+ )
773
+ payload["flags_without_effect"] = _no_effect(
774
+ args,
775
+ (
776
+ ("workspace", "--workspace", "a hosted run has no workbench folder here"),
777
+ ("run_dir", "--run-dir", "a hosted run has no build folder here"),
778
+ ("research_dir", "--research-dir", "the research notebook is rendered on the backend"),
779
+ # ⚠ NOT "the document is a file". The path-kind gate reads every sentence this
780
+ # package can print and refuses one that names a kind nothing looked at -- and
781
+ # nothing here looked at anything on disk. What is true and says as much is that the
782
+ # document was brought back rather than served.
783
+ (
784
+ "host",
785
+ "--host",
786
+ "nothing is served from this computer; the document is brought here",
787
+ ),
788
+ (
789
+ "port",
790
+ "--port",
791
+ "nothing is served from this computer; the document is brought here",
792
+ ),
793
+ ),
794
+ )
795
+ return payload
796
+
797
+
798
+ def _fetch_notebook(
799
+ args: argparse.Namespace,
800
+ client: StudioRunClient | None,
801
+ *,
802
+ wanted: Sequence[str],
803
+ open_html: bool,
804
+ ) -> dict[str, Any]:
805
+ run_id = _run_id(getattr(args, "run", None) or getattr(args, "run_dir", None))
806
+ selected = client or _client(args)
807
+ try:
808
+ rendered = selected.run_notebook(run_id)
809
+ except ThinLaneError as refusal:
810
+ if refusal.code != "THIN_NOT_FOUND":
811
+ raise
812
+ raise ThinLaneError(
813
+ "THIN_NOTEBOOK_NOT_RENDERED",
814
+ "this run has no notebook yet; Studio renders one when the run finishes, and the "
815
+ "live view until then is mr-data watch",
816
+ ) from refusal
817
+ documents = rendered.get("documents")
818
+ if not isinstance(documents, Mapping):
819
+ raise ThinLaneError("THIN_RESPONSE_INVALID", "Studio described no notebook documents")
820
+ output = Path(getattr(args, "output", None) or DEFAULT_NOTEBOOK_OUTPUT).expanduser()
821
+ written: dict[str, Any] = {}
822
+ for name in wanted:
823
+ document = documents.get(name)
824
+ if not isinstance(document, Mapping):
825
+ raise ThinLaneError(
826
+ "THIN_RESPONSE_INVALID", f"Studio described no {name} notebook document"
827
+ )
828
+ destination = output / f"{run_id}{_notebooksuffix_for(name, document)}"
829
+ artifact = download_signed_artifact(
830
+ _notebook_session(document), destination, transport=selected.transport
831
+ )
832
+ written[name] = {
833
+ "path": str(artifact.path),
834
+ "media_type": artifact.media_type,
835
+ "size_bytes": artifact.size_bytes,
836
+ "content_digest": artifact.content_digest,
837
+ }
838
+ payload: dict[str, Any] = {
839
+ "schema_version": NOTEBOOK_SCHEMA,
840
+ "status": "notebook_ready",
841
+ **_hosted_head(selected.session, run_id),
842
+ "rendered_at": rendered.get("rendered_at"),
843
+ "run_status": rendered.get("run_status"),
844
+ "source_log_digest": rendered.get("source_log_digest"),
845
+ "source_event_count": rendered.get("source_event_count"),
846
+ "source_truncated": rendered.get("source_truncated"),
847
+ "output": str(output),
848
+ "documents": written,
849
+ }
850
+ if rendered.get("source_truncated") is True:
851
+ # Not a footnote. A notebook that covers part of a run is a notebook that can be read as
852
+ # covering all of it, and the one place somebody is guaranteed to look is the top.
853
+ payload["note"] = (
854
+ "This run's history was longer than the renderer reads, so the notebook covers only "
855
+ "its first events. The document says so in its own text as well."
856
+ )
857
+ if open_html and "html" in written:
858
+ payload["opened"] = _open_document(Path(written["html"]["path"]))
859
+ return payload
860
+
861
+
862
+ def _notebooksuffix_for(name: str, document: Mapping[str, Any]) -> str:
863
+ """``.ipynb`` for the notebook, and the project's own table for everything else.
864
+
865
+ The HTML arm compares the PARSED media type, not the header Studio sent. Studio serves the
866
+ rendered page as ``text/html; charset=utf-8``, and an equality test against ``text/html``
867
+ missed it silently: the one document a person opens in a browser was written as ``.bin``.
868
+ """
869
+
870
+ if name == "ipynb":
871
+ return ".ipynb"
872
+ if media_type_essence(document.get("media_type")) == "text/html":
873
+ return ".html"
874
+ return suffix_for(document.get("media_type"))
875
+
876
+
877
+ def _notebook_session(document: Mapping[str, Any]) -> dict[str, Any]:
878
+ """One notebook document, in the shape the signed-download verifier already checks.
879
+
880
+ The notebook resource carries its digest and size beside the URL rather than inside a signed
881
+ session object, so they are moved onto the shape :func:`download_signed_artifact` verifies
882
+ rather than a second verifier being written. Both facts stay mandatory: a document with no
883
+ digest is refused here, before a byte is fetched.
884
+ """
885
+
886
+ download = document.get("download")
887
+ if not isinstance(download, Mapping):
888
+ raise ThinLaneError("THIN_RESPONSE_INVALID", "a notebook document carried no download")
889
+ return {
890
+ "signed_url": download.get("url", ""),
891
+ "expected_content_digest": document.get("content_digest"),
892
+ "expected_size_bytes": document.get("size_bytes"),
893
+ "media_type": document.get("media_type"),
894
+ }
895
+
896
+
897
+ def _open_document(path: Path, opener: Callable[[str], bool] | None = None) -> bool:
898
+ """Open one written document in whatever this desktop opens HTML with.
899
+
900
+ ``webbrowser`` is standard library and never a hard failure: a machine with no browser -- a
901
+ build agent, a container, somebody over SSH -- gets the file it asked for and a receipt saying
902
+ it was not opened, rather than a refusal about a document that is already on disk.
903
+ """
904
+
905
+ import webbrowser
906
+
907
+ selected = opener or webbrowser.open
908
+ try:
909
+ return bool(selected(path.resolve().as_uri()))
910
+ except Exception: # pragma: no cover - a desktop that answers by raising
911
+ return False
912
+
913
+
914
+ # --------------------------------------------------------------------------------------------
915
+ # Looking at the data, and at a deployment's run
916
+ # --------------------------------------------------------------------------------------------
917
+
918
+
919
+ def peek(args: argparse.Namespace, *, client: StudioRunClient | None = None) -> dict[str, Any]:
920
+ """``mr-data peek``: the columns, types and first rows of one hosted table version.
921
+
922
+ The local command opens a file or an address on this computer. Hosted, the thing to look at is
923
+ the bounded preview Studio published beside the released version, named by the run that
924
+ released it. Every argument about opening bytes here -- the format, the Reader, the JSON
925
+ shaping -- is reported as having no effect, because the preview arrives already decoded by the
926
+ Reader the recipe pinned.
927
+
928
+ ``--rows`` narrows what is shown and cannot widen it: the preview's size was decided by the
929
+ release, and a client asking for more rows than were published would be asking Studio to
930
+ re-read the table.
931
+ """
932
+
933
+ run_id = _run_id(args.target)
934
+ selected = client or _client(args)
935
+ run = selected.get_run(run_id)
936
+ released = _released_version(selected, run)
937
+ if released is None:
938
+ raise ThinLaneError(
939
+ "THIN_NO_RELEASED_VERSION",
940
+ "this run has released no version to look at; a hosted preview is published by the "
941
+ "release, and Studio has this run as " + str(run.get("status")),
942
+ )
943
+ table_id, version_id = released
944
+ preview = selected.table_version_preview(table_id, version_id)
945
+ rows = preview.get("rows")
946
+ rows = rows if isinstance(rows, list) else []
947
+ wanted = getattr(args, "rows", None)
948
+ shown = rows[: int(wanted)] if isinstance(wanted, int) else rows
949
+ payload: dict[str, Any] = {
950
+ "schema_version": PEEK_SCHEMA,
951
+ "status": "peeked",
952
+ **_hosted_head(selected.session, run_id),
953
+ "source": f"{table_id}/{version_id}",
954
+ "origin": "hosted",
955
+ "table_id": table_id,
956
+ "table_version_id": version_id,
957
+ "row_count": preview.get("total_row_count"),
958
+ "columns": _column_names(preview),
959
+ "schema": _schema_payload(preview),
960
+ "sample": _numbered("row", shown),
961
+ "input_sha256": preview.get("source_parquet_digest"),
962
+ "candidate_digest": preview.get("candidate_digest"),
963
+ "flags_without_effect": _no_effect(
964
+ args,
965
+ (
966
+ (
967
+ "data_format",
968
+ "--format",
969
+ "the preview arrives decoded; there are no bytes here to name a format for",
970
+ ),
971
+ ("reader", "--reader", "the Reader was pinned by the recipe that built this"),
972
+ (
973
+ "reader_options",
974
+ "--reader-options",
975
+ "the Reader was pinned by the recipe that built this",
976
+ ),
977
+ (
978
+ "allow_redirect_hosts",
979
+ "--allow-redirect-host",
980
+ "nothing is fetched from a third-party address here",
981
+ ),
982
+ (
983
+ "json_records_pointer",
984
+ "--json-records-pointer",
985
+ "the preview is already rows and columns",
986
+ ),
987
+ ("json_expand", "--json-expand", "the preview is already rows and columns"),
988
+ ("json_column", "--json-column", "the preview is already rows and columns"),
989
+ (
990
+ "json_optional_column",
991
+ "--json-optional-column",
992
+ "the preview is already rows and columns",
993
+ ),
994
+ (
995
+ "json_document_format",
996
+ "--json-document-format",
997
+ "the preview is already rows and columns",
998
+ ),
999
+ ),
1000
+ ),
1001
+ }
1002
+ if preview.get("truncated") is True or len(shown) != len(rows):
1003
+ payload["note"] = (
1004
+ f"Showing {len(shown)} of the {len(rows)} rows this version published a preview for, "
1005
+ f"out of {preview.get('total_row_count')} in the table."
1006
+ )
1007
+ return payload
1008
+
1009
+
1010
+ def deployment_status(
1011
+ args: argparse.Namespace, *, client: StudioRunClient | None = None
1012
+ ) -> dict[str, Any]:
1013
+ """``mr-data deploy-status``: what Studio says about the run one deployment queued.
1014
+
1015
+ The local command finds the run identifier in the deployment journal this computer wrote. A
1016
+ hosted profile never wrote one, so the identifier is named directly -- and everything after
1017
+ that is the same reading of the same record, under the same status word.
1018
+ """
1019
+
1020
+ run_id = _run_id(args.run_dir)
1021
+ selected = client or _client(args)
1022
+ run = selected.get_run(run_id)
1023
+ state = str(run.get("status"))
1024
+ return {
1025
+ "schema_version": DEPLOYMENT_STATUS_SCHEMA,
1026
+ "status": "run_status_reported",
1027
+ **_hosted_head(selected.session, run_id),
1028
+ "run_status": state,
1029
+ "terminal": state in TERMINAL_RUN_STATUSES,
1030
+ "workspace_id": str(selected.session.workspace_id),
1031
+ "run": run_receipt(run),
1032
+ "flags_without_effect": _no_effect(
1033
+ args,
1034
+ (
1035
+ (
1036
+ "watch_seconds",
1037
+ "--watch-seconds",
1038
+ "this profile follows a run with mr-data watch, which streams it rather than "
1039
+ "asking again on a timer",
1040
+ ),
1041
+ ),
1042
+ ),
1043
+ }
1044
+
1045
+
1046
+ __all__ = [
1047
+ "DEFAULT_NOTEBOOK_OUTPUT",
1048
+ "DEPLOYMENT_STATUS_SCHEMA",
1049
+ "DIFF_SCHEMA",
1050
+ "FLEET_NEEDS_A_PERSON",
1051
+ "FLEET_SCHEMA",
1052
+ "FLEET_STATE",
1053
+ "INSPECT_SCHEMA",
1054
+ "LIST_SCHEMA",
1055
+ "NOTEBOOK_SCHEMA",
1056
+ "PEEK_SCHEMA",
1057
+ "SHOW_SCHEMA",
1058
+ "VERIFY_SCHEMA",
1059
+ "deployment_status",
1060
+ "diff",
1061
+ "fleet",
1062
+ "inspect",
1063
+ "listing",
1064
+ "notebook",
1065
+ "peek",
1066
+ "show",
1067
+ "verify",
1068
+ "verify_exit_code",
1069
+ "view",
1070
+ ]