mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,3037 @@
1
+ """Crash-safe deployment of one exact reviewed Build through Cloud and Studio."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import fcntl
6
+ import hashlib
7
+ import importlib
8
+ import ipaddress
9
+ import os
10
+ import re
11
+ import ssl
12
+ import stat
13
+ import time
14
+ from collections.abc import Callable, Mapping
15
+ from dataclasses import dataclass
16
+ from datetime import UTC, datetime
17
+ from pathlib import Path
18
+ from typing import Any, Protocol
19
+ from urllib.error import HTTPError
20
+ from urllib.parse import urlsplit
21
+ from urllib.request import HTTPRedirectHandler, HTTPSHandler, ProxyHandler, Request, build_opener
22
+ from uuid import NAMESPACE_URL, UUID, uuid5
23
+
24
+ from croniter import CroniterBadCronError, CroniterBadDateError, croniter
25
+
26
+ from mostlyright.data_harness import canonical, deployment_evidence, pipeline
27
+ from mostlyright.data_harness.deploy import DeploymentRequest
28
+ from mostlyright.data_harness.hosted_crawler_protocol import (
29
+ OPENLIGADB_EGRESS_POLICY_ATTESTATION,
30
+ PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION,
31
+ )
32
+ from mostlyright.data_harness.studio_boundary import StudioBoundaryError, client_contract_identity
33
+ from mostlyright.data_harness.ux.credentials import ResolvedCloudCredentials
34
+
35
+ CLOUD_TOKEN_PATH = "/api/cli/studio-token"
36
+ CLOUD_DATASET_BINDINGS_PATH = "/api/cli/table-bindings"
37
+ CLOUD_TABLE_BINDING_SCHEMA_VERSION = "mostlyright-cloud-table-binding.v1"
38
+ CLOUD_TABLE_BINDING_CONTRACT_SHA256 = (
39
+ "f408d3ce54994624778e74186077557228bfd89f95ab882a152efa1b5af111fd"
40
+ )
41
+ DEPLOYMENT_STATE_SCHEMA = "mostlyright-hosted-deployment-state.v3"
42
+ DEPLOYMENT_RECEIPT_SCHEMA = "mostlyright-hosted-deployment-receipt.v3"
43
+ STUDIO_SCHEMA_VERSION = "3.0.0"
44
+ REQUEST_TIMEOUT_SECONDS = 30.0
45
+ #: How many EXTRA attempts one transient Cloud 503 earns before this deployment
46
+ #: refuses.
47
+ #:
48
+ #: ⚠ WHY THIS EXISTS AT ALL. Studio runs behind Cloud and scales to zero; a cold
49
+ #: start takes ~20s to become servable. Cloud provisions this workspace's Studio
50
+ #: tenancy on EVERY token exchange, so the request that WAKES Studio is the same
51
+ #: request that has to wait for it. Before 2026-08-20 that request simply
52
+ #: answered 503 and this function raised DEPLOY_STUDIO_UNAVAILABLE immediately --
53
+ #: which meant `mr-data` could never deploy after any idle period, because every
54
+ #: attempt woke Studio, gave up on it, and left it to scale back to zero. Eight
55
+ #: such failures were recorded against production on 2026-08-20 alone.
56
+ #:
57
+ #: Two attempts is deliberate rather than generous: Cloud's own provisioning
58
+ #: budget already clears the observed cold start, so a retry here is the second
59
+ #: line, covering the case where Studio was not merely cold but genuinely
60
+ #: restarting.
61
+ TOKEN_EXCHANGE_RETRY_ATTEMPTS = 2
62
+ #: Ceiling on any single retry wait, however large a `retry-after` Cloud sends.
63
+ #: A deployment must stay a bounded, interruptible foreground operation.
64
+ TOKEN_EXCHANGE_MAX_RETRY_SECONDS = 30.0
65
+ MAX_TOKEN_RESPONSE_BYTES = 32 * 1024
66
+ MAX_BINDING_RESPONSE_BYTES = 32 * 1024
67
+ MAX_STATE_BYTES = 8 * 1024 * 1024
68
+ STATE_RELATIVE_PATH = Path("evidence") / "deployment" / "state.json"
69
+
70
+ _DIGEST = re.compile(r"^[0-9a-f]{64}$")
71
+ _PREFIXED_DIGEST = re.compile(r"^sha256:[0-9a-f]{64}$")
72
+ _IMMUTABLE_IMAGE = re.compile(r"^[a-z0-9][a-z0-9._/-]{0,254}@sha256:[0-9a-f]{64}$")
73
+ _OBJECT_GENERATION = re.compile(r"^[1-9][0-9]{0,31}$")
74
+ _CLI_KEY = re.compile(r"^mr_cli_[A-Za-z0-9_-]{64}$")
75
+
76
+
77
+ class HostedDeployError(RuntimeError):
78
+ """A typed, credential-safe deployment refusal."""
79
+
80
+ def __init__(self, code: str, detail: str) -> None:
81
+ self.code = code
82
+ self.detail = detail
83
+ super().__init__(f"{code}: {detail}")
84
+
85
+
86
+ @dataclass(frozen=True)
87
+ class StudioToken:
88
+ studio_base_url: str
89
+ workspace_id: UUID
90
+ token: str
91
+ expires_at: datetime
92
+
93
+ def __repr__(self) -> str:
94
+ return (
95
+ "StudioToken("
96
+ f"studio_base_url={self.studio_base_url!r}, workspace_id={self.workspace_id!r}, "
97
+ f"token='<redacted>', expires_at={self.expires_at!r})"
98
+ )
99
+
100
+
101
+ @dataclass(frozen=True)
102
+ class StudioResponse:
103
+ status_code: int
104
+ body: Mapping[str, Any]
105
+ etag: str | None = None
106
+
107
+
108
+ class TokenExchangeTransport(Protocol):
109
+ def request(
110
+ self,
111
+ method: str,
112
+ url: str,
113
+ headers: Mapping[str, str],
114
+ body: bytes | None,
115
+ maximum: int,
116
+ ) -> tuple[int, bytes, Mapping[str, str]]: ...
117
+
118
+
119
+ class StudioDeploymentClient(Protocol):
120
+ def call(
121
+ self,
122
+ operation: str,
123
+ body: Mapping[str, Any] | None,
124
+ *,
125
+ idempotency_key: str | None = None,
126
+ resource_id: UUID | None = None,
127
+ if_match: str | None = None,
128
+ ) -> StudioResponse: ...
129
+
130
+ def preflight(self, body: Mapping[str, Any]) -> StudioResponse: ...
131
+
132
+ def read(self, operation: str, *, resource_id: UUID | None = None) -> StudioResponse: ...
133
+
134
+ def upload(self, session: Mapping[str, Any], content: bytes) -> str: ...
135
+
136
+ def close(self) -> None: ...
137
+
138
+
139
+ class DeploymentStateStore(Protocol):
140
+ @property
141
+ def path(self) -> Path: ...
142
+
143
+ def exists(self) -> bool: ...
144
+
145
+ def load(self) -> dict[str, Any]: ...
146
+
147
+ def save(self, state: Mapping[str, Any]) -> None: ...
148
+
149
+ def close(self) -> None: ...
150
+
151
+
152
+ class _NoRedirect(HTTPRedirectHandler):
153
+ def redirect_request(
154
+ self,
155
+ request: Request,
156
+ file_pointer: Any,
157
+ code: int,
158
+ message: str,
159
+ headers: Any,
160
+ new_url: str,
161
+ ) -> None:
162
+ return None
163
+
164
+
165
+ class UrlLibTokenExchangeTransport:
166
+ """Bounded Cloud token exchange with ambient proxies and redirects disabled."""
167
+
168
+ def __init__(self) -> None:
169
+ self._opener = build_opener(
170
+ ProxyHandler({}), HTTPSHandler(context=ssl.create_default_context()), _NoRedirect()
171
+ )
172
+
173
+ def request(
174
+ self,
175
+ method: str,
176
+ url: str,
177
+ headers: Mapping[str, str],
178
+ body: bytes | None,
179
+ maximum: int,
180
+ ) -> tuple[int, bytes, Mapping[str, str]]:
181
+ request = Request(url, data=body, headers=dict(headers), method=method)
182
+ try:
183
+ response = self._opener.open(request, timeout=REQUEST_TIMEOUT_SECONDS)
184
+ except HTTPError as error:
185
+ response = error
186
+ with response:
187
+ raw = response.read(maximum + 1)
188
+ status = int(response.status)
189
+ response_headers = {key.lower(): value for key, value in response.headers.items()}
190
+ if len(raw) > maximum:
191
+ raise HostedDeployError(
192
+ "DEPLOY_TOKEN_EXCHANGE_FAILED", "the Cloud token response exceeds its byte limit"
193
+ )
194
+ return status, raw, response_headers
195
+
196
+
197
+ def _transient_retry_seconds(response_headers: Mapping[str, str]) -> float | None:
198
+ """Seconds Cloud asked this deployment to wait, or ``None`` if it asked at all.
199
+
200
+ ⚠ THE ABSENCE OF THE HEADER IS THE SIGNAL, not a missing detail to paper
201
+ over. Cloud answers a 503 in exactly two shapes, and they mean opposite
202
+ things:
203
+
204
+ * ``serviceUnavailable()`` -- "our failure, not yours; please retry" --
205
+ stamps ``retry-after``. This is the cold-start case, and it is worth
206
+ retrying.
207
+ * ``studioUnconfigured()`` -- Studio is not configured on that deployment
208
+ at all -- stamps NO ``retry-after``. Retrying it would burn a bounded
209
+ deployment budget on a condition no amount of waiting can change.
210
+
211
+ Both bodies are deliberately generic (Cloud does not leak which of its
212
+ environment variables is at fault), so this header is the ONLY thing that
213
+ distinguishes them from out here. Returning ``None`` for a malformed or
214
+ out-of-range value fails toward the safe reading -- refuse now rather than
215
+ sleep on a number we do not trust.
216
+ """
217
+
218
+ raw = response_headers.get("retry-after", "")
219
+ if not raw.isdecimal():
220
+ return None
221
+ seconds = int(raw)
222
+ if not 1 <= seconds <= 3600:
223
+ return None
224
+ return float(min(seconds, TOKEN_EXCHANGE_MAX_RETRY_SECONDS))
225
+
226
+
227
+ def exchange_studio_token(
228
+ credentials: ResolvedCloudCredentials,
229
+ *,
230
+ transport: TokenExchangeTransport | None = None,
231
+ now: Callable[[], datetime] = lambda: datetime.now(UTC),
232
+ sleep: Callable[[float], None] = time.sleep,
233
+ ) -> StudioToken:
234
+ """Exchange the selected Cloud credential for a short-lived human Studio token."""
235
+
236
+ _require_cli_credential(credentials)
237
+ cloud_url = _service_url(credentials.cloud_url, "Cloud URL", allow_loopback_http=False)
238
+ selected = transport or UrlLibTokenExchangeTransport()
239
+ attempts_remaining = TOKEN_EXCHANGE_RETRY_ATTEMPTS
240
+ while True:
241
+ try:
242
+ status, raw, response_headers = selected.request(
243
+ "POST",
244
+ f"{cloud_url}{CLOUD_TOKEN_PATH}",
245
+ {"Accept": "application/json", "x-api-key": credentials.raw_key},
246
+ b"",
247
+ MAX_TOKEN_RESPONSE_BYTES,
248
+ )
249
+ except HostedDeployError:
250
+ raise
251
+ except Exception as error:
252
+ raise HostedDeployError(
253
+ "DEPLOY_TOKEN_EXCHANGE_FAILED", "the Cloud token exchange could not be reached"
254
+ ) from error
255
+ if status != 503 or attempts_remaining == 0:
256
+ break
257
+ delay = _transient_retry_seconds(response_headers)
258
+ if delay is None:
259
+ break
260
+ attempts_remaining -= 1
261
+ sleep(delay)
262
+ try:
263
+ parsed = canonical.parse_json(raw)
264
+ except canonical.CanonicalJSONError:
265
+ parsed = None
266
+ if status == 401:
267
+ raise HostedDeployError(
268
+ "DEPLOY_AUTHENTICATION_FAILED", "the selected Cloud credential was rejected"
269
+ )
270
+ if status == 402:
271
+ raise HostedDeployError(
272
+ "DEPLOY_SUBSCRIPTION_REQUIRED", "this workspace has no hosted deployment access"
273
+ )
274
+ if status == 429:
275
+ retry_after = response_headers.get("retry-after", "")
276
+ suffix = (
277
+ f" after {retry_after} seconds"
278
+ if retry_after.isdecimal() and 1 <= int(retry_after) <= 3600
279
+ else ""
280
+ )
281
+ raise HostedDeployError(
282
+ "DEPLOY_RATE_LIMITED", f"Cloud asked this deployment to retry{suffix}"
283
+ )
284
+ if status == 503:
285
+ raise HostedDeployError(
286
+ "DEPLOY_STUDIO_UNAVAILABLE",
287
+ "Studio is not configured, or stayed unavailable across every attempt",
288
+ )
289
+ if status != 200 or not isinstance(parsed, dict):
290
+ raise HostedDeployError(
291
+ "DEPLOY_TOKEN_EXCHANGE_FAILED", f"the Cloud token exchange failed (HTTP {status})"
292
+ )
293
+ base_url = _required_text(parsed, "studio_base_url")
294
+ workspace_id = _uuid(_required_text(parsed, "workspace_id"), "workspace_id")
295
+ token = _required_text(parsed, "token")
296
+ expires_at = _timestamp(_required_text(parsed, "expires_at"), "expires_at")
297
+ if expires_at <= now():
298
+ raise HostedDeployError(
299
+ "DEPLOY_TOKEN_EXCHANGE_FAILED", "the Cloud returned an expired Studio token"
300
+ )
301
+ return StudioToken(
302
+ studio_base_url=_service_url(base_url, "Studio base URL", allow_loopback_http=True),
303
+ workspace_id=workspace_id,
304
+ token=token,
305
+ expires_at=expires_at,
306
+ )
307
+
308
+
309
+ _OPERATIONS: dict[str, tuple[str, str]] = {
310
+ "create_dataset": ("ContractCreateDatasetRequest", "create_dataset"),
311
+ "create_question": ("ContractCreateQuestionRequest", "create_question"),
312
+ "put_requirements": ("ContractUpsertCommand", "put_question_requirements"),
313
+ "register_source": ("ContractRegisterCommand", "register_source"),
314
+ "create_plan": ("ContractCreateCommand", "create_table_plan"),
315
+ "create_source_staging": ("CreateSourceStagingCommand", "create_source_staging"),
316
+ "complete_artifact_upload": (
317
+ "ContractCompleteArtifactUploadCommand",
318
+ "complete_artifact_upload",
319
+ ),
320
+ "finalize_source_staging": ("FinalizeSourceStagingCommand", "finalize_source_staging"),
321
+ "register_connector": (
322
+ "ConnectorConfigurationRegistrationCommand",
323
+ "register_connector_configuration",
324
+ ),
325
+ "create_proposal": ("ContractProposalCommand", "create_recipe_proposal"),
326
+ "request_approval": ("ContractApprovalCommand", "request_recipe_approval"),
327
+ # This is intentionally the dedicated V3 Table-recipe operation. The coordinated Studio client
328
+ # exposes only confirmation of the exact Table Recipe approval the Editor just requested, not
329
+ # generic approval decisions. Never map this to the generic V3 approval route or the retired V2
330
+ # recipe confirmation route.
331
+ "confirm_table_recipe_approval": (
332
+ "ContractDecisionCommand",
333
+ "confirm_table_recipe_approval",
334
+ ),
335
+ "activate_recipe": ("ContractActivateRecipeCommand", "activate_recipe"),
336
+ }
337
+
338
+ #: The read operations, as the generated method each one resolves to and the keyword it names its
339
+ #: resource with.
340
+ #:
341
+ #: ⚠ A READ IS A DIFFERENT SHAPE, NOT A RARER WRITE. Every entry in `_OPERATIONS` above is built
342
+ #: the same way: a request model from a body, an idempotency key, and a journaled plan written
343
+ #: before the request is sent, because each of those operations changes something in Studio and
344
+ #: has to be replayable. A read carries no body and no idempotency key, so there is nothing for
345
+ #: that path to build and nothing for the journal to replay -- which is why `get_run` is here
346
+ #: rather than as a thirteenth entry above, and why nothing on this path can mutate a Run by
347
+ #: accident: the method names reachable through `read` are exactly these.
348
+ _READ_OPERATIONS: dict[str, tuple[str, str]] = {
349
+ "get_proposal": ("get_recipe_proposal", "recipe_proposal_id"),
350
+ "get_run": ("get_run", "run_id"),
351
+ }
352
+
353
+ # The generated V3 distribution exports operations as modules. Keep the deployment surface
354
+ # closed by resolving only this audited inventory; no caller-provided operation name reaches an
355
+ # import path.
356
+ _GENERATED_OPERATION_MODULES = {
357
+ "create_dataset": ("datasets", "create_dataset"),
358
+ "create_question": ("questions", "create_question"),
359
+ "put_question_requirements": ("questions", "put_question_requirements"),
360
+ "register_source": ("sources", "register_source"),
361
+ "create_table_plan": ("plans", "create_table_plan"),
362
+ "create_source_staging": ("artifacts", "create_source_staging"),
363
+ "complete_artifact_upload": ("artifacts", "complete_artifact_upload"),
364
+ "finalize_source_staging": ("artifacts", "finalize_source_staging"),
365
+ "register_connector_configuration": ("connectors", "register_connector_configuration"),
366
+ "create_recipe_proposal": ("recipes", "create_recipe_proposal"),
367
+ "create_worker_policy_preflight": ("recipes", "create_worker_policy_preflight"),
368
+ "request_recipe_approval": ("recipes", "request_recipe_approval"),
369
+ "activate_recipe": ("recipes", "activate_recipe"),
370
+ "get_recipe_proposal": ("recipes", "get_recipe_proposal"),
371
+ "get_run": ("runs", "get_run"),
372
+ }
373
+
374
+
375
+ def build_editor_client(token: StudioToken) -> tuple[Any, Any]:
376
+ """Return the pinned V3 Table-recipe confirmation facade and its transport.
377
+
378
+ Written once because two lanes need the same thing: the deployment writes through this facade
379
+ and the dataset handoff reads through it, and a second copy of the construction would be a
380
+ second place for the contract pin, the redirect policy, or the timeout to be got wrong.
381
+
382
+ Raises:
383
+ HostedDeployError: when the pinned client is absent or incompatible.
384
+ """
385
+
386
+ try:
387
+ client_contract_identity()
388
+ client_module = importlib.import_module("mostlyright_studio.client")
389
+ authority_module = importlib.import_module("mostlyright_studio.authority")
390
+ authenticated = client_module.AuthenticatedClient(
391
+ base_url=token.studio_base_url,
392
+ token=token.token,
393
+ timeout=REQUEST_TIMEOUT_SECONDS,
394
+ follow_redirects=False,
395
+ )
396
+ return authenticated, authority_module.StudioV3RecipeConfirmationClient(authenticated)
397
+ except (ImportError, AttributeError, StudioBoundaryError) as error:
398
+ raise HostedDeployError("DEPLOY_STUDIO_CLIENT_INVALID", str(error)) from error
399
+
400
+
401
+ class GeneratedStudioDeploymentClient:
402
+ """All remote deployment operations through the pinned generated editor facade."""
403
+
404
+ def __init__(self, token: StudioToken) -> None:
405
+ self._raw_client, self._editor_client = build_editor_client(token)
406
+ try:
407
+ self._models = importlib.import_module("mostlyright_studio.models")
408
+ except ImportError as error:
409
+ raise HostedDeployError("DEPLOY_STUDIO_CLIENT_INVALID", str(error)) from error
410
+ confirmation = getattr(self._editor_client, "confirm_table_recipe_approval", None)
411
+ if confirmation is None or not callable(getattr(confirmation, "sync_detailed", None)):
412
+ raise HostedDeployError(
413
+ "DEPLOY_STUDIO_CLIENT_INVALID",
414
+ "the pinned Studio client does not expose V3 Table-recipe confirmation; "
415
+ "install the coordinated regenerated Studio client before deploying",
416
+ )
417
+ try:
418
+ preflight = self._operation("create_worker_policy_preflight")
419
+ except ImportError as error:
420
+ raise HostedDeployError(
421
+ "DEPLOY_STUDIO_CLIENT_INVALID",
422
+ "the installed generated Studio client does not expose "
423
+ "createWorkerPolicyPreflight; merge Studio #80 and install its regenerated client "
424
+ "before deploying",
425
+ ) from error
426
+ if not callable(getattr(preflight, "sync_detailed", None)):
427
+ raise HostedDeployError(
428
+ "DEPLOY_STUDIO_CLIENT_INVALID",
429
+ "the installed generated Studio client has no callable "
430
+ "createWorkerPolicyPreflight operation; merge Studio #80 and install its "
431
+ "regenerated client before deploying",
432
+ )
433
+ self._base_url = token.studio_base_url
434
+
435
+ def call(
436
+ self,
437
+ operation: str,
438
+ body: Mapping[str, Any] | None,
439
+ *,
440
+ idempotency_key: str | None = None,
441
+ resource_id: UUID | None = None,
442
+ if_match: str | None = None,
443
+ ) -> StudioResponse:
444
+ if operation in _READ_OPERATIONS:
445
+ return self.read(operation, resource_id=resource_id)
446
+ coordinate = _OPERATIONS.get(operation)
447
+ if coordinate is None or body is None or idempotency_key is None:
448
+ raise HostedDeployError("DEPLOY_REQUEST_INVALID", "deployment operation is incomplete")
449
+ model_name, method_name = coordinate
450
+ model = self._model(model_name, body)
451
+ kwargs: dict[str, Any] = {"body": model, "idempotency_key": idempotency_key}
452
+ if operation == "put_requirements":
453
+ kwargs.update(question_id=_required_uuid(resource_id, "question_id"), if_match=if_match)
454
+ elif operation in {"finalize_source_staging"}:
455
+ kwargs["source_staging_id"] = _required_uuid(resource_id, "source_staging_id")
456
+ elif operation == "complete_artifact_upload":
457
+ kwargs["artifact_id"] = _required_uuid(resource_id, "artifact_id")
458
+ elif operation in {"request_approval", "activate_recipe"}:
459
+ kwargs.update(
460
+ recipe_proposal_id=_required_uuid(resource_id, "recipe_proposal_id"),
461
+ if_match=if_match,
462
+ )
463
+ elif operation == "confirm_table_recipe_approval":
464
+ kwargs.update(
465
+ approval_request_id=_required_uuid(resource_id, "approval_request_id"),
466
+ if_match=if_match,
467
+ )
468
+ if operation == "confirm_table_recipe_approval":
469
+ confirmation = getattr(self._editor_client, method_name, None)
470
+ if confirmation is None or not callable(getattr(confirmation, "sync_detailed", None)):
471
+ raise HostedDeployError(
472
+ "DEPLOY_STUDIO_CLIENT_INVALID",
473
+ "the installed generated Studio client does not expose "
474
+ "StudioV3RecipeConfirmationClient.confirm_table_recipe_approval; pin the "
475
+ "coordinated "
476
+ "regenerated Studio client before deployment",
477
+ )
478
+ response = confirmation.sync_detailed(**kwargs)
479
+ else:
480
+ response = self._operation(method_name).sync_detailed(client=self._raw_client, **kwargs)
481
+ return _generated_response(response)
482
+
483
+ def preflight(self, body: Mapping[str, Any]) -> StudioResponse:
484
+ """Resolve Studio's current worker policy without journaling it as a mutation."""
485
+
486
+ command = self._model("ContractWorkerPolicyPreflightCommand", body)
487
+ response = self._operation("create_worker_policy_preflight").sync_detailed(
488
+ client=self._raw_client,
489
+ body=command,
490
+ )
491
+ return _generated_response(response)
492
+
493
+ def read(self, operation: str, *, resource_id: UUID | None = None) -> StudioResponse:
494
+ """One bounded read of one Studio resource, sending no body and changing nothing.
495
+
496
+ Transport failures are deliberately not translated here. What a caller should say about an
497
+ unreachable Studio depends on what it was reading -- a status check that could not be made
498
+ is a retryable answer, an approved proposal that could not be fetched stops a deployment --
499
+ so the refusal is written where that is known rather than flattened into one code here.
500
+ """
501
+
502
+ coordinate = _READ_OPERATIONS.get(operation)
503
+ if coordinate is None:
504
+ raise HostedDeployError("DEPLOY_REQUEST_INVALID", "deployment operation is incomplete")
505
+ method_name, parameter = coordinate
506
+ response = self._operation(method_name).sync_detailed(
507
+ client=self._raw_client,
508
+ **{parameter: _required_uuid(resource_id, parameter)},
509
+ )
510
+ return _generated_response(response)
511
+
512
+ @staticmethod
513
+ def _operation(method_name: str) -> Any:
514
+ coordinate = _GENERATED_OPERATION_MODULES.get(method_name)
515
+ if coordinate is None:
516
+ raise HostedDeployError("DEPLOY_REQUEST_INVALID", "deployment operation is incomplete")
517
+ tag, module = coordinate
518
+ return importlib.import_module(f"mostlyright_studio.api.{tag}.{module}")
519
+
520
+ def upload(self, session: Mapping[str, Any], content: bytes) -> str:
521
+ expected_digest = "sha256:" + hashlib.sha256(content).hexdigest()
522
+ if (
523
+ session.get("direction") != "upload"
524
+ or session.get("method") != "PUT"
525
+ or session.get("route_authority") != "source_staging_editor"
526
+ or session.get("expected_content_digest") != expected_digest
527
+ or session.get("expected_size_bytes") != len(content)
528
+ ):
529
+ raise HostedDeployError(
530
+ "DEPLOY_SIGNED_SESSION_INVALID", "Studio returned a mismatched upload capability"
531
+ )
532
+ try:
533
+ import httpx
534
+ except ImportError as error:
535
+ raise HostedDeployError(
536
+ "DEPLOY_STUDIO_CLIENT_INVALID", "the hosted extra does not contain httpx"
537
+ ) from error
538
+ headers = _signed_headers(session, size=len(content))
539
+ try:
540
+ with httpx.Client(
541
+ timeout=REQUEST_TIMEOUT_SECONDS, follow_redirects=False, trust_env=False
542
+ ) as client:
543
+ response = client.put(
544
+ _transfer_url(_required_text(session, "signed_url"), self._base_url),
545
+ headers=headers,
546
+ content=content,
547
+ )
548
+ except httpx.RequestError as error:
549
+ raise HostedDeployError(
550
+ "DEPLOY_SOURCE_UPLOAD_FAILED", "the signed source upload could not be reached"
551
+ ) from error
552
+ if response.status_code not in {200, 201}:
553
+ if 400 <= response.status_code < 500:
554
+ raise HostedDeployError(
555
+ "DEPLOY_SOURCE_UPLOAD_SESSION_REJECTED",
556
+ "the signed source upload capability was rejected and must be renewed",
557
+ )
558
+ raise HostedDeployError(
559
+ "DEPLOY_SOURCE_UPLOAD_FAILED",
560
+ f"the signed source upload returned HTTP {response.status_code}",
561
+ )
562
+ generation = response.headers.get(
563
+ "x-mostlyright-object-generation", response.headers.get("x-goog-generation", "")
564
+ )
565
+ if _OBJECT_GENERATION.fullmatch(generation) is None:
566
+ raise HostedDeployError(
567
+ "DEPLOY_SOURCE_UPLOAD_FAILED", "the upload response omitted object generation"
568
+ )
569
+ return generation
570
+
571
+ def close(self) -> None:
572
+ client = getattr(self._raw_client, "_client", None)
573
+ if client is not None:
574
+ client.close()
575
+
576
+ def _model(self, name: str, body: Mapping[str, Any]) -> Any:
577
+ model_type = getattr(self._models, name, None)
578
+ if model_type is None:
579
+ raise HostedDeployError(
580
+ "DEPLOY_STUDIO_CLIENT_INVALID", f"the generated Studio client lacks {name}"
581
+ )
582
+ try:
583
+ model = model_type.from_dict(dict(body))
584
+ except (KeyError, TypeError, ValueError) as error:
585
+ raise HostedDeployError(
586
+ "DEPLOY_REQUEST_INVALID", f"the deployment handoff is not a valid {name}"
587
+ ) from error
588
+ if model.to_dict() != dict(body):
589
+ raise HostedDeployError(
590
+ "DEPLOY_REQUEST_INVALID", f"the deployment handoff is not the exact {name} shape"
591
+ )
592
+ return model
593
+
594
+
595
+ class FileDeploymentStateStore:
596
+ """Canonical, atomically replaced non-secret journal under the reviewed run directory."""
597
+
598
+ def __init__(self, run_dir: Path) -> None:
599
+ self._run_dir = Path(run_dir).absolute()
600
+ self._path = self._run_dir / STATE_RELATIVE_PATH
601
+ self._run_handle = None
602
+ self._directory_descriptors: list[int] = []
603
+ self._directory_bindings: list[tuple[int, str, int, int, int]] = []
604
+ try:
605
+ self._run_handle = pipeline._open_candidate_run_handle(self._run_dir)
606
+ evidence = self._open_directory(self._run_handle.run_fd, "evidence")
607
+ deployment = self._open_directory(evidence, "deployment")
608
+ self._deployment_fd = deployment
609
+ fcntl.flock(self._deployment_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
610
+ self._validate()
611
+ except (OSError, pipeline.BuildError) as error:
612
+ self.close()
613
+ raise HostedDeployError(
614
+ "DEPLOY_STATE_UNAVAILABLE",
615
+ "the deployment journal directory is unavailable, unsafe, or already in use",
616
+ ) from error
617
+
618
+ @property
619
+ def path(self) -> Path:
620
+ return self._path
621
+
622
+ def exists(self) -> bool:
623
+ self._validate()
624
+ try:
625
+ os.stat("state.json", dir_fd=self._deployment_fd, follow_symlinks=False)
626
+ except FileNotFoundError:
627
+ return False
628
+ except OSError as error:
629
+ raise HostedDeployError(
630
+ "DEPLOY_STATE_INVALID", "the deployment journal is unavailable or unsafe"
631
+ ) from error
632
+ return True
633
+
634
+ def load(self) -> dict[str, Any]:
635
+ try:
636
+ descriptor = os.open(
637
+ "state.json",
638
+ os.O_RDONLY
639
+ | getattr(os, "O_NOFOLLOW", 0)
640
+ | getattr(os, "O_NONBLOCK", 0)
641
+ | getattr(os, "O_CLOEXEC", 0),
642
+ dir_fd=self._deployment_fd,
643
+ )
644
+ except OSError as error:
645
+ raise HostedDeployError(
646
+ "DEPLOY_STATE_INVALID", "the deployment journal is unavailable or unsafe"
647
+ ) from error
648
+ try:
649
+ info = os.fstat(descriptor)
650
+ if (
651
+ not stat.S_ISREG(info.st_mode)
652
+ or not 0 < info.st_size <= MAX_STATE_BYTES
653
+ or info.st_nlink != 1
654
+ ):
655
+ raise HostedDeployError(
656
+ "DEPLOY_STATE_INVALID", "the deployment journal is not a bounded regular file"
657
+ )
658
+ chunks: list[bytes] = []
659
+ remaining = MAX_STATE_BYTES + 1
660
+ while remaining:
661
+ chunk = os.read(descriptor, min(64 * 1024, remaining))
662
+ if not chunk:
663
+ break
664
+ chunks.append(chunk)
665
+ remaining -= len(chunk)
666
+ raw = b"".join(chunks)
667
+ after = os.fstat(descriptor)
668
+ named = os.stat("state.json", dir_fd=self._deployment_fd, follow_symlinks=False)
669
+ if _file_identity(after) != _file_identity(info) or _file_identity(
670
+ named
671
+ ) != _file_identity(info):
672
+ raise HostedDeployError(
673
+ "DEPLOY_STATE_INVALID", "the deployment journal changed while being read"
674
+ )
675
+ finally:
676
+ os.close(descriptor)
677
+ self._validate()
678
+ try:
679
+ value = canonical.parse_canonical_json(raw)
680
+ except canonical.CanonicalJSONError as error:
681
+ raise HostedDeployError(
682
+ "DEPLOY_STATE_INVALID", "the deployment journal is not canonical JSON"
683
+ ) from error
684
+ if not isinstance(value, dict):
685
+ raise HostedDeployError(
686
+ "DEPLOY_STATE_INVALID", "the deployment journal must be one object"
687
+ )
688
+ return value
689
+
690
+ def save(self, state: Mapping[str, Any]) -> None:
691
+ raw = canonical.canonical_json_bytes(dict(state))
692
+ if len(raw) > MAX_STATE_BYTES:
693
+ raise HostedDeployError("DEPLOY_STATE_INVALID", "the deployment journal is too large")
694
+ self._validate()
695
+ temporary = f".state.{os.getpid()}.{os.urandom(8).hex()}.partial"
696
+ descriptor = -1
697
+ try:
698
+ descriptor = os.open(
699
+ temporary,
700
+ os.O_WRONLY
701
+ | os.O_CREAT
702
+ | os.O_EXCL
703
+ | getattr(os, "O_NOFOLLOW", 0)
704
+ | getattr(os, "O_CLOEXEC", 0),
705
+ 0o600,
706
+ dir_fd=self._deployment_fd,
707
+ )
708
+ offset = 0
709
+ while offset < len(raw):
710
+ offset += os.write(descriptor, raw[offset:])
711
+ os.fsync(descriptor)
712
+ os.close(descriptor)
713
+ descriptor = -1
714
+ self._validate()
715
+ os.replace(
716
+ temporary,
717
+ "state.json",
718
+ src_dir_fd=self._deployment_fd,
719
+ dst_dir_fd=self._deployment_fd,
720
+ )
721
+ os.fsync(self._deployment_fd)
722
+ self._validate()
723
+ except Exception:
724
+ if descriptor >= 0:
725
+ os.close(descriptor)
726
+ try:
727
+ os.unlink(temporary, dir_fd=self._deployment_fd)
728
+ except FileNotFoundError:
729
+ pass
730
+ raise
731
+
732
+ def close(self) -> None:
733
+ descriptors = list(reversed(getattr(self, "_directory_descriptors", [])))
734
+ self._directory_descriptors = []
735
+ for descriptor in descriptors:
736
+ try:
737
+ os.close(descriptor)
738
+ except OSError:
739
+ pass
740
+ handle = getattr(self, "_run_handle", None)
741
+ self._run_handle = None
742
+ if handle is not None:
743
+ handle.close()
744
+
745
+ def _open_directory(self, parent_fd: int, name: str) -> int:
746
+ try:
747
+ os.mkdir(name, 0o700, dir_fd=parent_fd)
748
+ except FileExistsError:
749
+ pass
750
+ descriptor = os.open(
751
+ name,
752
+ os.O_RDONLY
753
+ | getattr(os, "O_DIRECTORY", 0)
754
+ | getattr(os, "O_NOFOLLOW", 0)
755
+ | getattr(os, "O_CLOEXEC", 0),
756
+ dir_fd=parent_fd,
757
+ )
758
+ opened = os.fstat(descriptor)
759
+ if not stat.S_ISDIR(opened.st_mode):
760
+ os.close(descriptor)
761
+ raise OSError("deployment state component is not a directory")
762
+ self._directory_descriptors.append(descriptor)
763
+ self._directory_bindings.append((parent_fd, name, descriptor, opened.st_dev, opened.st_ino))
764
+ return descriptor
765
+
766
+ def _validate(self) -> None:
767
+ if self._run_handle is None:
768
+ raise HostedDeployError("DEPLOY_STATE_INVALID", "the deployment journal is closed")
769
+ try:
770
+ pipeline._validate_candidate_run_handle(self._run_handle)
771
+ for parent_fd, name, descriptor, device, inode in self._directory_bindings:
772
+ opened = os.fstat(descriptor)
773
+ named = os.stat(name, dir_fd=parent_fd, follow_symlinks=False)
774
+ expected = (device, inode)
775
+ if (
776
+ not stat.S_ISDIR(opened.st_mode)
777
+ or not stat.S_ISDIR(named.st_mode)
778
+ or (opened.st_dev, opened.st_ino) != expected
779
+ or (named.st_dev, named.st_ino) != expected
780
+ ):
781
+ raise OSError("deployment state directory changed")
782
+ except (OSError, pipeline.BuildError) as error:
783
+ raise HostedDeployError(
784
+ "DEPLOY_STATE_INVALID", "the deployment journal path changed during deployment"
785
+ ) from error
786
+
787
+
788
+ def journaled_reviewed_build(run_dir: Path) -> Mapping[str, Any] | None:
789
+ """Return a legacy reviewed-Build binding recorded in a deployment journal.
790
+
791
+ This compatibility reader does not participate in the ordinary Editor deploy path. That path
792
+ always rebuilds and re-verifies the local Build, then confirms its internal Studio approval in
793
+ the same invocation; ``--resume`` only replays an interrupted mutation chain.
794
+
795
+ Raises:
796
+ HostedDeployError: when there is no journal to resume, or it cannot be read safely.
797
+ """
798
+
799
+ store = FileDeploymentStateStore(run_dir)
800
+ try:
801
+ if not store.exists():
802
+ raise HostedDeployError(
803
+ "DEPLOY_STATE_MISSING", f"no deployment journal exists at {store.path}"
804
+ )
805
+ state = store.load()
806
+ finally:
807
+ store.close()
808
+ resources = state.get("resources")
809
+ proposal = resources.get("proposal") if isinstance(resources, dict) else None
810
+ evidence = proposal.get("bootstrap_evidence") if isinstance(proposal, dict) else None
811
+ reviewed = evidence.get("reviewed_build") if isinstance(evidence, dict) else None
812
+ return reviewed if isinstance(reviewed, Mapping) else None
813
+
814
+
815
+ def _legacy_local_build_bootstrap_table_digest(bootstrap: Mapping[str, Any]) -> str:
816
+ """Return the local-Build activation digest or refuse a Studio-owned first build.
817
+
818
+ ``deploy_managed_dataset`` predates hosted-first Recipes. It owns the local Build it has just
819
+ staged, confirmed, and then activates by passing that Build's digest to Studio. A
820
+ ``hosted_first_build`` has no such local Build: Studio/Cloud owns its atomic approval and
821
+ first-run transition. Treating that object as local evidence would either leak a raw
822
+ ``KeyError`` or activate the same Recipe a second time.
823
+
824
+ Omitted ``origin`` remains the historical local shape for backwards compatibility.
825
+ """
826
+
827
+ origin = bootstrap.get("origin", "local_build")
828
+ if origin == "hosted_first_build":
829
+ raise HostedDeployError(
830
+ "DEPLOY_HOSTED_FIRST_STUDIO_OWNED",
831
+ "this Recipe's hosted first build is owned by Studio/Cloud; approve and start it "
832
+ "there instead of using the legacy local-Build deploy command",
833
+ )
834
+ if origin != "local_build":
835
+ raise HostedDeployError(
836
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
837
+ "Studio returned an unknown Recipe bootstrap origin",
838
+ )
839
+ digest = bootstrap.get("bootstrap_table_digest")
840
+ if not isinstance(digest, str) or _DIGEST.fullmatch(digest) is None:
841
+ raise HostedDeployError(
842
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
843
+ "Studio's local-Build Recipe bootstrap omitted a valid bootstrap_table_digest",
844
+ )
845
+ return digest
846
+
847
+
848
+ def deploy_managed_dataset(
849
+ request: DeploymentRequest,
850
+ *,
851
+ run_dir: Path,
852
+ previous_run_dir: Path | None,
853
+ token: StudioToken,
854
+ credentials: ResolvedCloudCredentials,
855
+ resume: bool,
856
+ cron: str = "0 0 * * *",
857
+ max_concurrent_runs: int = 1,
858
+ state_store: DeploymentStateStore | None = None,
859
+ client_factory: Callable[
860
+ [StudioToken], StudioDeploymentClient
861
+ ] = GeneratedStudioDeploymentClient,
862
+ cloud_transport: TokenExchangeTransport | None = None,
863
+ ) -> dict[str, Any]:
864
+ """Prepare and activate one exact Dataset, or resume it after interruption.
865
+
866
+ ``previous_run_dir`` is the run directory whose own deployment published the Dataset this one
867
+ updates, and naming it is what makes this a successor rather than a first deployment. Its
868
+ *absence* is refused here, by :func:`_deployment_predecessor`, and only for the one shape this
869
+ layer can judge alone: a Build whose Recipe is a successor version, which exists because a
870
+ Dataset it follows exists. The opposite rule -- that a Build executed as an ``initial`` run
871
+ must not name a predecessor, because it never read one -- is enforced where the request is
872
+ built, in :func:`deploy.build_deployment_request`, which is the only layer that can see the
873
+ execution mode. So this function trusts its caller for that half; every command in this
874
+ repository builds its request through that function first.
875
+ """
876
+
877
+ _deployment_options(request.dataset_name, cron, max_concurrent_runs)
878
+ if token.workspace_id is None:
879
+ raise HostedDeployError("DEPLOY_TOKEN_EXCHANGE_FAILED", "Studio workspace is missing")
880
+ predecessor = _deployment_predecessor(request, previous_run_dir)
881
+ owns_store = state_store is None
882
+ store = state_store or FileDeploymentStateStore(run_dir)
883
+ cloud_url = _service_url(credentials.cloud_url, "Cloud URL", allow_loopback_http=False)
884
+ identity = canonical.canonical_sha256(
885
+ {
886
+ "request_digest": request.digest(),
887
+ "workspace_id": str(token.workspace_id),
888
+ "cloud_url": cloud_url,
889
+ "cron": cron,
890
+ "max_concurrent_runs": max_concurrent_runs,
891
+ # Which live Dataset this updates is an input like the schedule is, and it is the one
892
+ # a resume can get wrong without changing the request at all. The key is present only
893
+ # when there is a predecessor, so every deployment that could already be in flight --
894
+ # all of them, until this release gave the command a flag to name one -- keeps the
895
+ # exact identity it was staged under.
896
+ **({} if predecessor is None else {"previous_run_dir": predecessor}),
897
+ }
898
+ )
899
+ inputs = _identity_inputs(
900
+ request,
901
+ workspace_id=token.workspace_id,
902
+ cloud_url=cloud_url,
903
+ cron=cron,
904
+ max_concurrent_runs=max_concurrent_runs,
905
+ previous_run_dir=predecessor,
906
+ )
907
+ if store.exists():
908
+ state = store.load()
909
+ _validate_state(
910
+ state,
911
+ identity,
912
+ token.workspace_id,
913
+ inputs=inputs,
914
+ request_digest=request.digest(),
915
+ )
916
+ _refuse_legacy_approval_journal(state)
917
+ if not resume and state.get("status") != "initialized":
918
+ raise HostedDeployError(
919
+ "DEPLOY_STATE_EXISTS",
920
+ f"deployment journal already exists at {store.path}; rerun with --resume",
921
+ )
922
+ else:
923
+ if resume:
924
+ raise HostedDeployError(
925
+ "DEPLOY_STATE_MISSING", f"no deployment journal exists at {store.path}"
926
+ )
927
+ state = {
928
+ "schema_version": DEPLOYMENT_STATE_SCHEMA,
929
+ "deployment_identity": identity,
930
+ "request_digest": request.digest(),
931
+ # The inputs behind that identity, recorded so a refusal hours or days later can name
932
+ # the one that moved. The identity itself is unchanged and stays the only thing any
933
+ # check compares; this is what turns its refusal into a sentence.
934
+ "identity_inputs": inputs,
935
+ "workspace_id": str(token.workspace_id),
936
+ "status": "initialized",
937
+ "schedule": {
938
+ "cron": cron,
939
+ "timezone": "UTC",
940
+ "mode": "incremental_refresh",
941
+ "max_concurrent_runs": max_concurrent_runs,
942
+ },
943
+ "operations": {},
944
+ "resources": {},
945
+ }
946
+ store.save(state)
947
+
948
+ client: StudioDeploymentClient | None = None
949
+ try:
950
+ client = client_factory(token)
951
+ proposal = _prepare_and_decide_approval(
952
+ request,
953
+ state,
954
+ store,
955
+ client,
956
+ run_dir=Path(run_dir),
957
+ previous_run_dir=previous_run_dir,
958
+ workspace_id=token.workspace_id,
959
+ )
960
+ proposal_id = _uuid(proposal["recipe_proposal_id"], "recipe_proposal_id")
961
+ fetched = client.call("get_proposal", None, resource_id=proposal_id)
962
+ fetched_body = _success(fetched, {200}, "DEPLOY_PROPOSAL_READ_FAILED")
963
+ _same(fetched_body, "recipe_digest", request.recipe_digest)
964
+ activation_intent = {
965
+ "table_name": request.dataset_name,
966
+ "schedule": state["schedule"],
967
+ }
968
+ _same(fetched_body, "activation_intent", activation_intent)
969
+ if fetched_body.get("status") not in {"approved", "activation_started"}:
970
+ raise HostedDeployError(
971
+ "DEPLOY_APPROVAL_DECISION_INVALID",
972
+ "Studio has not recorded this authenticated Editor's approval for the exact "
973
+ "Recipe proposal",
974
+ )
975
+ etag = _etag(fetched, "proposal read")
976
+ approval_request_id = _uuid_field(fetched_body, "approval_request_id")
977
+ approval_decision_id = _uuid_field(fetched_body, "approval_decision_id")
978
+ bootstrap = fetched_body.get("bootstrap_evidence")
979
+ if not isinstance(bootstrap, dict):
980
+ raise HostedDeployError(
981
+ "DEPLOY_STUDIO_RESPONSE_INVALID", "the approved proposal omitted Build evidence"
982
+ )
983
+ bootstrap_table_digest = _legacy_local_build_bootstrap_table_digest(bootstrap)
984
+ activation_body = {
985
+ "schema_version": STUDIO_SCHEMA_VERSION,
986
+ "workspace_id": str(token.workspace_id),
987
+ "recipe_proposal_id": str(proposal_id),
988
+ "table_recipe_id": request.recipe_id,
989
+ "recipe_version": request.recipe_version,
990
+ "recipe_digest": request.recipe_digest,
991
+ "source_authority_bindings": fetched_body["source_authority_bindings"],
992
+ "source_authority_bindings_digest": fetched_body["source_authority_bindings_digest"],
993
+ "approval_request_id": str(approval_request_id),
994
+ "approval_decision_id": str(approval_decision_id),
995
+ "bootstrap_table_digest": bootstrap_table_digest,
996
+ "table_name": activation_intent["table_name"],
997
+ "schedule": activation_intent["schedule"],
998
+ }
999
+ activation = _mutate(
1000
+ state,
1001
+ store,
1002
+ client,
1003
+ name="activation",
1004
+ operation="activate_recipe",
1005
+ body=activation_body,
1006
+ expected={202},
1007
+ resource_id=proposal_id,
1008
+ if_match=etag,
1009
+ )
1010
+ state["resources"]["activation"] = dict(activation)
1011
+ state["status"] = "activation_queued"
1012
+ store.save(state)
1013
+ binding = _bind_cloud_dataset(
1014
+ request,
1015
+ state,
1016
+ store,
1017
+ credentials=credentials,
1018
+ transport=cloud_transport,
1019
+ )
1020
+ state["resources"]["cloud_binding"] = dict(binding)
1021
+ state["status"] = "cloud_dataset_bound"
1022
+ store.save(state)
1023
+ return _activation_receipt(state, store, cloud_url)
1024
+ finally:
1025
+ if client is not None:
1026
+ client.close()
1027
+ if owns_store:
1028
+ store.close()
1029
+
1030
+
1031
+ def _worker_policy_preflight(
1032
+ request: DeploymentRequest,
1033
+ recipe_document: Mapping[str, Any],
1034
+ proposal: Mapping[str, Any],
1035
+ client: StudioDeploymentClient,
1036
+ ) -> dict[str, Any]:
1037
+ """Require Studio's selected fleet to match this runtime before approval."""
1038
+
1039
+ command_fields = (
1040
+ "workspace_id",
1041
+ "dataset_id",
1042
+ "table_plan_id",
1043
+ "table_recipe_id",
1044
+ "recipe_version",
1045
+ "recipe_digest",
1046
+ "recipe_proposal_id",
1047
+ )
1048
+ try:
1049
+ command = {field: proposal[field] for field in command_fields}
1050
+ except KeyError as error:
1051
+ raise HostedDeployError(
1052
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1053
+ "Studio's recipe proposal omitted worker-policy preflight scope",
1054
+ ) from error
1055
+ if "table_id" in proposal:
1056
+ command["table_id"] = proposal["table_id"]
1057
+ response = client.preflight(command)
1058
+ preflight = dict(_success(response, {200}, "DEPLOY_WORKER_POLICY_PREFLIGHT_FAILED"))
1059
+
1060
+ required = {
1061
+ "schema_version",
1062
+ "workspace_id",
1063
+ "dataset_id",
1064
+ "table_plan_id",
1065
+ "table_recipe_id",
1066
+ "recipe_version",
1067
+ "recipe_digest",
1068
+ "builder_image_digest",
1069
+ "verifier_image_digest",
1070
+ "validation_policy_digest",
1071
+ "selection_fence",
1072
+ "selection_id",
1073
+ "policy_mode",
1074
+ "validation_policy_digests",
1075
+ "policy_set_digest",
1076
+ }
1077
+ expected_fields = required | ({"table_id"} if "table_id" in command else set())
1078
+ if (
1079
+ set(preflight) != expected_fields
1080
+ or preflight.get("schema_version") != STUDIO_SCHEMA_VERSION
1081
+ ):
1082
+ raise HostedDeployError(
1083
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1084
+ "Studio returned a non-exact worker-policy preflight",
1085
+ )
1086
+ for field in (
1087
+ "workspace_id",
1088
+ "dataset_id",
1089
+ "table_plan_id",
1090
+ "table_id",
1091
+ "table_recipe_id",
1092
+ "recipe_version",
1093
+ "recipe_digest",
1094
+ ):
1095
+ if field in command and preflight.get(field) != command[field]:
1096
+ raise HostedDeployError(
1097
+ "DEPLOY_WORKER_POLICY_MISMATCH",
1098
+ f"Studio's worker-policy preflight {field} differs from the frozen Recipe proposal",
1099
+ )
1100
+
1101
+ transform_plan = recipe_document.get("transform_plan")
1102
+ plan_schema = (
1103
+ transform_plan.get("schema_version") if isinstance(transform_plan, Mapping) else None
1104
+ )
1105
+ modes = {
1106
+ "local-table-plan.v1": "table",
1107
+ "local-graph-table-plan.v1": "graph",
1108
+ }
1109
+ policy_mode = modes.get(plan_schema)
1110
+ if policy_mode is None:
1111
+ raise HostedDeployError(
1112
+ "DEPLOY_REQUEST_INVALID",
1113
+ "the frozen Recipe has no supported table or graph transform plan",
1114
+ )
1115
+ local_policies = {
1116
+ "table": "sha256:" + pipeline.validation_policy_digest(graph=False),
1117
+ "graph": "sha256:" + pipeline.validation_policy_digest(graph=True),
1118
+ }
1119
+ local_policy_set_digest = "sha256:" + canonical.canonical_sha256(local_policies)
1120
+ selected_policies = preflight.get("validation_policy_digests")
1121
+ selected_policy = preflight.get("validation_policy_digest")
1122
+ if (
1123
+ selected_policies != local_policies
1124
+ or preflight.get("policy_set_digest") != local_policy_set_digest
1125
+ or preflight.get("policy_mode") != policy_mode
1126
+ or selected_policy != local_policies[policy_mode]
1127
+ ):
1128
+ raise HostedDeployError(
1129
+ "DEPLOY_WORKER_POLICY_MISMATCH",
1130
+ "your CLI and the hosted workers disagree about the validation policy: "
1131
+ f"Studio selected {selected_policy} for {preflight.get('policy_mode')}, while this "
1132
+ f"Recipe runtime computes {local_policies[policy_mode]} for {policy_mode}; re-export "
1133
+ "with 'mr-data recipe-export', or wait for the hosted fleet to be republished",
1134
+ )
1135
+
1136
+ builder_image = preflight.get("builder_image_digest")
1137
+ verifier_image = preflight.get("verifier_image_digest")
1138
+ selection_fence = preflight.get("selection_fence")
1139
+ selection_id = preflight.get("selection_id")
1140
+ if (
1141
+ not isinstance(builder_image, str)
1142
+ or _IMMUTABLE_IMAGE.fullmatch(builder_image) is None
1143
+ or not isinstance(verifier_image, str)
1144
+ or _IMMUTABLE_IMAGE.fullmatch(verifier_image) is None
1145
+ or not isinstance(selection_fence, str)
1146
+ or _PREFIXED_DIGEST.fullmatch(selection_fence) is None
1147
+ or not isinstance(selection_id, str)
1148
+ or _DIGEST.fullmatch(selection_id) is None
1149
+ ):
1150
+ raise HostedDeployError(
1151
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1152
+ "Studio returned an invalid worker-policy selection identity",
1153
+ )
1154
+ selection = {
1155
+ "schema_version": "mostlyright-worker-policy-binding.v2",
1156
+ "producer_image_digest": builder_image,
1157
+ "verifier_image_digest": verifier_image,
1158
+ "validation_policy_digests": local_policies,
1159
+ "policy_set_digest": local_policy_set_digest,
1160
+ "release_revision_coordinate": selection_fence,
1161
+ }
1162
+ if canonical.canonical_sha256(selection) != selection_id:
1163
+ raise HostedDeployError(
1164
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1165
+ "Studio's worker-policy selection ID does not bind its exact images, policy set, and "
1166
+ "release fence",
1167
+ )
1168
+ if request.recipe_digest != preflight["recipe_digest"]:
1169
+ raise HostedDeployError(
1170
+ "DEPLOY_WORKER_POLICY_MISMATCH",
1171
+ "Studio's worker-policy preflight does not bind the exact local Recipe",
1172
+ )
1173
+ return preflight
1174
+
1175
+
1176
+ def _prepare_and_decide_approval(
1177
+ request: DeploymentRequest,
1178
+ state: dict[str, Any],
1179
+ store: DeploymentStateStore,
1180
+ client: StudioDeploymentClient,
1181
+ *,
1182
+ run_dir: Path,
1183
+ previous_run_dir: Path | None,
1184
+ workspace_id: UUID,
1185
+ ) -> Mapping[str, Any]:
1186
+ recipe_document = canonical.parse_canonical_json(request.canonical_recipe_json.encode("utf-8"))
1187
+ if not isinstance(recipe_document, dict):
1188
+ raise HostedDeployError("DEPLOY_REQUEST_INVALID", "Recipe must be one object")
1189
+ resources = state["resources"]
1190
+ # Whether this deployment founds a Dataset or updates one is decided by whether it named a
1191
+ # predecessor, which is the same question `recipe.verify_recipe_candidate` asks -- a Recipe
1192
+ # execution names its predecessor exactly when its mode is not `initial`, and `deploy.py`
1193
+ # refuses both shapes where the two would disagree. Branching on `recipe_version` instead
1194
+ # looked equivalent and is not: a refresh of an unchanged Recipe is a successor whose Recipe
1195
+ # version has not moved, and it would arrive here to have its project, question, plan and
1196
+ # Dataset created a second time.
1197
+ if previous_run_dir is None:
1198
+ project = _mutate(
1199
+ state,
1200
+ store,
1201
+ client,
1202
+ name="project",
1203
+ operation="create_dataset",
1204
+ body={
1205
+ "workspace_id": str(workspace_id),
1206
+ "name": request.dataset_name,
1207
+ "description": "Managed from an exact reviewed local Recipe Build.",
1208
+ },
1209
+ expected={201},
1210
+ )
1211
+ dataset_id = _uuid_field(project, "dataset_id")
1212
+ question = _mutate(
1213
+ state,
1214
+ store,
1215
+ client,
1216
+ name="question",
1217
+ operation="create_question",
1218
+ body={
1219
+ "workspace_id": str(workspace_id),
1220
+ "dataset_id": str(dataset_id),
1221
+ "question": recipe_document["question"]["text"],
1222
+ },
1223
+ expected={201},
1224
+ )
1225
+ question_id = _uuid_field(question, "question_id")
1226
+ requirements_document = recipe_document["requirements"]
1227
+ requirements = _mutate(
1228
+ state,
1229
+ store,
1230
+ client,
1231
+ name="requirements",
1232
+ operation="put_requirements",
1233
+ body={
1234
+ "schema_version": STUDIO_SCHEMA_VERSION,
1235
+ "workspace_id": str(workspace_id),
1236
+ "population": requirements_document["population"],
1237
+ "time_range": requirements_document["time_range"],
1238
+ "output_grain": requirements_document["output_grain"]["columns"],
1239
+ "required_fields": requirements_document["required_fields"],
1240
+ "target_policy": requirements_document["target_policy"],
1241
+ "success_criteria": requirements_document["success_criteria"],
1242
+ "feasibility": _studio_feasibility(requirements_document["feasibility"]),
1243
+ **(
1244
+ {"prediction_cutoff": requirements_document["prediction_cutoff"]}
1245
+ if "prediction_cutoff" in requirements_document
1246
+ else {}
1247
+ ),
1248
+ },
1249
+ expected={200, 201},
1250
+ resource_id=question_id,
1251
+ # The requirements upsert is preconditioned on the REQUIREMENTS
1252
+ # subresource's own version, not the question's: Studio's
1253
+ # put_requirements allows the initial upsert only as If-Match: "*"
1254
+ # (allow_initial), and the question ETag can never satisfy it.
1255
+ if_match="*",
1256
+ )
1257
+ requirements_id = _uuid_field(requirements, "requirements_id")
1258
+ source_registry_ids = _register_sources(
1259
+ request,
1260
+ recipe_document,
1261
+ state,
1262
+ store,
1263
+ client,
1264
+ run_dir=run_dir,
1265
+ previous_run_dir=previous_run_dir,
1266
+ dataset_id=dataset_id,
1267
+ workspace_id=workspace_id,
1268
+ )
1269
+ policy = recipe_document.get("backfill_policy") or _default_backfill_policy(
1270
+ requirements_document["time_range"]
1271
+ )
1272
+ plan = _mutate(
1273
+ state,
1274
+ store,
1275
+ client,
1276
+ name="plan",
1277
+ operation="create_plan",
1278
+ body={
1279
+ "schema_version": STUDIO_SCHEMA_VERSION,
1280
+ "workspace_id": str(workspace_id),
1281
+ "dataset_id": str(dataset_id),
1282
+ "question_id": str(question_id),
1283
+ "requirements_id": str(requirements_id),
1284
+ "source_ids": [str(value) for value in source_registry_ids],
1285
+ "output_grain": requirements_document["output_grain"]["columns"],
1286
+ "transformation_contract_version": "1.0.0",
1287
+ "execution_plan_digest": "sha256:" + request.execution_plan_digest,
1288
+ "validation_policy_digest": "sha256:"
1289
+ + request.bootstrap_evidence["bootstrap_verification_digest"],
1290
+ "approved_backfill_policy": policy,
1291
+ "backfill_policy_digest": "sha256:" + canonical.canonical_sha256(policy),
1292
+ },
1293
+ expected={201},
1294
+ )
1295
+ table_plan_id = _uuid_field(plan, "table_plan_id")
1296
+ else:
1297
+ previous = _previous_deployment_resources(previous_run_dir)
1298
+ dataset_id = _uuid(previous["dataset_id"], "dataset_id")
1299
+ table_plan_id = _uuid(previous["table_plan_id"], "table_plan_id")
1300
+ resources["previous_dataset_id"] = previous["dataset_id"]
1301
+ store.save(state)
1302
+
1303
+ authorities = _source_authorities(
1304
+ request,
1305
+ recipe_document,
1306
+ state,
1307
+ store,
1308
+ client,
1309
+ run_dir=run_dir,
1310
+ previous_run_dir=previous_run_dir,
1311
+ workspace_id=workspace_id,
1312
+ )
1313
+ inventory_digest = canonical.canonical_sha256(
1314
+ [
1315
+ {
1316
+ "source_id": item["source_id"],
1317
+ "source_authority_digest": item["source_authority_digest"],
1318
+ }
1319
+ for item in authorities
1320
+ ]
1321
+ )
1322
+ proposal_body = {
1323
+ "schema_version": STUDIO_SCHEMA_VERSION,
1324
+ "workspace_id": str(workspace_id),
1325
+ "dataset_id": str(dataset_id),
1326
+ "table_plan_id": str(table_plan_id),
1327
+ "recipe_media_type": "application/vnd.mostlyright.frozen-recipe+json",
1328
+ "recipe_schema_version": request.recipe_schema_version,
1329
+ "table_recipe_id": request.recipe_id,
1330
+ "recipe_version": request.recipe_version,
1331
+ "predecessor_recipe_digest": recipe_document.get("predecessor_recipe_digest"),
1332
+ "canonical_recipe_json": request.canonical_recipe_json,
1333
+ "recipe_digest": request.recipe_digest,
1334
+ "source_authority_bindings": authorities,
1335
+ "source_authority_bindings_digest": inventory_digest,
1336
+ "bootstrap_evidence": request.bootstrap_evidence,
1337
+ "activation_intent": {
1338
+ "table_name": request.dataset_name,
1339
+ "schedule": state["schedule"],
1340
+ },
1341
+ **({} if previous_run_dir is None else {"dataset_id": resources["previous_dataset_id"]}),
1342
+ }
1343
+ proposal = _mutate(
1344
+ state,
1345
+ store,
1346
+ client,
1347
+ name="proposal",
1348
+ operation="create_proposal",
1349
+ body=proposal_body,
1350
+ expected={201},
1351
+ # The approval request below conditions on this ETag.
1352
+ requires_etag=True,
1353
+ )
1354
+ proposal_id = _uuid_field(proposal, "recipe_proposal_id")
1355
+ _same(proposal, "recipe_digest", request.recipe_digest)
1356
+ activation_operation = state["operations"].get("activation")
1357
+ if (
1358
+ state.get("status") == "cloud_dataset_bound"
1359
+ and isinstance(resources.get("activation"), dict)
1360
+ and isinstance(activation_operation, dict)
1361
+ and activation_operation.get("status") == "completed"
1362
+ ):
1363
+ return proposal
1364
+ worker_policy_preflight = _worker_policy_preflight(
1365
+ request,
1366
+ recipe_document,
1367
+ proposal,
1368
+ client,
1369
+ )
1370
+ proposal_etag = _resource_etag(state, "proposal")
1371
+ approval = _mutate(
1372
+ state,
1373
+ store,
1374
+ client,
1375
+ name="approval_request",
1376
+ operation="request_approval",
1377
+ body={
1378
+ "schema_version": STUDIO_SCHEMA_VERSION,
1379
+ "workspace_id": str(workspace_id),
1380
+ "recipe_proposal_id": str(proposal_id),
1381
+ "recipe_digest": request.recipe_digest,
1382
+ "worker_policy_preflight": worker_policy_preflight,
1383
+ },
1384
+ expected={201},
1385
+ resource_id=proposal_id,
1386
+ if_match=proposal_etag,
1387
+ # The confirmation below conditions on this ETag, and reads it from the journal.
1388
+ requires_etag=True,
1389
+ )
1390
+ approval_request_id = _uuid_field(approval, "approval_request_id")
1391
+ _same(approval, "status", "pending")
1392
+ _same(approval, "subject_digest", request.recipe_digest)
1393
+ approval_version = _positive_int_field(approval, "version")
1394
+ approval_etag = _resource_etag(state, "approval_request")
1395
+ decision = _mutate(
1396
+ state,
1397
+ store,
1398
+ client,
1399
+ name="approval_confirmation",
1400
+ operation="confirm_table_recipe_approval",
1401
+ body={
1402
+ "schema_version": STUDIO_SCHEMA_VERSION,
1403
+ "workspace_id": str(workspace_id),
1404
+ "decision": "approved",
1405
+ "expected_request_version": approval_version,
1406
+ "subject_digest": request.recipe_digest,
1407
+ },
1408
+ expected={200},
1409
+ resource_id=approval_request_id,
1410
+ if_match=approval_etag,
1411
+ )
1412
+ _same(decision, "approval_request_id", str(approval_request_id))
1413
+ _same(decision, "decision", "approved")
1414
+ _same(decision, "subject_digest", request.recipe_digest)
1415
+ resources["approval_request_id"] = str(approval_request_id)
1416
+ resources["approval_decision_id"] = str(_uuid_field(decision, "approval_decision_id"))
1417
+ resources["dataset_id"] = str(dataset_id)
1418
+ resources["table_plan_id"] = str(table_plan_id)
1419
+ state["status"] = "approval_decided"
1420
+ store.save(state)
1421
+ return proposal
1422
+
1423
+
1424
+ def _acquisition_bundle(
1425
+ request: DeploymentRequest,
1426
+ *,
1427
+ run_dir: Path,
1428
+ previous_run_dir: Path | None,
1429
+ ) -> Mapping[str, Any]:
1430
+ """Return the closed acquisition-evidence bundle this request names, from either home.
1431
+
1432
+ A request that carries a reviewed-Build binding carries the bundle document inside it, and that
1433
+ copy is used unchanged -- it is the exact document the binding's signature covers, so reading
1434
+ it from anywhere else would check a different object than the one that was signed.
1435
+
1436
+ A request built for a dataset the caller publishes itself carries no binding, so the bundle is
1437
+ read back out of the run directory under ``request.acquisition_bundle_digest``, which
1438
+ :func:`deployment_evidence.load_acquisition_bundle_document` authenticates to this exact Build
1439
+ before returning it. That digest is a field of the request, so it is inside the request digest,
1440
+ so it is inside the deployment identity the journal is held to. This is the same route
1441
+ ``_source_authorities`` already takes for the source bytes themselves.
1442
+ """
1443
+
1444
+ reviewed = request.bootstrap_evidence.get("reviewed_build")
1445
+ if isinstance(reviewed, Mapping):
1446
+ bundle = reviewed["acquisition_bundle"]
1447
+ if not isinstance(bundle, Mapping):
1448
+ raise HostedDeployError(
1449
+ "DEPLOY_REQUEST_INVALID", "the reviewed Build carries no acquisition bundle"
1450
+ )
1451
+ return bundle
1452
+ try:
1453
+ return deployment_evidence.load_acquisition_bundle_document(
1454
+ run_dir,
1455
+ request.acquisition_bundle_digest,
1456
+ previous_run_dir=previous_run_dir,
1457
+ )
1458
+ except deployment_evidence.DeploymentEvidenceError as error:
1459
+ raise HostedDeployError(
1460
+ "DEPLOY_REQUEST_INVALID",
1461
+ f"the acquisition evidence for {run_dir} could not be read [{error.code}]",
1462
+ ) from error
1463
+
1464
+
1465
+ def _register_sources(
1466
+ request: DeploymentRequest,
1467
+ recipe_document: Mapping[str, Any],
1468
+ state: dict[str, Any],
1469
+ store: DeploymentStateStore,
1470
+ client: StudioDeploymentClient,
1471
+ *,
1472
+ run_dir: Path,
1473
+ previous_run_dir: Path | None,
1474
+ dataset_id: UUID,
1475
+ workspace_id: UUID,
1476
+ ) -> list[UUID]:
1477
+ proposals = {item["source_id"]: item for item in recipe_document["source_proposals"]}
1478
+ recipe_sources = {item["source_id"]: item for item in recipe_document["sources"]}
1479
+ bundle_sources = {
1480
+ item["source_id"]: item
1481
+ for item in _acquisition_bundle(
1482
+ request, run_dir=run_dir, previous_run_dir=previous_run_dir
1483
+ )["sources"]
1484
+ }
1485
+ result: list[UUID] = []
1486
+ for source_id in sorted(recipe_sources):
1487
+ proposal = proposals[source_id]
1488
+ source = recipe_sources[source_id]
1489
+ bundle = bundle_sources[source_id]
1490
+ locator = _source_locator(source, proposal, bundle)
1491
+ response = _mutate(
1492
+ state,
1493
+ store,
1494
+ client,
1495
+ name=f"source_registry:{source_id}",
1496
+ operation="register_source",
1497
+ body={
1498
+ "schema_version": STUDIO_SCHEMA_VERSION,
1499
+ "workspace_id": str(workspace_id),
1500
+ "dataset_id": str(dataset_id),
1501
+ "name": proposal["display_name"],
1502
+ "source_class": proposal["source_class"],
1503
+ "locator": locator,
1504
+ "credential_reference_ids": [],
1505
+ "data_classification": source["declared_classification"],
1506
+ "rights_claim": {
1507
+ "claimed_basis": (
1508
+ "permission_asserted"
1509
+ if proposal.get("rights_status") == "approved"
1510
+ else "unknown"
1511
+ ),
1512
+ "claim_evidence_digest": "sha256:"
1513
+ + canonical.canonical_sha256(proposal["evidence"]),
1514
+ "claim_note": source["lawful_basis"],
1515
+ },
1516
+ "retention_policy": {
1517
+ "raw_days": 30,
1518
+ "derived_days": 365,
1519
+ "tombstone_required": False,
1520
+ },
1521
+ },
1522
+ expected={201},
1523
+ )
1524
+ result.append(_uuid_field(response, "source_id"))
1525
+ return result
1526
+
1527
+
1528
+ def _source_authorities(
1529
+ request: DeploymentRequest,
1530
+ recipe_document: Mapping[str, Any],
1531
+ state: dict[str, Any],
1532
+ store: DeploymentStateStore,
1533
+ client: StudioDeploymentClient,
1534
+ *,
1535
+ run_dir: Path,
1536
+ previous_run_dir: Path | None,
1537
+ workspace_id: UUID,
1538
+ ) -> list[dict[str, Any]]:
1539
+ bundle = _acquisition_bundle(request, run_dir=run_dir, previous_run_dir=previous_run_dir)
1540
+ recipe_sources = {item["source_id"]: item for item in recipe_document["sources"]}
1541
+ resources = state["resources"]
1542
+ result: list[dict[str, Any]] = []
1543
+ for entry in bundle["sources"]:
1544
+ source_id = entry["source_id"]
1545
+ if entry["authority_kind"] == "connector":
1546
+ source = recipe_sources[source_id]
1547
+ if source["adapter_id"] not in {"public.https", "external.openligadb"}:
1548
+ raise HostedDeployError(
1549
+ "DEPLOY_CONNECTOR_UNSUPPORTED",
1550
+ f"{source_id} does not use a deployed refresh connector",
1551
+ )
1552
+ egress_attestation = (
1553
+ PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION
1554
+ if source["adapter_id"] == "public.https"
1555
+ else OPENLIGADB_EGRESS_POLICY_ATTESTATION
1556
+ )
1557
+ body = {
1558
+ "schema_version": STUDIO_SCHEMA_VERSION,
1559
+ "workspace_id": str(workspace_id),
1560
+ "source_id": source_id,
1561
+ "adapter_id": source["adapter_id"],
1562
+ "credential_mode": "none",
1563
+ "crawler_egress_policy_attestation": egress_attestation,
1564
+ **(
1565
+ {"query": source["query_template"]}
1566
+ if source["adapter_id"] == "public.https"
1567
+ else {}
1568
+ ),
1569
+ }
1570
+ configuration = _mutate(
1571
+ state,
1572
+ store,
1573
+ client,
1574
+ name=f"connector:{source_id}",
1575
+ operation="register_connector",
1576
+ body=body,
1577
+ expected={201},
1578
+ )
1579
+ authority = {
1580
+ "source_id": source_id,
1581
+ "authority_kind": "connector",
1582
+ "source_authority_digest": "",
1583
+ "adapter_id": source["adapter_id"],
1584
+ "connector_configuration_id": configuration["connector_configuration_id"],
1585
+ "connector_configuration_digest": configuration["configuration_digest"],
1586
+ "credential_mode": "none",
1587
+ "crawler_egress_policy_attestation": egress_attestation,
1588
+ }
1589
+ authority["source_authority_digest"] = canonical.canonical_sha256(
1590
+ {key: value for key, value in authority.items() if key != "source_authority_digest"}
1591
+ )
1592
+ result.append(authority)
1593
+ continue
1594
+ payload = deployment_evidence.load_acquisition_source_payload(
1595
+ run_dir,
1596
+ request.acquisition_bundle_digest,
1597
+ source_id,
1598
+ previous_run_dir=previous_run_dir,
1599
+ )
1600
+ descriptors = {
1601
+ role: {
1602
+ "media_type": reference["media_type"],
1603
+ "expected_content_digest": "sha256:" + reference["content_sha256"],
1604
+ "expected_size_bytes": reference["size_bytes"],
1605
+ }
1606
+ for role, reference in (
1607
+ ("source_data", entry["source_data"]),
1608
+ ("acquisition_receipt", entry["acquisition_receipt"]),
1609
+ ("source_observation", entry["source_observation"]),
1610
+ )
1611
+ }
1612
+ content_by_role = {
1613
+ "source_data": payload.source_data,
1614
+ "acquisition_receipt": payload.acquisition_receipt,
1615
+ "source_observation": payload.source_observation,
1616
+ }
1617
+ attempts = state.setdefault("source_staging_attempts", {})
1618
+ attempt = attempts.get(source_id, 0)
1619
+ if isinstance(attempt, bool) or not isinstance(attempt, int) or attempt < 0:
1620
+ raise HostedDeployError(
1621
+ "DEPLOY_STATE_INVALID", f"source staging attempt for {source_id} is invalid"
1622
+ )
1623
+ renewed = False
1624
+ while True:
1625
+ suffix = "" if attempt == 0 else f":renew:{attempt}"
1626
+ staging_id = _source_staging_id(request.digest(), workspace_id, source_id, attempt)
1627
+ staging_command = {
1628
+ "schema_version": STUDIO_SCHEMA_VERSION,
1629
+ "source_staging_id": str(staging_id),
1630
+ "workspace_id": str(workspace_id),
1631
+ "source_id": source_id,
1632
+ **descriptors,
1633
+ }
1634
+ staging = _mutate(
1635
+ state,
1636
+ store,
1637
+ client,
1638
+ name=f"source_staging:{source_id}{suffix}",
1639
+ operation="create_source_staging",
1640
+ body=staging_command,
1641
+ expected={201},
1642
+ )
1643
+ artifacts = _validate_source_staging(
1644
+ staging,
1645
+ command=staging_command,
1646
+ staging_id=staging_id,
1647
+ workspace_id=workspace_id,
1648
+ source_id=source_id,
1649
+ )
1650
+ if staging.get("status") == "finalized":
1651
+ break
1652
+ # A staging Studio has already sealed is replayed from its journaled finalize
1653
+ # response instead of being walked again.
1654
+ #
1655
+ # ⚠ WHY THIS EXISTS. The create response journals signed upload sessions that live for
1656
+ # minutes, and the whole point of a resume is that it arrives hours or days later,
1657
+ # after a human has approved in Cloud. Walking the uploads again therefore always found
1658
+ # those sessions expired, took the renewal path, and minted a SECOND staging: every
1659
+ # source byte uploaded to Studio a second time, new artifact ids, and new source
1660
+ # authorities -- which are inputs to the Recipe proposal the human already approved.
1661
+ # The proposal short-circuit hid it, so the rebuilt bindings were discarded in silence
1662
+ # and the resume looked fine. Reading the journaled request digest is what made it
1663
+ # visible: the rebuilt proposal no longer matched the one Studio holds, because this
1664
+ # invocation had just moved it.
1665
+ #
1666
+ # The sealed objects, their artifact ids and the source authority all come back in the
1667
+ # finalize response, which is journaled. Replaying it keeps the authorities -- and the
1668
+ # proposal built from them -- exactly what the approval was given against, and stops a
1669
+ # resume re-uploading a Build that is already staged.
1670
+ if isinstance(resources.get(f"source_finalize:{source_id}{suffix}"), dict):
1671
+ staging = _mutate(
1672
+ state,
1673
+ store,
1674
+ client,
1675
+ name=f"source_finalize:{source_id}{suffix}",
1676
+ operation="finalize_source_staging",
1677
+ body={
1678
+ "schema_version": STUDIO_SCHEMA_VERSION,
1679
+ "workspace_id": str(workspace_id),
1680
+ },
1681
+ expected={200},
1682
+ resource_id=staging_id,
1683
+ )
1684
+ if staging.get("status") != "finalized":
1685
+ raise HostedDeployError(
1686
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1687
+ f"the journaled finalize for {source_id} does not report a sealed staging",
1688
+ )
1689
+ artifacts = _validate_source_staging(
1690
+ staging,
1691
+ command=staging_command,
1692
+ staging_id=staging_id,
1693
+ workspace_id=workspace_id,
1694
+ source_id=source_id,
1695
+ )
1696
+ # The journaled finalize is what reconciled any transfer whose response was lost,
1697
+ # so a replay records that here exactly as the live finalize does. Skipping it
1698
+ # would leave the journal describing a transfer Studio has sealed as still planned.
1699
+ _reconcile_lost_uploads(state, store, source_id)
1700
+ break
1701
+ try:
1702
+ for role, artifact in artifacts.items():
1703
+ session = artifact["upload_session"]
1704
+ _validate_source_upload_session(
1705
+ session,
1706
+ role=role,
1707
+ artifact_id=_uuid(artifact["artifact_id"], "artifact_id"),
1708
+ staging_id=staging_id,
1709
+ workspace_id=workspace_id,
1710
+ descriptor=descriptors[role],
1711
+ )
1712
+ generation = _upload_once(
1713
+ state,
1714
+ store,
1715
+ client,
1716
+ name=f"source_upload:{source_id}:{role}{suffix}",
1717
+ session=session,
1718
+ content=content_by_role[role],
1719
+ )
1720
+ if generation is not None:
1721
+ _mutate(
1722
+ state,
1723
+ store,
1724
+ client,
1725
+ name=f"source_complete:{source_id}:{role}{suffix}",
1726
+ operation="complete_artifact_upload",
1727
+ body={
1728
+ "schema_version": STUDIO_SCHEMA_VERSION,
1729
+ "workspace_id": str(workspace_id),
1730
+ "reservation_id": session["reservation_id"],
1731
+ "reservation_digest": session["reservation_digest"],
1732
+ "upload_session_id": session["session_id"],
1733
+ "object_generation": generation,
1734
+ "observed_size_bytes": len(content_by_role[role]),
1735
+ "observed_content_digest": "sha256:"
1736
+ + hashlib.sha256(content_by_role[role]).hexdigest(),
1737
+ },
1738
+ expected={201},
1739
+ resource_id=_uuid(artifact["artifact_id"], "artifact_id"),
1740
+ )
1741
+ except HostedDeployError as error:
1742
+ if error.code != "DEPLOY_SOURCE_UPLOAD_SESSION_REJECTED" or renewed:
1743
+ raise
1744
+ attempt += 1
1745
+ attempts[source_id] = attempt
1746
+ store.save(state)
1747
+ renewed = True
1748
+ continue
1749
+ staging = _mutate(
1750
+ state,
1751
+ store,
1752
+ client,
1753
+ name=f"source_finalize:{source_id}{suffix}",
1754
+ operation="finalize_source_staging",
1755
+ body={"schema_version": STUDIO_SCHEMA_VERSION, "workspace_id": str(workspace_id)},
1756
+ expected={200},
1757
+ resource_id=staging_id,
1758
+ )
1759
+ artifacts = _validate_source_staging(
1760
+ staging,
1761
+ command=staging_command,
1762
+ staging_id=staging_id,
1763
+ workspace_id=workspace_id,
1764
+ source_id=source_id,
1765
+ )
1766
+ _reconcile_lost_uploads(state, store, source_id)
1767
+ break
1768
+ authority = _validate_source_authority(
1769
+ staging.get("authority"),
1770
+ source_id=source_id,
1771
+ workspace_id=workspace_id,
1772
+ artifacts=artifacts,
1773
+ descriptors=descriptors,
1774
+ )
1775
+ result.append(authority)
1776
+ if [item["source_id"] for item in result] != sorted(recipe_sources):
1777
+ raise HostedDeployError(
1778
+ "DEPLOY_SOURCE_INVENTORY_MISMATCH", "Studio source authority inventory is not exact"
1779
+ )
1780
+ return result
1781
+
1782
+
1783
+ def _reconcile_lost_uploads(
1784
+ state: dict[str, Any], store: DeploymentStateStore, source_id: str
1785
+ ) -> None:
1786
+ """Retire this source's still-planned transfers once Studio has sealed the staging.
1787
+
1788
+ A transfer can commit remotely while its response is lost, and :func:`_upload_once` leaves that
1789
+ operation ``planned`` on purpose. The finalize response is what settles it: Studio names the
1790
+ exact sealed object, so the plan is neither outstanding nor a failure. Recording it is not
1791
+ bookkeeping for its own sake -- the journal is the evidence of what this deployment did.
1792
+ """
1793
+
1794
+ for name, operation in list(state["operations"].items()):
1795
+ if (
1796
+ name.startswith(f"source_upload:{source_id}:")
1797
+ and isinstance(operation, dict)
1798
+ and operation.get("status") == "planned"
1799
+ ):
1800
+ state["operations"][name] = {**operation, "status": "reconciled"}
1801
+ store.save(state)
1802
+
1803
+
1804
+ def _source_staging_id(
1805
+ request_digest: str,
1806
+ workspace_id: UUID,
1807
+ source_id: str,
1808
+ attempt: int,
1809
+ ) -> UUID:
1810
+ return uuid5(
1811
+ NAMESPACE_URL,
1812
+ f"mostlyright:source-staging:{request_digest}:{workspace_id}:{source_id}:attempt:{attempt}",
1813
+ )
1814
+
1815
+
1816
+ _SOURCE_STAGING_ROLES = {
1817
+ "source_data": ("raw_snapshot", "source_staging_source_data"),
1818
+ "acquisition_receipt": (
1819
+ "candidate_evidence",
1820
+ "source_staging_acquisition_receipt",
1821
+ ),
1822
+ "source_observation": (
1823
+ "candidate_evidence",
1824
+ "source_staging_source_observation",
1825
+ ),
1826
+ }
1827
+
1828
+
1829
+ def _validate_source_staging(
1830
+ staging: Mapping[str, Any],
1831
+ *,
1832
+ command: Mapping[str, Any],
1833
+ staging_id: UUID,
1834
+ workspace_id: UUID,
1835
+ source_id: str,
1836
+ ) -> dict[str, Mapping[str, Any]]:
1837
+ if (
1838
+ staging.get("schema_version") != STUDIO_SCHEMA_VERSION
1839
+ or staging.get("source_staging_id") != str(staging_id)
1840
+ or staging.get("workspace_id") != str(workspace_id)
1841
+ or staging.get("source_id") != source_id
1842
+ or staging.get("requested_artifacts")
1843
+ != {role: command[role] for role in _SOURCE_STAGING_ROLES}
1844
+ or staging.get("status") not in {"preparing", "uploading", "finalized"}
1845
+ ):
1846
+ raise HostedDeployError(
1847
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1848
+ f"Studio returned a source staging outside the request for {source_id}",
1849
+ )
1850
+ raw = staging.get("artifacts")
1851
+ if not isinstance(raw, list):
1852
+ raise HostedDeployError(
1853
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1854
+ f"Studio returned an invalid source staging inventory for {source_id}",
1855
+ )
1856
+ artifacts: dict[str, Mapping[str, Any]] = {}
1857
+ for item in raw:
1858
+ if not isinstance(item, dict):
1859
+ raise HostedDeployError(
1860
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1861
+ f"Studio returned an invalid source staging inventory for {source_id}",
1862
+ )
1863
+ role = item.get("role")
1864
+ if role not in _SOURCE_STAGING_ROLES or role in artifacts:
1865
+ raise HostedDeployError(
1866
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1867
+ f"Studio returned a duplicate or unknown source staging role for {source_id}",
1868
+ )
1869
+ _uuid(item.get("artifact_id"), "artifact_id")
1870
+ if not isinstance(item.get("upload_session"), dict):
1871
+ raise HostedDeployError(
1872
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1873
+ f"Studio omitted a source staging session for {source_id}",
1874
+ )
1875
+ artifacts[role] = item
1876
+ if set(artifacts) != set(_SOURCE_STAGING_ROLES):
1877
+ raise HostedDeployError(
1878
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1879
+ f"Studio returned an incomplete source staging for {source_id}",
1880
+ )
1881
+ return artifacts
1882
+
1883
+
1884
+ def _validate_source_upload_session(
1885
+ session: Mapping[str, Any],
1886
+ *,
1887
+ role: str,
1888
+ artifact_id: UUID,
1889
+ staging_id: UUID,
1890
+ workspace_id: UUID,
1891
+ descriptor: Mapping[str, Any],
1892
+ ) -> None:
1893
+ kind, purpose = _SOURCE_STAGING_ROLES[role]
1894
+ forbidden = {"attempt_id", "producer_generation", "fence", "verification_generation"}
1895
+ if (
1896
+ session.get("schema_version") != STUDIO_SCHEMA_VERSION
1897
+ or session.get("workspace_id") != str(workspace_id)
1898
+ or session.get("run_id") != str(staging_id)
1899
+ or session.get("artifact_id") != str(artifact_id)
1900
+ or session.get("kind") != kind
1901
+ or session.get("purpose") != purpose
1902
+ or session.get("classification") != "restricted"
1903
+ or session.get("media_type") != descriptor["media_type"]
1904
+ or session.get("contract_version") != STUDIO_SCHEMA_VERSION
1905
+ or session.get("direction") != "upload"
1906
+ or session.get("route_authority") != "source_staging_editor"
1907
+ or session.get("method") != "PUT"
1908
+ or session.get("expected_content_digest") != descriptor["expected_content_digest"]
1909
+ or session.get("expected_size_bytes") != descriptor["expected_size_bytes"]
1910
+ or session.get("single_use") is not True
1911
+ or any(field in session for field in forbidden)
1912
+ ):
1913
+ raise HostedDeployError(
1914
+ "DEPLOY_SIGNED_SESSION_INVALID",
1915
+ f"Studio returned an upload capability outside the {role} reservation",
1916
+ )
1917
+ for field in ("session_id", "reservation_id"):
1918
+ _uuid(session.get(field), field)
1919
+ for field in ("reservation_digest", "object_key_digest"):
1920
+ value = session.get(field)
1921
+ if (
1922
+ not isinstance(value, str)
1923
+ or not value.startswith("sha256:")
1924
+ or _DIGEST.fullmatch(value.removeprefix("sha256:")) is None
1925
+ ):
1926
+ raise HostedDeployError(
1927
+ "DEPLOY_SIGNED_SESSION_INVALID", f"Studio returned an invalid {field}"
1928
+ )
1929
+ issued_at = _timestamp(_required_text(session, "issued_at"), "issued_at")
1930
+ expires_at = _timestamp(_required_text(session, "expires_at"), "expires_at")
1931
+ if expires_at <= issued_at:
1932
+ raise HostedDeployError(
1933
+ "DEPLOY_SIGNED_SESSION_INVALID", "Studio returned an invalid upload session lifetime"
1934
+ )
1935
+ if expires_at <= datetime.now(UTC):
1936
+ raise HostedDeployError(
1937
+ "DEPLOY_SOURCE_UPLOAD_SESSION_REJECTED",
1938
+ "the signed source upload session expired before it could be used",
1939
+ )
1940
+
1941
+
1942
+ def _validate_source_authority(
1943
+ raw: Any,
1944
+ *,
1945
+ source_id: str,
1946
+ workspace_id: UUID,
1947
+ artifacts: Mapping[str, Mapping[str, Any]],
1948
+ descriptors: Mapping[str, Mapping[str, Any]],
1949
+ ) -> dict[str, Any]:
1950
+ if not isinstance(raw, dict):
1951
+ raise HostedDeployError(
1952
+ "DEPLOY_STUDIO_RESPONSE_INVALID", f"Studio omitted source authority for {source_id}"
1953
+ )
1954
+ authority = dict(raw)
1955
+ reference_fields = {
1956
+ "source_data": "source_artifact",
1957
+ "acquisition_receipt": "acquisition_receipt_artifact",
1958
+ "source_observation": "source_observation_artifact",
1959
+ }
1960
+ if (
1961
+ authority.get("source_id") != source_id
1962
+ or authority.get("authority_kind") != "sealed_artifact"
1963
+ ):
1964
+ raise HostedDeployError(
1965
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1966
+ f"Studio returned mismatched authority for {source_id}",
1967
+ )
1968
+ for role, field in reference_fields.items():
1969
+ reference = authority.get(field)
1970
+ if (
1971
+ not isinstance(reference, dict)
1972
+ or reference.get("workspace_id") != str(workspace_id)
1973
+ or reference.get("artifact_id") != artifacts[role].get("artifact_id")
1974
+ or reference.get("contract_version") != STUDIO_SCHEMA_VERSION
1975
+ or reference.get("content_digest") != descriptors[role]["expected_content_digest"]
1976
+ ):
1977
+ raise HostedDeployError(
1978
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1979
+ f"Studio returned a mismatched {role} authority for {source_id}",
1980
+ )
1981
+ digest = authority.get("source_authority_digest")
1982
+ expected = canonical.canonical_sha256(
1983
+ {key: value for key, value in authority.items() if key != "source_authority_digest"}
1984
+ )
1985
+ if digest != expected:
1986
+ raise HostedDeployError(
1987
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
1988
+ f"Studio returned a mismatched authority digest for {source_id}",
1989
+ )
1990
+ return authority
1991
+
1992
+
1993
+ #: Where a journaled plan records one digest per top-level field of the request body.
1994
+ #:
1995
+ #: The whole-body digest says *that* a replayed request differs. This says *which* field does, and
1996
+ #: it is the difference between a refusal a person can act on and one they can only stare at. It is
1997
+ #: derived from the same body ``request_digest`` already covers, so it adds naming and never
1998
+ #: authority: no check reads it to decide whether a request matches, only to describe how it does
1999
+ #: not.
2000
+ _REQUEST_FIELDS = "request_field_digests"
2001
+
2002
+
2003
+ def _request_field_digests(body: Mapping[str, Any]) -> dict[str, str]:
2004
+ """One digest per top-level field of a request body, so a replay can name what moved."""
2005
+
2006
+ return {str(field): canonical.canonical_sha256(value) for field, value in body.items()}
2007
+
2008
+
2009
+ def _journaled_request_digest(journaled: Any) -> str | None:
2010
+ """The request digest a journaled operation records, or ``None`` when it records none."""
2011
+
2012
+ if not isinstance(journaled, Mapping):
2013
+ return None
2014
+ recorded = journaled.get("request_digest")
2015
+ if not isinstance(recorded, str) or _DIGEST.fullmatch(recorded) is None:
2016
+ return None
2017
+ return recorded
2018
+
2019
+
2020
+ def _drifted_request_fields(journaled: Any, fields: Mapping[str, str]) -> str:
2021
+ """The request fields that differ from the journaled plan, named for a person to read.
2022
+
2023
+ A plan journaled before per-field digests existed can only say that the request as a whole
2024
+ differs, and that is exactly what this says about it. Naming a field the journal cannot see
2025
+ would be a guess, and a guess in a refusal sends an operator to fix the wrong input.
2026
+ """
2027
+
2028
+ recorded = journaled.get(_REQUEST_FIELDS) if isinstance(journaled, Mapping) else None
2029
+ if not isinstance(recorded, Mapping):
2030
+ return "the request"
2031
+ drifted = sorted(
2032
+ field for field in {*recorded, *fields} if recorded.get(field) != fields.get(field)
2033
+ )
2034
+ return ", ".join(drifted) if drifted else "the request"
2035
+
2036
+
2037
+ def _same_plan(journaled: Any, planned: Mapping[str, Any]) -> bool:
2038
+ """Whether a journaled plan is the same plan, including one journaled before field naming.
2039
+
2040
+ ``request_field_digests`` is derived from the body ``request_digest`` already covers, so a plan
2041
+ journaled without it is held to exactly the evidence it does carry, and one that carries it is
2042
+ compared whole. A deployment interrupted before this field existed stays resumable rather than
2043
+ having a recoverable lost response turned into a refusal by an upgrade.
2044
+ """
2045
+
2046
+ if not isinstance(journaled, Mapping):
2047
+ return False
2048
+ if _REQUEST_FIELDS in journaled:
2049
+ return dict(journaled) == dict(planned)
2050
+ return dict(journaled) == {
2051
+ field: value for field, value in planned.items() if field != _REQUEST_FIELDS
2052
+ }
2053
+
2054
+
2055
+ def _mutate(
2056
+ state: dict[str, Any],
2057
+ store: DeploymentStateStore,
2058
+ client: StudioDeploymentClient,
2059
+ *,
2060
+ name: str,
2061
+ operation: str,
2062
+ body: Mapping[str, Any],
2063
+ expected: set[int],
2064
+ resource_id: UUID | None = None,
2065
+ if_match: str | None = None,
2066
+ requires_etag: bool = False,
2067
+ ) -> Mapping[str, Any]:
2068
+ operations = state["operations"]
2069
+ resources = state["resources"]
2070
+ request_digest = canonical.canonical_sha256(dict(body))
2071
+ fields = _request_field_digests(body)
2072
+ existing = resources.get(name)
2073
+ if isinstance(existing, dict):
2074
+ # A completed operation short-circuits to its journaled response, and the request this
2075
+ # invocation rebuilt is discarded rather than sent. So a local input that moved since
2076
+ # staging is not transmitted -- it is IGNORED, silently, and the operator is handed the
2077
+ # result of the request they no longer have. The journal has recorded the digest of what
2078
+ # was actually sent since v3 for exactly this comparison; until now nothing read it.
2079
+ journaled = operations.get(name)
2080
+ recorded = _journaled_request_digest(journaled)
2081
+ if recorded is None:
2082
+ raise HostedDeployError(
2083
+ "DEPLOY_STATE_INVALID",
2084
+ f"deployment operation {name} recorded a result without the request that "
2085
+ "produced it",
2086
+ )
2087
+ if recorded != request_digest:
2088
+ raise HostedDeployError(
2089
+ "DEPLOY_STATE_MISMATCH",
2090
+ f"deployment operation {name} already ran, and "
2091
+ f"{_drifted_request_fields(journaled, fields)} here no longer matches what it "
2092
+ "sent; this run would keep the recorded result and ignore the change, so start a "
2093
+ "new deployment for it",
2094
+ )
2095
+ return existing
2096
+ idempotency_key = _idempotency(name, state["deployment_identity"])
2097
+ operation_state = operations.get(name)
2098
+ planned = {
2099
+ "operation": operation,
2100
+ "request_digest": request_digest,
2101
+ _REQUEST_FIELDS: fields,
2102
+ "idempotency_key": idempotency_key,
2103
+ "status": "planned",
2104
+ }
2105
+ if operation_state is None:
2106
+ operations[name] = planned
2107
+ store.save(state)
2108
+ elif not _same_plan(operation_state, planned):
2109
+ raise HostedDeployError(
2110
+ "DEPLOY_STATE_MISMATCH",
2111
+ f"deployment operation {name} changed after journaling: "
2112
+ f"{_drifted_request_fields(operation_state, fields)} is not what was journaled",
2113
+ )
2114
+ response = client.call(
2115
+ operation,
2116
+ body,
2117
+ idempotency_key=idempotency_key,
2118
+ resource_id=resource_id,
2119
+ if_match=if_match,
2120
+ )
2121
+ result = dict(_success(response, expected, "DEPLOY_STUDIO_MUTATION_FAILED"))
2122
+ if requires_etag and response.etag is None:
2123
+ # Refuse to journal this as completed. A later step conditions its write on this
2124
+ # response's ETag and reads it back from the journal, never from the wire -- so
2125
+ # recording "completed" without one produces a deployment that cannot be finished
2126
+ # AND cannot be resumed, because a completed operation is never re-issued. The
2127
+ # proposal it already created still holds the Recipe digest, so the Build cannot be
2128
+ # retried either.
2129
+ #
2130
+ # Leaving the operation "planned" is what makes this recoverable: a resume re-sends
2131
+ # it under the same idempotency key, and a server that has learned to return the
2132
+ # ETag answers the replay with one. That is exactly how this was recovered when
2133
+ # Studio omitted it (mostlyrightmd/mostlyright-studio#57).
2134
+ raise HostedDeployError(
2135
+ "DEPLOY_STUDIO_RESPONSE_INVALID",
2136
+ f"Studio omitted the ETag after {name}, which the next step must write against; "
2137
+ "nothing was journaled as complete, so rerunning with --resume will ask again",
2138
+ )
2139
+ resources[name] = result
2140
+ if response.etag is not None:
2141
+ state.setdefault("etags", {})[name] = response.etag
2142
+ operations[name] = {**planned, "status": "completed"}
2143
+ store.save(state)
2144
+ return result
2145
+
2146
+
2147
+ def _upload_once(
2148
+ state: dict[str, Any],
2149
+ store: DeploymentStateStore,
2150
+ client: StudioDeploymentClient,
2151
+ *,
2152
+ name: str,
2153
+ session: Mapping[str, Any],
2154
+ content: bytes,
2155
+ ) -> str | None:
2156
+ operations = state["operations"]
2157
+ completed = operations.get(name, {})
2158
+ status = completed.get("status")
2159
+ content_digest = hashlib.sha256(content).hexdigest()
2160
+ if status in {"completed", "reconciled"}:
2161
+ # Both of these short-circuit an object Studio has already sealed, and neither one sends
2162
+ # the bytes this invocation just read. Local source evidence that moved since staging is
2163
+ # therefore ignored rather than uploaded, so it is compared against the digest the journal
2164
+ # recorded when the transfer actually happened.
2165
+ recorded = _journaled_request_digest(completed)
2166
+ if recorded is None:
2167
+ raise HostedDeployError(
2168
+ "DEPLOY_STATE_INVALID",
2169
+ f"source upload {name} recorded a result without the content it sent",
2170
+ )
2171
+ if recorded != content_digest:
2172
+ raise HostedDeployError(
2173
+ "DEPLOY_STATE_MISMATCH",
2174
+ f"source upload {name} already ran, and the content here no longer matches what "
2175
+ "it sent; this run would keep the sealed object and ignore the change, so start a "
2176
+ "new deployment for it",
2177
+ )
2178
+ if status == "reconciled":
2179
+ return None
2180
+ generation = completed.get("object_generation")
2181
+ if not isinstance(generation, str) or _OBJECT_GENERATION.fullmatch(generation) is None:
2182
+ raise HostedDeployError(
2183
+ "DEPLOY_STATE_INVALID", f"source upload {name} has no object generation"
2184
+ )
2185
+ return generation
2186
+ planned = {
2187
+ "operation": "signed_upload",
2188
+ "request_digest": content_digest,
2189
+ "session_id": session.get("session_id"),
2190
+ "status": "planned",
2191
+ }
2192
+ current = operations.get(name)
2193
+ if current is None:
2194
+ operations[name] = planned
2195
+ store.save(state)
2196
+ elif current != planned:
2197
+ raise HostedDeployError(
2198
+ "DEPLOY_STATE_MISMATCH", f"source upload {name} changed after journaling"
2199
+ )
2200
+ try:
2201
+ generation = client.upload(session, content)
2202
+ except HostedDeployError as error:
2203
+ if error.code != "DEPLOY_SOURCE_UPLOAD_FAILED":
2204
+ raise
2205
+ # A transfer can commit remotely while its response is lost. Leave the upload planned;
2206
+ # finalization or the next idempotent replay will reconcile the exact sealed object.
2207
+ return None
2208
+ operations[name] = {**planned, "status": "completed", "object_generation": generation}
2209
+ store.save(state)
2210
+ return generation
2211
+
2212
+
2213
+ def _approval_receipt(
2214
+ state: Mapping[str, Any], store: DeploymentStateStore, cloud_url: str
2215
+ ) -> dict[str, Any]:
2216
+ proposal = state["resources"]["proposal"]
2217
+ return {
2218
+ "schema_version": DEPLOYMENT_RECEIPT_SCHEMA,
2219
+ "status": "deployment_approval_required",
2220
+ "workspace_id": state["workspace_id"],
2221
+ "recipe_proposal_id": proposal["recipe_proposal_id"],
2222
+ "approval_request_id": state["resources"]["approval_request_id"],
2223
+ "state": str(store.path),
2224
+ "dashboard_url": (
2225
+ f"{_service_url(cloud_url, 'Cloud URL', allow_loopback_http=False)}/dashboard/approvals"
2226
+ ),
2227
+ "next": (
2228
+ "approve the exact Recipe in Cloud Approvals, then rerun this command with --resume"
2229
+ ),
2230
+ }
2231
+
2232
+
2233
+ def _activation_receipt(
2234
+ state: Mapping[str, Any], store: DeploymentStateStore, cloud_url: str
2235
+ ) -> dict[str, Any]:
2236
+ activation = state["resources"]["activation"]
2237
+ binding = state["resources"]["cloud_binding"]
2238
+ table_id = _uuid_field(activation, "table_id")
2239
+ run_id = _uuid_field(activation, "run_id")
2240
+ return {
2241
+ "schema_version": DEPLOYMENT_RECEIPT_SCHEMA,
2242
+ "status": "dataset_activation_queued",
2243
+ "workspace_id": state["workspace_id"],
2244
+ "dataset_id": state["resources"]["dataset_id"],
2245
+ "table_plan_id": state["resources"]["table_plan_id"],
2246
+ "table_id": str(table_id),
2247
+ "cloud_dataset_id": _required_text(binding, "cloud_dataset_id"),
2248
+ "current_path": _required_text(binding, "current_path"),
2249
+ "run_id": str(run_id),
2250
+ # The reviewed schedule is the coordinate that says when this Dataset refreshes itself,
2251
+ # and it is the one activation coordinate the receipt used to omit -- so the operator who
2252
+ # just deployed had no way to read back the cadence Studio activated without reopening
2253
+ # the journal. It is the exact object the activation body carried, held in ``state`` from
2254
+ # the moment the deployment was initialized.
2255
+ "schedule": dict(state["schedule"]),
2256
+ "recipe_proposal_id": state["resources"]["proposal"]["recipe_proposal_id"],
2257
+ "state": str(store.path),
2258
+ "dashboard_url": (
2259
+ f"{_service_url(cloud_url, 'Cloud URL', allow_loopback_http=False)}/dashboard/datasets"
2260
+ ),
2261
+ # The queued Run is not a released version, and this receipt is where somebody learns that
2262
+ # for the first time. Its sibling above has always ended by naming the next command; this
2263
+ # one ended at a dashboard URL, which is exactly the trip to Cloud the status command now
2264
+ # removes.
2265
+ "next": "run mr-data deploy-status RUN_DIR to follow the run this queued",
2266
+ }
2267
+
2268
+
2269
+ def cloud_dashboard_url(cloud_url: str, page: str) -> str:
2270
+ """The Cloud page an operator opens about this deployment, from one validated origin.
2271
+
2272
+ Every receipt that points somebody at Cloud goes through here, so the origin is checked the
2273
+ same way each time and the three surfaces cannot drift into three spellings of one URL.
2274
+ """
2275
+
2276
+ origin = _service_url(cloud_url, "Cloud URL", allow_loopback_http=False)
2277
+ return f"{origin}/dashboard/workspace/{page}"
2278
+
2279
+
2280
+ def _bind_cloud_dataset(
2281
+ request: DeploymentRequest,
2282
+ state: dict[str, Any],
2283
+ store: DeploymentStateStore,
2284
+ *,
2285
+ credentials: ResolvedCloudCredentials,
2286
+ transport: TokenExchangeTransport | None,
2287
+ ) -> Mapping[str, Any]:
2288
+ _require_cli_credential(credentials)
2289
+ resources = state["resources"]
2290
+ activation = resources.get("activation")
2291
+ if not isinstance(activation, dict):
2292
+ raise HostedDeployError(
2293
+ "DEPLOY_STATE_INVALID", "Cloud Dataset binding requires a completed activation"
2294
+ )
2295
+ body = {
2296
+ "schema_version": CLOUD_TABLE_BINDING_SCHEMA_VERSION,
2297
+ "studio_workspace_id": state["workspace_id"],
2298
+ "studio_dataset_id": _required_text(resources, "dataset_id"),
2299
+ "studio_table_id": str(_uuid_field(activation, "table_id")),
2300
+ "table_recipe": {
2301
+ "table_recipe_id": request.recipe_id,
2302
+ "recipe_version": request.recipe_version,
2303
+ "recipe_digest": request.recipe_digest,
2304
+ },
2305
+ }
2306
+ request_digest = canonical.canonical_sha256(body)
2307
+ fields = _request_field_digests(body)
2308
+ operations = state["operations"]
2309
+ existing = resources.get("cloud_binding")
2310
+ if isinstance(existing, dict):
2311
+ # The binding Cloud already persisted is returned without contacting Cloud again, so a
2312
+ # rebuilt body that moved is ignored rather than sent. The body is built before this
2313
+ # short-circuit for that comparison alone.
2314
+ journaled = operations.get("cloud_binding")
2315
+ recorded = _journaled_request_digest(journaled)
2316
+ if recorded is None:
2317
+ raise HostedDeployError(
2318
+ "DEPLOY_STATE_INVALID",
2319
+ "the Cloud Dataset binding recorded a result without the request that produced it",
2320
+ )
2321
+ if recorded != request_digest:
2322
+ raise HostedDeployError(
2323
+ "DEPLOY_STATE_MISMATCH",
2324
+ "the Cloud Dataset binding already ran, and "
2325
+ f"{_drifted_request_fields(journaled, fields)} here no longer matches what it "
2326
+ "sent; this run would keep the recorded binding and ignore the change, so start a "
2327
+ "new deployment for it",
2328
+ )
2329
+ return existing
2330
+ planned = {
2331
+ "operation": "bind_cloud_dataset",
2332
+ "request_digest": request_digest,
2333
+ _REQUEST_FIELDS: fields,
2334
+ "status": "planned",
2335
+ }
2336
+ current = operations.get("cloud_binding")
2337
+ if current is None:
2338
+ operations["cloud_binding"] = planned
2339
+ store.save(state)
2340
+ elif not _same_plan(current, planned):
2341
+ raise HostedDeployError(
2342
+ "DEPLOY_STATE_MISMATCH",
2343
+ "the Cloud Dataset binding changed after journaling: "
2344
+ f"{_drifted_request_fields(current, fields)} is not what was journaled",
2345
+ )
2346
+ selected = transport or UrlLibTokenExchangeTransport()
2347
+ cloud_url = _service_url(credentials.cloud_url, "Cloud URL", allow_loopback_http=False)
2348
+ try:
2349
+ status, raw, _response_headers = selected.request(
2350
+ "POST",
2351
+ f"{cloud_url}{CLOUD_DATASET_BINDINGS_PATH}",
2352
+ {
2353
+ "Accept": "application/json",
2354
+ "Content-Type": "application/json",
2355
+ "x-api-key": credentials.raw_key,
2356
+ "Idempotency-Key": _idempotency(
2357
+ "cloud-table-binding", state["deployment_identity"]
2358
+ ),
2359
+ },
2360
+ canonical.canonical_json_bytes(body),
2361
+ MAX_BINDING_RESPONSE_BYTES,
2362
+ )
2363
+ except HostedDeployError:
2364
+ raise
2365
+ except Exception as error:
2366
+ raise HostedDeployError(
2367
+ "DEPLOY_CLOUD_BINDING_FAILED",
2368
+ "the Cloud Dataset binding could not be reached; rerun with --resume",
2369
+ ) from error
2370
+ try:
2371
+ parsed = canonical.parse_json(raw)
2372
+ except canonical.CanonicalJSONError:
2373
+ parsed = None
2374
+ if status == 401:
2375
+ raise HostedDeployError(
2376
+ "DEPLOY_AUTHENTICATION_FAILED", "the selected CLI credential was rejected"
2377
+ )
2378
+ if status == 402:
2379
+ raise HostedDeployError(
2380
+ "DEPLOY_SUBSCRIPTION_REQUIRED", "this workspace has no hosted deployment access"
2381
+ )
2382
+ if status == 409:
2383
+ raise HostedDeployError(
2384
+ "DEPLOY_CLOUD_BINDING_CONFLICT",
2385
+ "Cloud could not bind the exact accepted Studio activation",
2386
+ )
2387
+ if status in {429, 503}:
2388
+ raise HostedDeployError(
2389
+ "DEPLOY_CLOUD_BINDING_UNAVAILABLE",
2390
+ "Cloud could not finish Dataset binding; rerun with --resume",
2391
+ )
2392
+ if status != 200 or not isinstance(parsed, dict):
2393
+ raise HostedDeployError(
2394
+ "DEPLOY_CLOUD_BINDING_FAILED",
2395
+ f"the Cloud Dataset binding failed (HTTP {status})",
2396
+ )
2397
+ if set(parsed) != {
2398
+ "schema_version",
2399
+ "cloud_dataset_id",
2400
+ "cloud_table_id",
2401
+ "studio_dataset_id",
2402
+ "studio_table_id",
2403
+ "current_path",
2404
+ }:
2405
+ raise HostedDeployError(
2406
+ "DEPLOY_CLOUD_BINDING_FAILED", "Cloud returned an invalid Dataset binding"
2407
+ )
2408
+ if parsed["schema_version"] != CLOUD_TABLE_BINDING_SCHEMA_VERSION:
2409
+ raise HostedDeployError(
2410
+ "DEPLOY_CLOUD_BINDING_FAILED", "Cloud returned an invalid binding schema"
2411
+ )
2412
+ cloud_dataset_id = _uuid(_required_text(parsed, "cloud_dataset_id"), "cloud_dataset_id")
2413
+ cloud_table_id = _uuid(_required_text(parsed, "cloud_table_id"), "cloud_table_id")
2414
+ studio_dataset_id = _uuid(_required_text(parsed, "studio_dataset_id"), "studio_dataset_id")
2415
+ studio_table_id = _uuid(_required_text(parsed, "studio_table_id"), "studio_table_id")
2416
+ expected_studio_dataset_id = _uuid(_required_text(resources, "dataset_id"), "dataset_id")
2417
+ expected_studio_table_id = _uuid_field(activation, "table_id")
2418
+ current_path = _required_text(parsed, "current_path")
2419
+ if (
2420
+ studio_dataset_id != expected_studio_dataset_id
2421
+ or studio_table_id != expected_studio_table_id
2422
+ or current_path != f"/api/v2/tables/{cloud_table_id}/current"
2423
+ ):
2424
+ raise HostedDeployError(
2425
+ "DEPLOY_CLOUD_BINDING_FAILED", "Cloud returned a mismatched Dataset binding"
2426
+ )
2427
+ result = {
2428
+ "cloud_dataset_id": str(cloud_dataset_id),
2429
+ "cloud_table_id": str(cloud_table_id),
2430
+ "studio_dataset_id": str(studio_dataset_id),
2431
+ "studio_table_id": str(studio_table_id),
2432
+ "current_path": current_path,
2433
+ }
2434
+ resources["cloud_binding"] = result
2435
+ operations["cloud_binding"] = {**planned, "status": "completed"}
2436
+ store.save(state)
2437
+ return result
2438
+
2439
+
2440
+ def _deployment_predecessor(
2441
+ request: DeploymentRequest, previous_run_dir: Path | None
2442
+ ) -> str | None:
2443
+ """Settle which live Dataset this deployment updates, before it writes or sends anything of
2444
+ its own.
2445
+
2446
+ ``mr-data deploy`` has one predecessor, not two. ``--previous-run-dir`` names the run directory
2447
+ whose deployment published the Dataset this version updates, and the same directory is what
2448
+ :func:`recipe.verify_recipe_candidate` authenticates the version chain out of. The two layers
2449
+ ask one question because :func:`deploy.build_deployment_request` refuses both shapes where they
2450
+ could differ: a Recipe execution whose mode is not ``initial`` must name its predecessor, and
2451
+ one whose mode is ``initial`` must not. So by the time a request reaches here, "a predecessor
2452
+ was named" and "this Build follows a sealed one" are the same fact, and that -- not the Recipe
2453
+ version -- is what :func:`_prepare_and_request_approval` branches on. A refresh of an unchanged
2454
+ Recipe is a successor whose Recipe version has not moved.
2455
+
2456
+ What the Recipe version still decides is the one shape it alone can see: a successor Recipe
2457
+ exists because a Dataset it follows exists, so deploying one that names no predecessor at all
2458
+ would found a second Dataset and strand the first. That is refused here rather than at the
2459
+ Studio call, so it leaves no journal behind -- ``DEPLOY_PREDECESSOR_REQUIRED`` names a flag,
2460
+ and the rerun that supplies it has to be able to stage cleanly instead of meeting the identity
2461
+ gate holding the journal of the run that refused.
2462
+
2463
+ Returns:
2464
+ The predecessor coordinate this deployment's identity is computed from, or None when there
2465
+ is no predecessor.
2466
+ """
2467
+
2468
+ if previous_run_dir is None:
2469
+ if request.recipe_version > 1:
2470
+ # `_previous_deployment_resources` owns this sentence; calling it with nothing is what
2471
+ # moves its refusal ahead of the journal this run would otherwise have written.
2472
+ _previous_deployment_resources(previous_run_dir)
2473
+ return None
2474
+ # Resolved, because a resume is run from whatever directory the operator happens to be standing
2475
+ # in and the predecessor it names is the same one either way.
2476
+ coordinate = str(Path(previous_run_dir).resolve())
2477
+ # Reads the predecessor's own activation receipt, so one that never finished activating is
2478
+ # refused here rather than at the Studio call. This does open a journal store on the
2479
+ # predecessor, which creates its `evidence/deployment` directory and takes a lock there; what
2480
+ # is not yet written is anything belonging to the deployment being started.
2481
+ _previous_deployment_resources(previous_run_dir)
2482
+ return coordinate
2483
+
2484
+
2485
+ def _previous_deployment_resources(previous_run_dir: Path | None) -> Mapping[str, str]:
2486
+ if previous_run_dir is None:
2487
+ raise HostedDeployError(
2488
+ "DEPLOY_PREDECESSOR_REQUIRED", "a successor deployment needs --previous-run-dir"
2489
+ )
2490
+ store = FileDeploymentStateStore(previous_run_dir)
2491
+ try:
2492
+ state = store.load()
2493
+ finally:
2494
+ store.close()
2495
+ resources = state.get("resources")
2496
+ activation = resources.get("activation") if isinstance(resources, dict) else None
2497
+ if not isinstance(resources, dict) or not isinstance(activation, dict):
2498
+ raise HostedDeployError(
2499
+ "DEPLOY_PREDECESSOR_INVALID", "the previous Build has no completed activation receipt"
2500
+ )
2501
+ return {
2502
+ "dataset_id": _required_text(resources, "dataset_id"),
2503
+ "table_plan_id": _required_text(resources, "table_plan_id"),
2504
+ "table_id": str(_uuid_field(activation, "table_id")),
2505
+ }
2506
+
2507
+
2508
+ def _source_locator(
2509
+ source: Mapping[str, Any], proposal: Mapping[str, Any], bundle: Mapping[str, Any]
2510
+ ) -> dict[str, Any]:
2511
+ if bundle["authority_kind"] == "sealed_artifact":
2512
+ return {
2513
+ "kind": "artifact",
2514
+ "display_locator": "urn:sha256:" + bundle["source_data"]["content_sha256"],
2515
+ }
2516
+ if source["adapter_id"] == "public.https":
2517
+ return {
2518
+ "kind": "https_url",
2519
+ "display_locator": source["query_template"]["url"],
2520
+ "connector_contract_version": source["adapter_version"],
2521
+ }
2522
+ return {
2523
+ "kind": "sdk_connector",
2524
+ "display_locator": proposal["locator"],
2525
+ "connector_contract_version": source["adapter_version"],
2526
+ }
2527
+
2528
+
2529
+ def _studio_feasibility(value: Mapping[str, Any]) -> dict[str, Any]:
2530
+ mapping = {
2531
+ "sufficient_evidence": "SUFFICIENT_EVIDENCE",
2532
+ "limited_coverage": "LIMITED_COVERAGE",
2533
+ "unacceptable_delay": "UNACCEPTABLE_DELAY",
2534
+ "rights_unclear": "RIGHTS_UNCLEAR",
2535
+ "no_reliable_source": "NO_RELIABLE_SOURCE",
2536
+ "target_not_observable": "TARGET_NOT_OBSERVABLE",
2537
+ }
2538
+ return {
2539
+ "decision": value["decision"],
2540
+ "reason_codes": [mapping[item] for item in value["reason_codes"]],
2541
+ "narrative": value["narrative"],
2542
+ }
2543
+
2544
+
2545
+ def _default_backfill_policy(time_range: Mapping[str, Any]) -> dict[str, Any]:
2546
+ start = _timestamp(_required_text(time_range, "start_inclusive"), "start_inclusive")
2547
+ end = _timestamp(_required_text(time_range, "end_exclusive"), "end_exclusive")
2548
+ seconds = int((end - start).total_seconds())
2549
+ if seconds < 1:
2550
+ raise HostedDeployError("DEPLOY_REQUEST_INVALID", "Recipe time range is empty")
2551
+ return {
2552
+ "schema_version": "local-backfill-policy.v1",
2553
+ "coverage_envelope": dict(time_range),
2554
+ "max_window_seconds": seconds,
2555
+ "overlap_policy": "disallow_released_coverage",
2556
+ "partition_scope": {"fields": [], "allowed_values": {}, "max_partitions_per_run": 1},
2557
+ }
2558
+
2559
+
2560
+ #: Every operator-visible input the deployment identity is computed from, and what to call it when
2561
+ #: it moves. A resume is by definition something a person comes back to hours later, after an
2562
+ #: approval, with the original command long gone from their scrollback, so getting one of these
2563
+ #: wrong is the expected case rather than the exceptional one.
2564
+ #:
2565
+ #: Two of them are not arguments at all. The Cloud URL comes from the environment, and the review
2566
+ #: authority comes from whether this shell is inside a coordinator review session -- which is why a
2567
+ #: refusal that names the review authority must not offer an argument list as the fix.
2568
+ _IDENTITY_INPUT_LABELS: dict[str, str] = {
2569
+ "dataset_name": "--dataset-name",
2570
+ "cron": "--cron",
2571
+ "previous_run_dir": "--previous-run-dir",
2572
+ "max_runs_per_month": "--max-runs-per-month",
2573
+ "max_concurrent_runs": "--max-concurrent-runs",
2574
+ "max_monthly_cost_units": "--max-monthly-cost-units",
2575
+ "recipe_id": "the Recipe id",
2576
+ "recipe_version": "the Recipe version",
2577
+ "recipe_digest": "the Recipe",
2578
+ "candidate_digest": "the reviewed Build",
2579
+ "cloud_url": "the Cloud URL",
2580
+ "workspace_id": "the workspace",
2581
+ "review_authority": "the review authority",
2582
+ }
2583
+
2584
+ #: The journal key under which a staged deployment records the inputs above.
2585
+ _IDENTITY_INPUTS = "identity_inputs"
2586
+
2587
+ #: Which authority the staged request carried. A request carries the signed reviewed-Build binding
2588
+ #: or it carries none, and which one it is moves the request digest; nothing on the command line
2589
+ #: says so.
2590
+ _REVIEW_SESSION = "review_session"
2591
+ _UNATTESTED = "unattested"
2592
+
2593
+ #: What can still move a request digest once every named input matches. None of it is an argument,
2594
+ #: which is what makes listing it honest rather than unhelpful: it tells an operator that no rerun
2595
+ #: of the same command will converge.
2596
+ _UNNAMED_REQUEST_INPUTS = (
2597
+ "the sealed Build evidence the request carries",
2598
+ "the release of this command that staged it",
2599
+ )
2600
+
2601
+
2602
+ def _identity_inputs(
2603
+ request: DeploymentRequest,
2604
+ *,
2605
+ workspace_id: UUID,
2606
+ cloud_url: str,
2607
+ cron: str,
2608
+ max_concurrent_runs: int,
2609
+ previous_run_dir: str | None,
2610
+ ) -> dict[str, Any]:
2611
+ """The named inputs behind one deployment identity, in the shape the journal records them.
2612
+
2613
+ ``previous_run_dir`` is recorded on every deployment, including the initial ones where it is
2614
+ None. The identity digest carries it only when there is one -- an absent key there would move
2615
+ every identity staged before this release -- but the journal has no such constraint, and
2616
+ recording the absence is what lets a resume that adds a predecessor be named as the input that
2617
+ moved rather than listed as one of the inputs the journal could not resolve.
2618
+ """
2619
+
2620
+ return {
2621
+ "candidate_digest": request.candidate_digest,
2622
+ "cloud_url": cloud_url,
2623
+ "cron": cron,
2624
+ "table_name": request.dataset_name,
2625
+ "max_concurrent_runs": max_concurrent_runs,
2626
+ "max_monthly_cost_units": request.caps.max_monthly_cost_units,
2627
+ "max_runs_per_month": request.caps.max_runs_per_month,
2628
+ "previous_run_dir": previous_run_dir,
2629
+ "recipe_digest": request.recipe_digest,
2630
+ "table_recipe_id": request.recipe_id,
2631
+ "recipe_version": request.recipe_version,
2632
+ "review_authority": (
2633
+ _REVIEW_SESSION if request.review_decision_digest is not None else _UNATTESTED
2634
+ ),
2635
+ "workspace_id": str(workspace_id),
2636
+ }
2637
+
2638
+
2639
+ def _journaled_identity_inputs(state: Mapping[str, Any]) -> dict[str, Any]:
2640
+ """What this journal can say about the inputs it was staged with.
2641
+
2642
+ A journal written since staging began recording them answers for every input. An older one
2643
+ answers only for what it happened to need: the schedule it sent to Studio, and -- once Studio
2644
+ held a Recipe proposal -- the Dataset name inside the activation intent. Whatever it cannot
2645
+ answer is left out here, so a refusal lists it as a possibility rather than saying it matched.
2646
+ """
2647
+
2648
+ recorded = state.get(_IDENTITY_INPUTS)
2649
+ if isinstance(recorded, Mapping):
2650
+ return {name: recorded[name] for name in _IDENTITY_INPUT_LABELS if name in recorded}
2651
+ inputs: dict[str, Any] = {}
2652
+ schedule = state.get("schedule")
2653
+ if isinstance(schedule, Mapping):
2654
+ for name in ("cron", "max_concurrent_runs"):
2655
+ if name in schedule:
2656
+ inputs[name] = schedule[name]
2657
+ resources = state.get("resources")
2658
+ proposal = resources.get("proposal") if isinstance(resources, Mapping) else None
2659
+ intent = proposal.get("activation_intent") if isinstance(proposal, Mapping) else None
2660
+ if isinstance(intent, Mapping) and "dataset_name" in intent:
2661
+ inputs["table_name"] = intent["table_name"]
2662
+ return inputs
2663
+
2664
+
2665
+ def _drifted_identity_inputs(journaled: Mapping[str, Any], inputs: Mapping[str, Any]) -> list[str]:
2666
+ """The named inputs this journal can prove moved, in a stable order."""
2667
+
2668
+ return sorted(name for name, value in journaled.items() if inputs.get(name) != value)
2669
+
2670
+
2671
+ def _identity_drift_sentence(
2672
+ journaled: Mapping[str, Any], inputs: Mapping[str, Any], names: list[str]
2673
+ ) -> str:
2674
+ """Each differing input, as the value it was staged with and the value it has here."""
2675
+
2676
+ return "; ".join(
2677
+ f"{_IDENTITY_INPUT_LABELS[name]} was staged as {journaled[name]} and is "
2678
+ f"{inputs.get(name)} here"
2679
+ for name in names
2680
+ )
2681
+
2682
+
2683
+ def _validate_state(
2684
+ state: Mapping[str, Any],
2685
+ identity: str,
2686
+ workspace_id: UUID,
2687
+ *,
2688
+ inputs: Mapping[str, Any],
2689
+ request_digest: str,
2690
+ ) -> None:
2691
+ """Hold one journal to one exact handoff, and say which input moved when it does not.
2692
+
2693
+ The identity comparison is unchanged: one digest over the request, the workspace, the Cloud
2694
+ URL, the schedule and the concurrency cap, and it either names this handoff or it does not.
2695
+ What changes is what a person is told when it does not. The journal is holding the answer --
2696
+ it recorded the inputs it was staged with -- and until now the refusal said only that the two
2697
+ disagreed, which is the one thing the operator already knew.
2698
+ """
2699
+
2700
+ if (
2701
+ state.get("schema_version") != DEPLOYMENT_STATE_SCHEMA
2702
+ or not isinstance(state.get("operations"), dict)
2703
+ or not isinstance(state.get("resources"), dict)
2704
+ ):
2705
+ raise HostedDeployError(
2706
+ "DEPLOY_STATE_MISMATCH", "the deployment journal belongs to another exact handoff"
2707
+ )
2708
+ staged_workspace = state.get("workspace_id")
2709
+ if staged_workspace != str(workspace_id):
2710
+ raise HostedDeployError(
2711
+ "DEPLOY_STATE_MISMATCH",
2712
+ f"this deployment journal was staged in workspace {staged_workspace} and this run is "
2713
+ f"authenticated to workspace {workspace_id}; sign in to the one it was staged in, or "
2714
+ "start a new deployment",
2715
+ )
2716
+ if state.get("deployment_identity") == identity:
2717
+ return
2718
+ journaled = _journaled_identity_inputs(state)
2719
+ drifted = _drifted_identity_inputs(journaled, inputs)
2720
+ if "review_authority" in drifted:
2721
+ # The sixth input, and the only one that is invisible from the command line: `mr-data
2722
+ # deploy` reads the review authority off the ambient shell, so a session-staged deployment
2723
+ # resumed from an ordinary terminal rebuilds an unattested request, moves the digest, and
2724
+ # lands here with no differing flag to name. Printing an argument list for this one would
2725
+ # send an operator to retype a command that already matches.
2726
+ if journaled["review_authority"] == _REVIEW_SESSION:
2727
+ raise HostedDeployError(
2728
+ "DEPLOY_STATE_MISMATCH",
2729
+ "this deployment was staged inside a coordinator review session and this run has "
2730
+ "none, so it rebuilt a different request; no argument list can bridge that -- "
2731
+ "resume it from a review session, or start a new deployment",
2732
+ )
2733
+ raise HostedDeployError(
2734
+ "DEPLOY_STATE_MISMATCH",
2735
+ "this deployment was staged outside a coordinator review session and this run is "
2736
+ "inside one, so it rebuilt a different request; no argument list can bridge that -- "
2737
+ "resume it outside the session, or start a new deployment",
2738
+ )
2739
+ if drifted:
2740
+ raise HostedDeployError(
2741
+ "DEPLOY_STATE_MISMATCH",
2742
+ "this deployment journal was staged with different inputs: "
2743
+ f"{_identity_drift_sentence(journaled, inputs, drifted)}; rerun it with the values it "
2744
+ "was staged with, or start a new deployment",
2745
+ )
2746
+ if state.get("request_digest") != request_digest:
2747
+ candidates = [
2748
+ *sorted(
2749
+ _IDENTITY_INPUT_LABELS[name]
2750
+ for name in _IDENTITY_INPUT_LABELS
2751
+ if name not in journaled and name != "workspace_id"
2752
+ ),
2753
+ *_UNNAMED_REQUEST_INPUTS,
2754
+ ]
2755
+ raise HostedDeployError(
2756
+ "DEPLOY_STATE_MISMATCH",
2757
+ "this deployment journal was staged from a different request, and every input it "
2758
+ f"records still matches, so what moved is one of: {', '.join(candidates)} -- rerun it "
2759
+ "the way it was staged, or start a new deployment",
2760
+ )
2761
+ raise HostedDeployError(
2762
+ "DEPLOY_STATE_MISMATCH", "the deployment journal belongs to another exact handoff"
2763
+ )
2764
+
2765
+
2766
+ def _refuse_legacy_approval_journal(state: Mapping[str, Any]) -> None:
2767
+ """Refuse every old human-gate pause rather than inventing a direct confirmation.
2768
+
2769
+ The direct Editor flow never writes ``approval_required``. Even an old journal that happens to
2770
+ carry a complete approval-request response and ETag has no journaled Editor confirmation, so
2771
+ that status alone is sufficient to identify a handoff that cannot safely cross into activation.
2772
+ """
2773
+
2774
+ if state.get("status") != "approval_required":
2775
+ return
2776
+ raise HostedDeployError(
2777
+ "DEPLOY_LEGACY_APPROVAL_JOURNAL",
2778
+ "this deployment journal was paused by the former human-approval flow and has no "
2779
+ "journaled Editor confirmation; it cannot be activated safely, so start a new deployment "
2780
+ "from a fresh verified Build directory",
2781
+ )
2782
+
2783
+
2784
+ def _resource_etag(state: Mapping[str, Any], name: str) -> str:
2785
+ value = state.get("etags", {}).get(name) if isinstance(state.get("etags"), dict) else None
2786
+ if not isinstance(value, str) or not value.startswith('"') or not value.endswith('"'):
2787
+ raise HostedDeployError(
2788
+ "DEPLOY_STUDIO_RESPONSE_INVALID", f"Studio omitted the ETag after {name}"
2789
+ )
2790
+ return value
2791
+
2792
+
2793
+ def _generated_response(response: Any) -> StudioResponse:
2794
+ parsed = getattr(response, "parsed", None)
2795
+ body = parsed.to_dict() if parsed is not None and hasattr(parsed, "to_dict") else {}
2796
+ headers = getattr(response, "headers", {})
2797
+ return StudioResponse(
2798
+ status_code=int(response.status_code),
2799
+ body=body,
2800
+ etag=headers.get("ETag") if hasattr(headers, "get") else None,
2801
+ )
2802
+
2803
+
2804
+ def _success(response: StudioResponse, expected: set[int], code: str) -> Mapping[str, Any]:
2805
+ if response.status_code not in expected:
2806
+ api_code = response.body.get("code")
2807
+ suffix = f" ({api_code})" if isinstance(api_code, str) else ""
2808
+ raise HostedDeployError(
2809
+ code, f"Studio refused the operation (HTTP {response.status_code}){suffix}"
2810
+ )
2811
+ return response.body
2812
+
2813
+
2814
+ def _etag(response: StudioResponse, operation: str) -> str:
2815
+ value = response.etag
2816
+ if not isinstance(value, str) or not value.startswith('"') or not value.endswith('"'):
2817
+ raise HostedDeployError(
2818
+ "DEPLOY_STUDIO_RESPONSE_INVALID", f"Studio omitted the ETag after {operation}"
2819
+ )
2820
+ return value
2821
+
2822
+
2823
+ def _idempotency(operation: str, identity: str) -> str:
2824
+ digest = hashlib.sha256(f"{operation}:{identity}".encode()).hexdigest()
2825
+ return f"mr-data.deploy.v3:{digest}"
2826
+
2827
+
2828
+ def _same(value: Mapping[str, Any], field: str, expected: Any) -> None:
2829
+ if value.get(field) != expected:
2830
+ raise HostedDeployError(
2831
+ "DEPLOY_STATE_MISMATCH", f"{field} differs from the authenticated deployment scope"
2832
+ )
2833
+
2834
+
2835
+ def _required_text(value: Mapping[str, Any], field: str) -> str:
2836
+ selected = value.get(field)
2837
+ if not isinstance(selected, str) or not selected:
2838
+ raise HostedDeployError("DEPLOY_RESPONSE_INVALID", f"the response omitted {field}")
2839
+ return selected
2840
+
2841
+
2842
+ def _required_uuid(value: UUID | None, field: str) -> UUID:
2843
+ if value is None:
2844
+ raise HostedDeployError("DEPLOY_REQUEST_INVALID", f"{field} is missing")
2845
+ return value
2846
+
2847
+
2848
+ def _uuid(value: Any, field: str) -> UUID:
2849
+ try:
2850
+ parsed = UUID(value) if isinstance(value, str) else None
2851
+ except ValueError:
2852
+ parsed = None
2853
+ if parsed is None or str(parsed) != value:
2854
+ raise HostedDeployError("DEPLOY_RESPONSE_INVALID", f"{field} is not a canonical UUID")
2855
+ return parsed
2856
+
2857
+
2858
+ def _uuid_field(value: Mapping[str, Any], field: str) -> UUID:
2859
+ try:
2860
+ return _uuid(value.get(field), field)
2861
+ except HostedDeployError as error:
2862
+ raise HostedDeployError(
2863
+ "DEPLOY_STUDIO_RESPONSE_INVALID", f"Studio omitted a valid {field}"
2864
+ ) from error
2865
+
2866
+
2867
+ def _positive_int_field(value: Mapping[str, Any], field: str) -> int:
2868
+ selected = value.get(field)
2869
+ if isinstance(selected, bool) or not isinstance(selected, int) or selected < 1:
2870
+ raise HostedDeployError(
2871
+ "DEPLOY_STUDIO_RESPONSE_INVALID", f"Studio omitted a positive {field}"
2872
+ )
2873
+ return selected
2874
+
2875
+
2876
+ def _timestamp(value: str, field: str) -> datetime:
2877
+ try:
2878
+ parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
2879
+ except ValueError as error:
2880
+ raise HostedDeployError("DEPLOY_RESPONSE_INVALID", f"{field} is not a timestamp") from error
2881
+ if parsed.tzinfo is None or parsed.utcoffset() != UTC.utcoffset(parsed):
2882
+ raise HostedDeployError("DEPLOY_RESPONSE_INVALID", f"{field} is not UTC")
2883
+ return parsed
2884
+
2885
+
2886
+ def _deployment_options(dataset_name: Any, cron: Any, maximum: Any) -> None:
2887
+ if (
2888
+ not isinstance(dataset_name, str)
2889
+ or not 1 <= len(dataset_name) <= 200
2890
+ or dataset_name.strip() != dataset_name
2891
+ or not dataset_name
2892
+ ):
2893
+ raise HostedDeployError(
2894
+ "DEPLOY_REQUEST_INVALID", "dataset name must be 1-200 trimmed nonblank characters"
2895
+ )
2896
+ if (
2897
+ not isinstance(cron, str)
2898
+ or not 9 <= len(cron) <= 100
2899
+ or cron.strip() != cron
2900
+ or len(cron.split()) != 5
2901
+ or "\n" in cron
2902
+ or "\r" in cron
2903
+ ):
2904
+ raise HostedDeployError(
2905
+ "DEPLOY_REQUEST_INVALID", "cron must be one satisfiable five-field UTC expression"
2906
+ )
2907
+ try:
2908
+ croniter(cron, datetime(2000, 1, 1, tzinfo=UTC)).get_next(datetime)
2909
+ except (CroniterBadCronError, CroniterBadDateError, ValueError) as error:
2910
+ raise HostedDeployError(
2911
+ "DEPLOY_REQUEST_INVALID", "cron must be one satisfiable five-field UTC expression"
2912
+ ) from error
2913
+ if isinstance(maximum, bool) or not isinstance(maximum, int) or not 1 <= maximum <= 10:
2914
+ raise HostedDeployError(
2915
+ "DEPLOY_REQUEST_INVALID", "max concurrent runs must be between 1 and 10"
2916
+ )
2917
+
2918
+
2919
+ def _service_url(value: str, label: str, *, allow_loopback_http: bool) -> str:
2920
+ if not isinstance(value, str) or not 1 <= len(value) <= 2048 or value.endswith("/"):
2921
+ raise HostedDeployError("DEPLOY_CONFIG_INVALID", f"{label} is not a usable origin")
2922
+ parsed = urlsplit(value)
2923
+ loopback = False
2924
+ if parsed.hostname:
2925
+ try:
2926
+ loopback = ipaddress.ip_address(parsed.hostname).is_loopback
2927
+ except ValueError:
2928
+ loopback = parsed.hostname == "localhost"
2929
+ if (
2930
+ parsed.scheme not in ({"https", "http"} if allow_loopback_http else {"https"})
2931
+ or (parsed.scheme == "http" and not loopback)
2932
+ or not parsed.hostname
2933
+ or parsed.username is not None
2934
+ or parsed.password is not None
2935
+ or parsed.fragment
2936
+ or parsed.path not in {"", "/"}
2937
+ or parsed.query
2938
+ ):
2939
+ raise HostedDeployError(
2940
+ "DEPLOY_CONFIG_INVALID",
2941
+ f"{label} must be an HTTPS origin without user information, path, query, or fragment",
2942
+ )
2943
+ return value
2944
+
2945
+
2946
+ def _require_cli_credential(credentials: ResolvedCloudCredentials) -> None:
2947
+ if _CLI_KEY.fullmatch(credentials.raw_key) is None:
2948
+ raise HostedDeployError(
2949
+ "DEPLOY_AUTHENTICATION_FAILED",
2950
+ "managed deployment requires an mr_cli_ credential; "
2951
+ "a live data key has read-only scope",
2952
+ )
2953
+
2954
+
2955
+ def _transfer_url(url: str, base_url: str) -> str:
2956
+ parsed = urlsplit(url)
2957
+ base = urlsplit(base_url)
2958
+ if (
2959
+ parsed.scheme not in {"https", "http"}
2960
+ or not parsed.hostname
2961
+ or parsed.username is not None
2962
+ or parsed.password is not None
2963
+ or parsed.fragment
2964
+ ):
2965
+ raise HostedDeployError("DEPLOY_SIGNED_SESSION_INVALID", "signed upload URL is invalid")
2966
+ if parsed.scheme == "http" and (
2967
+ base.scheme != "http" or parsed.hostname != base.hostname or parsed.port != base.port
2968
+ ):
2969
+ raise HostedDeployError(
2970
+ "DEPLOY_SIGNED_SESSION_INVALID",
2971
+ "insecure signed upload URL differs from the local Studio origin",
2972
+ )
2973
+ return url
2974
+
2975
+
2976
+ def _signed_headers(session: Mapping[str, Any], *, size: int) -> dict[str, str]:
2977
+ result: dict[str, str] = {}
2978
+ required = session.get("required_headers")
2979
+ if not isinstance(required, list):
2980
+ raise HostedDeployError("DEPLOY_SIGNED_SESSION_INVALID", "signed upload headers are absent")
2981
+ for item in required:
2982
+ if not isinstance(item, dict):
2983
+ raise HostedDeployError(
2984
+ "DEPLOY_SIGNED_SESSION_INVALID", "signed upload header is invalid"
2985
+ )
2986
+ name = item.get("name")
2987
+ value = item.get("value")
2988
+ if (
2989
+ not isinstance(name, str)
2990
+ or not isinstance(value, str)
2991
+ or name.lower() in {"authorization", "cookie", "host", "proxy-authorization"}
2992
+ or re.fullmatch(r"[A-Za-z0-9-]{1,128}", name) is None
2993
+ or "\r" in value
2994
+ or "\n" in value
2995
+ or name.lower() in result
2996
+ ):
2997
+ raise HostedDeployError(
2998
+ "DEPLOY_SIGNED_SESSION_INVALID", "signed upload header is unsafe"
2999
+ )
3000
+ result[name.lower()] = value
3001
+ media_type = _required_text(session, "media_type")
3002
+ if result.get("content-type", media_type) != media_type:
3003
+ raise HostedDeployError(
3004
+ "DEPLOY_SIGNED_SESSION_INVALID", "signed upload content type differs"
3005
+ )
3006
+ result["content-type"] = media_type
3007
+ result["content-length"] = str(size)
3008
+ return result
3009
+
3010
+
3011
+ def _file_identity(value: os.stat_result) -> tuple[int, int, int, int, int, int, int]:
3012
+ return (
3013
+ value.st_dev,
3014
+ value.st_ino,
3015
+ value.st_mode,
3016
+ value.st_nlink,
3017
+ value.st_size,
3018
+ value.st_mtime_ns,
3019
+ value.st_ctime_ns,
3020
+ )
3021
+
3022
+
3023
+ __all__ = [
3024
+ "DEPLOYMENT_RECEIPT_SCHEMA",
3025
+ "DEPLOYMENT_STATE_SCHEMA",
3026
+ "FileDeploymentStateStore",
3027
+ "GeneratedStudioDeploymentClient",
3028
+ "HostedDeployError",
3029
+ "StudioDeploymentClient",
3030
+ "StudioResponse",
3031
+ "StudioToken",
3032
+ "TokenExchangeTransport",
3033
+ "build_editor_client",
3034
+ "deploy_managed_dataset",
3035
+ "exchange_studio_token",
3036
+ "journaled_reviewed_build",
3037
+ ]