mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,2759 @@
1
+ """``mr-data propose``: the hosted authoring front door ADR 0021 left open.
2
+
3
+ WHAT WAS MISSING. ADR 0021 makes the product hosted-only and the client thin, but the only thing
4
+ that created a recipe proposal was the LOCAL lane -- ``mr-data deploy`` from a sealed local Build,
5
+ through :mod:`mostlyright.data_harness.hosted_deploy`, which imports pyarrow, croniter and the
6
+ generated client and is forbidden in the thin profile. ``mr-data build --recipe-proposal ID
7
+ --recipe-digest D`` therefore needed a proposal that nothing thin could make. This command makes
8
+ one, over the same V3 routes ``hosted_deploy`` drives, in pure Python and urllib only.
9
+
10
+ THE DOCUMENTS AN AGENT AUTHORS. A bundle directory holds the documents the hosted API needs, each
11
+ in the exact shape ``docs/HOSTED-AUTHORING.md`` records and each checked here, against that shape,
12
+ BEFORE a credential is minted or a request is sent:
13
+
14
+ * ``dataset.json`` -> ``POST /v3/datasets``
15
+ * ``question.json`` -> ``POST /v3/questions`` then ``PUT /v3/questions/{id}/requirements``
16
+ * ``sources.json`` (a list) -> one ``POST /v3/sources`` and one
17
+ ``POST /v3/connector-configurations`` per entry
18
+ * ``table_plan.json`` -> ``POST /v3/table-plans``
19
+ * ``recipe.json`` -> the recipe document itself, whose coordinates become the proposal's and whose
20
+ canonical bytes are the ``recipe_digest``
21
+ * ``activation.json`` -> the ``activation_intent`` sealed into the proposal
22
+
23
+ Each of the first four may instead ADOPT a resource that already exists, by naming its id
24
+ (``{"dataset_id": "..."}``), so a second bundle can attach to what a first one registered.
25
+
26
+ WHY ``--through sources`` EXISTS. A recipe names its sources by the Studio source id, which does
27
+ not exist until the source is registered. So an agent runs ``propose --through sources`` FIRST to
28
+ register the Dataset, Question and Sources, opens a research session against them
29
+ (``mr-data research-open --bundle``), probes, and only then writes ``recipe.json`` referencing the
30
+ registered ids and re-runs ``propose`` to finish. The journal is what carries the registered ids
31
+ across the two runs.
32
+
33
+ WHAT THE JOURNAL PINS, AND WHEN. A document's digest is pinned the moment the first resource built
34
+ from it completes -- ``dataset.json`` when the Dataset exists, ``question.json`` when the
35
+ requirements are written, ``sources.json`` when the last connector is registered,
36
+ ``table_plan.json`` when the plan exists, ``recipe.json`` and ``activation.json`` when the
37
+ proposal exists. Until then a
38
+ document is free to change: a refused document is corrected and re-run, not abandoned. After it,
39
+ a changed document is refused with the document named, because the resource Studio holds was built
40
+ from the earlier one and a proposal a person is asked to approve must be the proposal that was
41
+ authored.
42
+
43
+ WHAT THIS COMMAND NEVER DOES. It records an approval ASK, exactly as ``mr-data approve`` does; it
44
+ never grants one. A Table recipe is confirmed from a signed-in editor session in the dashboard,
45
+ which a Cloud-minted service token cannot be, so the final receipt names the proposal, the recipe
46
+ digest, the approval request, the dashboard address a person opens, and the two commands that
47
+ follow -- ``mr-data recipe-approve`` (the ask a person settles) and then ``mr-data build``. It does
48
+ NOT validate the transform rules inside the recipe: that is the local lane's job and imports the
49
+ modules this profile forbids. Studio validates what Studio validates, and the hosted worker refuses
50
+ a recipe its engine cannot run.
51
+ """
52
+
53
+ from __future__ import annotations
54
+
55
+ import argparse
56
+ import hashlib
57
+ import json
58
+ import os
59
+ import re
60
+ import secrets
61
+ import stat
62
+ from collections.abc import Callable, Mapping, Sequence
63
+ from pathlib import Path
64
+ from typing import Any
65
+ from uuid import UUID
66
+
67
+ from mostlyright.data_harness import package_version
68
+ from mostlyright.data_harness.canonical import (
69
+ CanonicalJSONError,
70
+ canonical_json_bytes,
71
+ canonical_sha256,
72
+ parse_json,
73
+ )
74
+ from mostlyright.data_harness.formats import PARSER_FORMAT_ORDER
75
+
76
+ # The externally enforced public-crawler egress policies, as their canonical attestations. Reused
77
+ # rather than recomputed here for the reason the acquire lane reuses `session_probes`: a second
78
+ # copy of a security-critical constant is a second thing to keep in step, and this one is the
79
+ # digest Studio's own `STUDIO_PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION` is set to at deploy time.
80
+ from mostlyright.data_harness.hosted_crawler_protocol import (
81
+ OPENLIGADB_EGRESS_POLICY_ATTESTATION,
82
+ PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION,
83
+ )
84
+ from mostlyright.data_harness.thin import THIN_SCHEMA_PREFIX
85
+ from mostlyright.data_harness.thin.approvals import GET_APPROVAL_PATH
86
+ from mostlyright.data_harness.thin.research import GET_SESSION_PATH, RetryAfterRefusal
87
+ from mostlyright.data_harness.thin.runs import (
88
+ GET_RECIPE_PROPOSAL_PATH,
89
+ StudioApiClient,
90
+ )
91
+ from mostlyright.data_harness.thin.session import (
92
+ StudioSession,
93
+ approval_dashboard_url,
94
+ dataset_dashboard_url,
95
+ open_studio_session,
96
+ run_dashboard_url,
97
+ )
98
+ from mostlyright.data_harness.thin.transport import ThinLaneError
99
+ from mostlyright.data_harness.ux.credentials import resolve_cloud_credentials
100
+ from mostlyright.data_harness.ux.path_kind import UNKNOWN_KIND, presence_at
101
+ from mostlyright.data_harness.ux.plain_file import PlainFileRefusal, read_plain_file
102
+
103
+ PROPOSE_SCHEMA = f"{THIN_SCHEMA_PREFIX}-propose.v1"
104
+ JOURNAL_SCHEMA = f"{THIN_SCHEMA_PREFIX}-propose-journal.v1"
105
+
106
+ #: The V3 wire this lane speaks. The creates are each bound to an idempotency key, the one PUT
107
+ #: upserts requirements, and the GETs either adopt a resource that already exists or read the
108
+ #: proposal back for the ETag its approval is pinned to. None of them approves, activates,
109
+ #: releases or decides. They are the exact routes :mod:`mostlyright.data_harness.hosted_deploy`
110
+ #: drives, written out so a reviewer diffs them against ``contracts/openapi/studio-v3.yaml``.
111
+ CREATE_DATASET_PATH = "/v3/datasets"
112
+ GET_DATASET_PATH = "/v3/datasets/{dataset_id}"
113
+ CREATE_QUESTION_PATH = "/v3/questions"
114
+ GET_QUESTION_PATH = "/v3/questions/{question_id}"
115
+ PUT_REQUIREMENTS_PATH = "/v3/questions/{question_id}/requirements"
116
+ GET_REQUIREMENTS_PATH = "/v3/questions/{question_id}/requirements"
117
+ REGISTER_SOURCE_PATH = "/v3/sources"
118
+ GET_SOURCE_PATH = "/v3/sources/{source_id}"
119
+ REGISTER_CONNECTOR_PATH = "/v3/connector-configurations"
120
+ GET_CONNECTOR_PATH = "/v3/connector-configurations/{connector_configuration_id}"
121
+ CREATE_TABLE_PLAN_PATH = "/v3/table-plans"
122
+ GET_TABLE_PLAN_PATH = "/v3/table-plans/{table_plan_id}"
123
+ WORKER_POLICY_PREFLIGHT_PATH = "/v3/worker-policy-preflights"
124
+ CREATE_RECIPE_PROPOSAL_PATH = "/v3/recipe-proposals"
125
+ #: ``createRecipeProposal`` answers 201 with NO ETag -- Studio declares no ``etag_kind`` on that
126
+ #: route, and an idempotent replay never carries one either -- so the proposal is read back,
127
+ #: where the ETag is, before the approval is pinned to it (``runs.GET_RECIPE_PROPOSAL_PATH``, the
128
+ #: read ``build`` shares). A real run against Studio is what found this: a fake that invented an
129
+ #: ETag on the create kept every test green over a lane that could never open its approval.
130
+ REQUEST_RECIPE_APPROVAL_PATH = "/v3/recipe-proposals/{recipe_proposal_id}/approval"
131
+
132
+ #: The V3 contract version every command body carries.
133
+ STUDIO_SCHEMA_VERSION = "3.0.0"
134
+
135
+ #: The media type a recipe travels under, and the recipe schema versions Studio's
136
+ #: ``recipe_schema_version`` enum admits. Held here so a document naming a version Studio would
137
+ #: refuse is a sentence at the terminal rather than a 422 three routes in.
138
+ RECIPE_MEDIA_TYPE = "application/vnd.mostlyright.frozen-recipe+json"
139
+ RECIPE_SCHEMA_VERSIONS = ("frozen-recipe.v1", "frozen-recipe.v3")
140
+
141
+ #: The one bootstrap origin this lane sends. There is no local Build behind a hosted-first Recipe,
142
+ #: so the first hosted Build is the bootstrap and the in-product approval plus the hosted
143
+ #: independent check are the gates. Studio admits the hosted shape only on a first Recipe
144
+ #: (``recipe_version`` 1 with a null predecessor), which this lane refuses locally before it can
145
+ #: become a 422.
146
+ HOSTED_BOOTSTRAP_ORIGIN = "hosted_first_build"
147
+
148
+ #: The two credential-free public connector adapters, and the egress attestation each one carries.
149
+ #: A source outside this map is a credentialed or sealed-artifact source, which the hosted
150
+ #: authoring front door does not register -- those enter through the local lane's staged uploads.
151
+ CONNECTOR_EGRESS_ATTESTATION = {
152
+ "public.https": PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION,
153
+ "external.openligadb": OPENLIGADB_EGRESS_POLICY_ATTESTATION,
154
+ }
155
+
156
+ #: The stages ``--through`` may stop after, in the order a bundle is built. ``approval`` is the
157
+ #: whole run and the default.
158
+ STAGES = ("dataset", "question", "sources", "plan", "proposal", "approval")
159
+
160
+ #: The bundle documents, and the stage each is first read for. A document is refused as missing
161
+ #: only when a stage that needs it is reached, so ``--through sources`` never asks for a
162
+ #: ``recipe.json`` that has not been written yet. ``recipe.json`` is read for the plan stage as
163
+ #: well as the proposal, because the plan's digests describe the recipe.
164
+ DOCUMENTS = {
165
+ "dataset": ("dataset.json",),
166
+ "question": ("question.json",),
167
+ "sources": ("sources.json",),
168
+ "plan": ("table_plan.json", "recipe.json"),
169
+ "proposal": ("activation.json",),
170
+ }
171
+
172
+ #: Which documents each stage pins when it completes. A pin is the statement "a resource Studio
173
+ #: holds was built from this exact document"; a stage that has not completed has made no such
174
+ #: statement, and its documents stay editable.
175
+ PINS = {
176
+ "dataset": ("dataset.json",),
177
+ "question": ("question.json",),
178
+ "sources": ("sources.json",),
179
+ "plan": ("table_plan.json",),
180
+ "proposal": ("recipe.json", "activation.json"),
181
+ }
182
+
183
+ #: A bundle document is read whole under the same open-then-fstat rule every named path is read
184
+ #: under, and one four times the canonical recipe ceiling is refused before it is parsed.
185
+ MAX_DOCUMENT_BYTES = 4 * 262_144
186
+
187
+ #: The journal embeds every resource Studio answered with -- the proposal, with its canonical
188
+ #: recipe bytes, among them -- so its ceiling is above a single document's.
189
+ MAX_JOURNAL_BYTES = 4 * MAX_DOCUMENT_BYTES
190
+
191
+ #: ``canonical_recipe_json`` is bounded by the contract at 262144 UTF-8 bytes. Checked here so a
192
+ #: recipe that canonicalises larger is a sentence rather than a 422.
193
+ MAX_CANONICAL_RECIPE_BYTES = 262_144
194
+
195
+ #: Studio's ``Retry-After`` is delta-seconds. Bounded the way the research lane bounds its own: a
196
+ #: window this lane does not understand is reported as no window rather than as a number nobody
197
+ #: checked.
198
+ MAX_RETRY_AFTER_SECONDS = 3_600
199
+
200
+ #: The journal directory and file, created 0700 and 0600 so a bundle checked out on a shared
201
+ #: machine does not leave a workspace's created-resource ids world-readable.
202
+ JOURNAL_DIR = ".mr-data"
203
+ JOURNAL_FILE = "propose.journal.json"
204
+
205
+ #: The schedule a bundle gets when ``activation.json`` leaves one out: daily at midnight UTC, one
206
+ #: run at a time -- the same defaults ``mr-data deploy`` applies.
207
+ DEFAULT_SCHEDULE = {
208
+ "cron": "0 0 * * *",
209
+ "timezone": "UTC",
210
+ "mode": "incremental_refresh",
211
+ "max_concurrent_runs": 1,
212
+ }
213
+
214
+ # The structural rules Studio's JSON Schema files declare for the bodies this lane sends, restated
215
+ # here so a document is refused at the terminal in the words of the document rather than three
216
+ # routes in with a JSON pointer. Every enum and pattern is the contract's own, and the cross-repo
217
+ # gate in `tests/test_thin_propose.py` holds every body built from them against the schema files.
218
+ _HARNESS_VERSION = re.compile(r"^[0-9]+\.[0-9]+\.[0-9]+$")
219
+ _UUID_TEXT = re.compile(
220
+ r"^[0-9a-f]{8}-[0-9a-f]{4}-[1-8][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$"
221
+ )
222
+ _PREFIXED_DIGEST = re.compile(r"^sha256:[0-9a-f]{64}$")
223
+ _BARE_DIGEST = re.compile(r"^[0-9a-f]{64}$")
224
+ _IDENTIFIER = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.-]{0,127}$")
225
+ _COLUMN = re.compile(r"^[a-z][a-z0-9_]{0,62}$")
226
+ _SEMVER = re.compile(r"^[1-9][0-9]*\.[0-9]+\.[0-9]+$")
227
+ _TIMESTAMP = re.compile(
228
+ r"^(?:[0-9]{4})-(?:0[1-9]|1[0-2])-(?:0[1-9]|[12][0-9]|3[01])"
229
+ r"T(?:[01][0-9]|2[0-3]):[0-5][0-9]:[0-5][0-9](?:\.[0-9]{1,9})?Z$"
230
+ )
231
+ _HTTPS_URL = re.compile(r"^https://[^:/?#@]+(?:(?:/[^#]*)|(?:\?[^#]*))?$")
232
+ _FILENAME = re.compile(r"^[^/\\\x00-\x1f\x7f]+$")
233
+
234
+ _SOURCE_CLASSES = (
235
+ "mostlyright_sdk",
236
+ "external_adapter",
237
+ "user_file",
238
+ "user_url",
239
+ "user_api",
240
+ "database_extract",
241
+ "webhook",
242
+ "stream",
243
+ )
244
+ _LOCATOR_KINDS = ("sdk_connector", "https_url", "artifact", "webhook", "stream", "database")
245
+ _CLASSIFICATIONS = ("public", "internal", "confidential", "restricted")
246
+ _CLAIMED_BASES = (
247
+ "unknown",
248
+ "prohibited",
249
+ "permission_asserted",
250
+ "public_domain_asserted",
251
+ "contractual_license_asserted",
252
+ "terms_of_service_asserted",
253
+ )
254
+ _TARGET_POLICIES = ("none", "required_if_supportable")
255
+ _FEASIBILITY_DECISIONS = ("supportable", "supportable_with_limits", "unsupported")
256
+ _REASON_CODES = (
257
+ "SUFFICIENT_EVIDENCE",
258
+ "LIMITED_COVERAGE",
259
+ "UNACCEPTABLE_DELAY",
260
+ "RIGHTS_UNCLEAR",
261
+ "NO_RELIABLE_SOURCE",
262
+ "TARGET_NOT_OBSERVABLE",
263
+ )
264
+ _OVERLAP_POLICIES = ("disallow_released_coverage", "allow_within_envelope")
265
+ _BACKFILL_POLICY_SCHEMA = "local-backfill-policy.v1"
266
+ _QUERY_LIMIT_CEILINGS = {
267
+ "max_source_bytes": 268_435_456,
268
+ "max_normalized_bytes": 4_294_967_296,
269
+ "max_rows": 10_000_000,
270
+ "max_columns": 1_024,
271
+ }
272
+
273
+
274
+ class ProposeError(ThinLaneError):
275
+ """A typed refusal from the authoring front door, carrying an operator action.
276
+
277
+ Distinct from a plain :class:`ThinLaneError` only in that it always names what to do next --
278
+ the document to fix, or the command to run first -- so a refusal an agent reads is actionable
279
+ rather than merely typed.
280
+ """
281
+
282
+ def __init__(self, code: str, detail: str) -> None:
283
+ super().__init__(code, detail)
284
+
285
+
286
+ # --------------------------------------------------------------------------------------------
287
+ # The client
288
+ # --------------------------------------------------------------------------------------------
289
+
290
+
291
+ class StudioProposeClient(StudioApiClient):
292
+ """The V3 authoring calls: creates under an idempotency key, and reads. Nothing decides.
293
+
294
+ ``request_recipe_approval`` is the one that opens an approval, and it opens an ASK: Studio
295
+ writes one immutable request row and never a decision, which is why this lane can create the
296
+ thing ``mr-data recipe-approve`` later points a human at without ever being able to grant it.
297
+ """
298
+
299
+ def create(
300
+ self,
301
+ path: str,
302
+ body: Mapping[str, Any],
303
+ *,
304
+ idempotency_key: str,
305
+ expected: int,
306
+ method: str = "POST",
307
+ if_match: str | None = None,
308
+ response_headers: dict[str, str] | None = None,
309
+ ) -> dict[str, Any]:
310
+ headers = {"Idempotency-Key": idempotency_key}
311
+ if if_match is not None:
312
+ headers["If-Match"] = if_match
313
+ captured: dict[str, str] = {}
314
+ try:
315
+ result = self._call(
316
+ method,
317
+ path,
318
+ body=body,
319
+ extra_headers=headers,
320
+ expected=(expected,),
321
+ response_headers=captured,
322
+ )
323
+ except ThinLaneError as refusal:
324
+ raise _maybe_retry_after(refusal, captured) from refusal
325
+ if response_headers is not None:
326
+ response_headers.update(captured)
327
+ return result
328
+
329
+ def read(self, path: str, *, response_headers: dict[str, str] | None = None) -> dict[str, Any]:
330
+ """One bounded GET of one Studio resource, with the ETag it answered under."""
331
+
332
+ captured: dict[str, str] = {}
333
+ try:
334
+ result = self._call("GET", path, expected=(200,), response_headers=captured)
335
+ except ThinLaneError as refusal:
336
+ raise _maybe_retry_after(refusal, captured) from refusal
337
+ if response_headers is not None:
338
+ response_headers.update(captured)
339
+ return result
340
+
341
+ def worker_policy_preflight(self, body: Mapping[str, Any]) -> dict[str, Any]:
342
+ """Studio's current worker selection for this proposal, read without journaling a write.
343
+
344
+ The preflight is not a mutation: it resolves the deployed fleet's images and validation
345
+ policy for the recipe's mode, and the approval request below carries it back so Studio can
346
+ re-check it against the same immutable TablePlan. This lane never re-derives that policy --
347
+ that needs the engine this profile does not carry -- so it passes Studio's own answer
348
+ through, and Studio re-checks it at approval and at build.
349
+ """
350
+
351
+ captured: dict[str, str] = {}
352
+ try:
353
+ return self._call(
354
+ "POST",
355
+ WORKER_POLICY_PREFLIGHT_PATH,
356
+ body=body,
357
+ expected=(200,),
358
+ response_headers=captured,
359
+ )
360
+ except ThinLaneError as refusal:
361
+ raise _maybe_retry_after(refusal, captured) from refusal
362
+
363
+
364
+ def _maybe_retry_after(refusal: ThinLaneError, headers: Mapping[str, str]) -> ThinLaneError:
365
+ """Carry a ``Retry-After`` window on a refusal Studio stamped one on, never sleep through it.
366
+
367
+ A workspace Studio is rate-limiting answers with a window rather than a queue, and the honest
368
+ thing is to report the window and let the caller decide, the way the research lane reports a
369
+ full-workspace refusal. Any refusal that arrives with a usable ``Retry-After`` becomes a
370
+ :class:`RetryAfterRefusal` carrying the number the router already knows how to surface.
371
+ """
372
+
373
+ raw = headers.get("retry-after", "")
374
+ if not raw.isdecimal():
375
+ return refusal
376
+ window = int(raw)
377
+ if not 1 <= window <= MAX_RETRY_AFTER_SECONDS:
378
+ return refusal
379
+ windowed = RetryAfterRefusal(
380
+ refusal.code,
381
+ f"{refusal.detail}; Studio asked this command to retry after {window} seconds. "
382
+ "Nothing further was created, and this command did not wait",
383
+ retry_after_seconds=window,
384
+ )
385
+ # The facts about the answer ride along, so the applied-or-not question below can still be
386
+ # asked of a windowed refusal.
387
+ windowed.http_status = getattr(refusal, "http_status", None) # type: ignore[attr-defined]
388
+ windowed.studio_code = getattr(refusal, "studio_code", None) # type: ignore[attr-defined]
389
+ return windowed
390
+
391
+
392
+ #: The statuses under which Studio answered with a problem document BEFORE any write: the request
393
+ #: failed validation, named a resource that is not there, asserted a stale version, or carried no
394
+ #: authority for the route. Every one of them is raised before ``v3_idempotent`` reserves a key.
395
+ _REFUSED_BEFORE_WRITE = frozenset({401, 403, 404, 412, 422})
396
+
397
+
398
+ def _nothing_was_applied(refusal: ThinLaneError) -> bool:
399
+ """Whether a refusal means Studio answered and applied nothing -- decided narrowly.
400
+
401
+ Only a problem document from Studio under a validation or conflict status says that: 401, 403,
402
+ 404, 412 and 422, and a 409 whose code is not one of the idempotency codes. Everything else
403
+ keeps the plan. A 5xx may be a gateway answering after Studio applied the write; an answer
404
+ this lane could not read as a problem document says nothing about what happened behind it; and
405
+ the two ``IDEMPOTENCY_*`` 409s are Studio saying the key IS reserved -- in progress, or
406
+ reserved without its record because ``v3_idempotent`` writes the resource and completes the
407
+ record in two transactions -- so the write may well have landed. For all of those the only
408
+ honest resume is the identical request under the identical key, which Studio replays.
409
+ """
410
+
411
+ status = getattr(refusal, "http_status", None)
412
+ studio_code = getattr(refusal, "studio_code", None)
413
+ if not isinstance(status, int) or not isinstance(studio_code, str):
414
+ return False
415
+ if status in _REFUSED_BEFORE_WRITE:
416
+ return True
417
+ return status == 409 and not studio_code.startswith("IDEMPOTENCY_")
418
+
419
+
420
+ # --------------------------------------------------------------------------------------------
421
+ # The bundle, read and checked before anything is minted
422
+ # --------------------------------------------------------------------------------------------
423
+
424
+
425
+ def _invalid(name: str, path: str, what: str) -> ProposeError:
426
+ where = f"{name} {path}" if path else name
427
+ return ProposeError("THIN_PROPOSE_DOCUMENT_INVALID", f"{where} {what}")
428
+
429
+
430
+ def _bundle_directory(value: Any) -> Path:
431
+ """The bundle directory, refused before a credential is resolved if it is not one.
432
+
433
+ The refusals name what the open actually observed rather than asserting a kind: ``presence_at``
434
+ reads the mode the look found, so "there is nothing there" and "there is a file there" are
435
+ answers to a look, not guesses -- the same rule ``thin.acquire`` reads a named path under, and
436
+ the one ``scripts/path_kind_gate.py`` holds every surface to.
437
+ """
438
+
439
+ if not isinstance(value, str) or not value:
440
+ raise ProposeError("THIN_PROPOSE_BUNDLE_INVALID", "propose names one bundle directory")
441
+ path = Path(value).expanduser()
442
+ try:
443
+ info = path.lstat()
444
+ except OSError:
445
+ raise ProposeError(
446
+ "THIN_PROPOSE_BUNDLE_INVALID",
447
+ f"there is {presence_at(path) or UNKNOWN_KIND} at {path}: propose reads its documents "
448
+ "out of a directory",
449
+ ) from None
450
+ if stat.S_ISLNK(info.st_mode) or not path.is_dir():
451
+ raise ProposeError(
452
+ "THIN_PROPOSE_BUNDLE_INVALID",
453
+ f"there is {presence_at(path) or UNKNOWN_KIND} at {path}, not the directory propose "
454
+ "reads its documents out of",
455
+ )
456
+ return path
457
+
458
+
459
+ def _read_document(bundle: Path, name: str) -> tuple[Any, str]:
460
+ """One bundle document and the digest of its bytes, parsed as STRICT JSON.
461
+
462
+ Strict, not foreign: every number in a bundle reaches a canonical digest or a Studio body that
463
+ is digested there, and the Harness data model admits no fractional number anywhere on that
464
+ path. A float is refused here, by document and position, rather than three stages in as a
465
+ canonicalisation error out of a request body.
466
+ """
467
+
468
+ target = bundle / name
469
+ try:
470
+ raw = read_plain_file(target, max_bytes=MAX_DOCUMENT_BYTES)
471
+ except PlainFileRefusal as refusal:
472
+ raise ProposeError(
473
+ "THIN_PROPOSE_DOCUMENT_UNREADABLE",
474
+ f"{name} could not be read from the bundle ({refusal.reason}); author it at {target}",
475
+ ) from None
476
+ except OSError as error:
477
+ raise ProposeError(
478
+ "THIN_PROPOSE_DOCUMENT_UNREADABLE", f"{name} could not be read: {error}"
479
+ ) from error
480
+ try:
481
+ parsed = parse_json(raw)
482
+ except CanonicalJSONError as error:
483
+ raise ProposeError(
484
+ "THIN_PROPOSE_DOCUMENT_INVALID",
485
+ f"{name} is not strict JSON at {error.path}: {error.detail}; a bundle carries "
486
+ "integers, strings, booleans, null, objects and lists, and never a fractional number",
487
+ ) from error
488
+ return parsed, hashlib.sha256(raw).hexdigest()
489
+
490
+
491
+ # -- the checks, in the vocabulary of the document --------------------------------------------
492
+
493
+
494
+ def _object(value: Any, name: str, path: str = "") -> dict[str, Any]:
495
+ if not isinstance(value, dict):
496
+ raise _invalid(name, path, "must be a JSON object")
497
+ return value
498
+
499
+
500
+ def _keys(
501
+ value: Mapping[str, Any],
502
+ name: str,
503
+ path: str,
504
+ required: Sequence[str],
505
+ optional: Sequence[str] = (),
506
+ ) -> None:
507
+ missing = [key for key in required if key not in value]
508
+ if missing:
509
+ raise _invalid(name, path, f"is missing {', '.join(repr(key) for key in missing)}")
510
+ extra = sorted(set(value) - set(required) - set(optional))
511
+ if extra:
512
+ raise _invalid(
513
+ name,
514
+ path,
515
+ f"carries {', '.join(repr(key) for key in extra)}, which the shape has no room for",
516
+ )
517
+
518
+
519
+ def _string(
520
+ value: Any,
521
+ name: str,
522
+ path: str,
523
+ *,
524
+ minimum: int = 1,
525
+ maximum: int | None = None,
526
+ pattern: re.Pattern[str] | None = None,
527
+ what: str = "",
528
+ ) -> str:
529
+ if not isinstance(value, str):
530
+ raise _invalid(name, path, "must be a string")
531
+ if len(value) < minimum:
532
+ raise _invalid(name, path, f"must be at least {minimum} character(s)")
533
+ if maximum is not None and len(value) > maximum:
534
+ raise _invalid(name, path, f"must be at most {maximum} characters")
535
+ if pattern is not None and pattern.fullmatch(value) is None:
536
+ raise _invalid(name, path, f"is not {what or 'in the shape the contract admits'}")
537
+ return value
538
+
539
+
540
+ def _enum(value: Any, name: str, path: str, allowed: Sequence[str]) -> str:
541
+ if value not in allowed:
542
+ raise _invalid(name, path, f"must be one of: {', '.join(allowed)}")
543
+ return value
544
+
545
+
546
+ def _integer(value: Any, name: str, path: str, *, minimum: int, maximum: int) -> int:
547
+ if type(value) is not int or not minimum <= value <= maximum:
548
+ raise _invalid(name, path, f"must be an integer from {minimum} to {maximum}")
549
+ return value
550
+
551
+
552
+ def _boolean(value: Any, name: str, path: str) -> bool:
553
+ if type(value) is not bool:
554
+ raise _invalid(name, path, "must be true or false")
555
+ return value
556
+
557
+
558
+ def _uuid_text(value: Any, name: str, path: str) -> str:
559
+ if not isinstance(value, str) or _UUID_TEXT.fullmatch(value) is None:
560
+ raise _invalid(name, path, "must be an identifier Studio assigned (a UUID)")
561
+ return value
562
+
563
+
564
+ def _digest(value: Any, name: str, path: str) -> str:
565
+ return _string(value, name, path, pattern=_PREFIXED_DIGEST, what="a sha256:-prefixed digest")
566
+
567
+
568
+ def _timestamp(value: Any, name: str, path: str) -> str:
569
+ return _string(value, name, path, pattern=_TIMESTAMP, what="a UTC timestamp ending in Z")
570
+
571
+
572
+ def _time_range(value: Any, name: str, path: str) -> dict[str, str]:
573
+ window = _object(value, name, path)
574
+ _keys(window, name, path, ("start_inclusive", "end_exclusive"))
575
+ return {
576
+ "start_inclusive": _timestamp(window["start_inclusive"], name, f"{path}.start_inclusive"),
577
+ "end_exclusive": _timestamp(window["end_exclusive"], name, f"{path}.end_exclusive"),
578
+ }
579
+
580
+
581
+ def _column_list(
582
+ value: Any,
583
+ name: str,
584
+ path: str,
585
+ *,
586
+ minimum_items: int = 1,
587
+ maximum_items: int | None = None,
588
+ ) -> list[str]:
589
+ if not isinstance(value, list) or len(value) < minimum_items:
590
+ raise _invalid(name, path, f"must be a list of at least {minimum_items} column name(s)")
591
+ if maximum_items is not None and len(value) > maximum_items:
592
+ raise _invalid(name, path, f"must name at most {maximum_items} columns")
593
+ columns = [
594
+ _string(item, name, f"{path}[{index}]", pattern=_COLUMN, what="a lowercase column name")
595
+ for index, item in enumerate(value)
596
+ ]
597
+ if len(set(columns)) != len(columns):
598
+ raise _invalid(name, path, "must not name a column twice")
599
+ return columns
600
+
601
+
602
+ def check_dataset(document: Any) -> dict[str, Any]:
603
+ """``dataset.json``: a name and an optional description, or the id of a Dataset to adopt."""
604
+
605
+ value = _object(document, "dataset.json")
606
+ if "dataset_id" in value:
607
+ _keys(value, "dataset.json", "", ("dataset_id",))
608
+ return {"dataset_id": _uuid_text(value["dataset_id"], "dataset.json", "dataset_id")}
609
+ _keys(value, "dataset.json", "", ("name",), ("description",))
610
+ checked = {"name": _string(value["name"], "dataset.json", "name", maximum=160)}
611
+ if "description" in value:
612
+ checked["description"] = _string(
613
+ value["description"], "dataset.json", "description", minimum=0, maximum=4000
614
+ )
615
+ return checked
616
+
617
+
618
+ def check_requirements(value: Any, name: str, path: str) -> dict[str, Any]:
619
+ """The ``RequirementsUpsertCommand`` content, in the contract's own enums and patterns."""
620
+
621
+ requirements = _object(value, name, path)
622
+ _keys(
623
+ requirements,
624
+ name,
625
+ path,
626
+ (
627
+ "population",
628
+ "time_range",
629
+ "output_grain",
630
+ "required_fields",
631
+ "target_policy",
632
+ "success_criteria",
633
+ "feasibility",
634
+ ),
635
+ ("prediction_cutoff",),
636
+ )
637
+ feasibility = _object(requirements["feasibility"], name, f"{path}.feasibility")
638
+ _keys(feasibility, name, f"{path}.feasibility", ("decision", "reason_codes", "narrative"))
639
+ codes = feasibility["reason_codes"]
640
+ if not isinstance(codes, list) or len(set(map(str, codes))) != len(codes):
641
+ raise _invalid(name, f"{path}.feasibility.reason_codes", "must be a list without repeats")
642
+ criteria = requirements["success_criteria"]
643
+ if not isinstance(criteria, list) or not criteria:
644
+ raise _invalid(name, f"{path}.success_criteria", "must be a non-empty list")
645
+ checked: dict[str, Any] = {
646
+ "population": _string(requirements["population"], name, f"{path}.population", maximum=2000),
647
+ "time_range": _time_range(requirements["time_range"], name, f"{path}.time_range"),
648
+ "output_grain": _column_list(requirements["output_grain"], name, f"{path}.output_grain"),
649
+ "required_fields": _column_list(
650
+ requirements["required_fields"], name, f"{path}.required_fields"
651
+ ),
652
+ "target_policy": _enum(
653
+ requirements["target_policy"], name, f"{path}.target_policy", _TARGET_POLICIES
654
+ ),
655
+ "success_criteria": [
656
+ _string(item, name, f"{path}.success_criteria[{index}]", maximum=1000)
657
+ for index, item in enumerate(criteria)
658
+ ],
659
+ "feasibility": {
660
+ "decision": _enum(
661
+ feasibility["decision"],
662
+ name,
663
+ f"{path}.feasibility.decision",
664
+ _FEASIBILITY_DECISIONS,
665
+ ),
666
+ "reason_codes": [
667
+ _enum(item, name, f"{path}.feasibility.reason_codes[{index}]", _REASON_CODES)
668
+ for index, item in enumerate(codes)
669
+ ],
670
+ "narrative": _string(
671
+ feasibility["narrative"], name, f"{path}.feasibility.narrative", maximum=4000
672
+ ),
673
+ },
674
+ }
675
+ if "prediction_cutoff" in requirements:
676
+ checked["prediction_cutoff"] = _timestamp(
677
+ requirements["prediction_cutoff"], name, f"{path}.prediction_cutoff"
678
+ )
679
+ return checked
680
+
681
+
682
+ def check_question(document: Any) -> dict[str, Any]:
683
+ """``question.json``: the question text and its requirements, or the id of one to adopt."""
684
+
685
+ value = _object(document, "question.json")
686
+ if "question_id" in value:
687
+ _keys(value, "question.json", "", ("question_id",))
688
+ return {"question_id": _uuid_text(value["question_id"], "question.json", "question_id")}
689
+ _keys(value, "question.json", "", ("question", "requirements"))
690
+ return {
691
+ "question": _string(value["question"], "question.json", "question", maximum=8000),
692
+ "requirements": check_requirements(value["requirements"], "question.json", "requirements"),
693
+ }
694
+
695
+
696
+ def _check_query(value: Any, name: str, path: str) -> dict[str, Any]:
697
+ """The ``public_https_query`` a ``public.https`` connector is registered with."""
698
+
699
+ query = _object(value, name, path)
700
+ _keys(
701
+ query,
702
+ name,
703
+ path,
704
+ ("url", "data_format", "filename", "reader_pin", "resource_caps", "limits"),
705
+ )
706
+ limits = _object(query["limits"], name, f"{path}.limits")
707
+ _keys(limits, name, f"{path}.limits", tuple(_QUERY_LIMIT_CEILINGS))
708
+ pin = query["reader_pin"]
709
+ if pin is not None and not isinstance(pin, dict):
710
+ raise _invalid(name, f"{path}.reader_pin", "must be null or a Reader pin object")
711
+ caps = query["resource_caps"]
712
+ if caps is not None:
713
+ if pin is None:
714
+ raise _invalid(name, f"{path}.resource_caps", "must be null when reader_pin is null")
715
+ caps = _object(caps, name, f"{path}.resource_caps")
716
+ _keys(
717
+ caps,
718
+ name,
719
+ f"{path}.resource_caps",
720
+ ("max_declared_cells", "max_container_members", "max_nesting_depth"),
721
+ )
722
+ _integer(
723
+ caps["max_declared_cells"],
724
+ name,
725
+ f"{path}.resource_caps.max_declared_cells",
726
+ minimum=1,
727
+ maximum=2**53,
728
+ )
729
+ _integer(
730
+ caps["max_container_members"],
731
+ name,
732
+ f"{path}.resource_caps.max_container_members",
733
+ minimum=1,
734
+ maximum=2**53,
735
+ )
736
+ if caps["max_nesting_depth"] != 1:
737
+ raise _invalid(name, f"{path}.resource_caps.max_nesting_depth", "must be 1")
738
+ return {
739
+ "url": _string(
740
+ query["url"],
741
+ name,
742
+ f"{path}.url",
743
+ maximum=2048,
744
+ pattern=_HTTPS_URL,
745
+ what="an https:// address with no credential",
746
+ ),
747
+ "data_format": _enum(
748
+ query["data_format"], name, f"{path}.data_format", PARSER_FORMAT_ORDER
749
+ ),
750
+ "filename": _string(
751
+ query["filename"],
752
+ name,
753
+ f"{path}.filename",
754
+ maximum=255,
755
+ pattern=_FILENAME,
756
+ what="a file name without a path",
757
+ ),
758
+ "reader_pin": pin,
759
+ "resource_caps": caps,
760
+ "limits": {
761
+ key: _integer(limits[key], name, f"{path}.limits.{key}", minimum=1, maximum=ceiling)
762
+ for key, ceiling in _QUERY_LIMIT_CEILINGS.items()
763
+ },
764
+ }
765
+
766
+
767
+ def _check_connector(value: Any, name: str, path: str) -> dict[str, Any]:
768
+ connector = _object(value, name, path)
769
+ _keys(connector, name, path, ("adapter_id",), ("credential_mode", "query"))
770
+ adapter_id = connector["adapter_id"]
771
+ if adapter_id not in CONNECTOR_EGRESS_ATTESTATION:
772
+ allowed = ", ".join(sorted(CONNECTOR_EGRESS_ATTESTATION))
773
+ raise ProposeError(
774
+ "THIN_PROPOSE_CONNECTOR_UNSUPPORTED",
775
+ f"{name} {path}.adapter_id names {adapter_id!r}; the hosted authoring front door "
776
+ f"registers credential-free connectors ({allowed}), and a credentialed or staged-file "
777
+ "source enters through the local lane instead",
778
+ )
779
+ if connector.get("credential_mode", "none") != "none":
780
+ raise ProposeError(
781
+ "THIN_PROPOSE_CONNECTOR_UNSUPPORTED",
782
+ f"{name} {path}.credential_mode must be none: a {adapter_id} connector is "
783
+ "credential-free, and this lane registers no other kind",
784
+ )
785
+ checked: dict[str, Any] = {"adapter_id": adapter_id, "credential_mode": "none"}
786
+ if adapter_id == "public.https":
787
+ if "query" not in connector:
788
+ raise _invalid(name, path, "needs a 'query' for a public.https connector")
789
+ checked["query"] = _check_query(connector["query"], name, f"{path}.query")
790
+ elif "query" in connector:
791
+ raise _invalid(name, f"{path}.query", "is only carried by a public.https connector")
792
+ return checked
793
+
794
+
795
+ def check_sources(document: Any) -> list[dict[str, Any]]:
796
+ """``sources.json``: one entry per source, each registering or adopting one."""
797
+
798
+ name = "sources.json"
799
+ if not isinstance(document, list) or not document:
800
+ raise _invalid(name, "", "must be a non-empty list of source entries")
801
+ entries: list[dict[str, Any]] = []
802
+ for index, item in enumerate(document):
803
+ path = f"[{index}]"
804
+ entry = _object(item, name, path)
805
+ if "source_id" in entry:
806
+ _keys(entry, name, path, ("name", "source_id", "connector_configuration_id"))
807
+ entries.append(
808
+ {
809
+ "name": _string(entry["name"], name, f"{path}.name", maximum=200),
810
+ "source_id": _uuid_text(entry["source_id"], name, f"{path}.source_id"),
811
+ "connector_configuration_id": _uuid_text(
812
+ entry["connector_configuration_id"],
813
+ name,
814
+ f"{path}.connector_configuration_id",
815
+ ),
816
+ }
817
+ )
818
+ continue
819
+ _keys(
820
+ entry,
821
+ name,
822
+ path,
823
+ (
824
+ "name",
825
+ "source_class",
826
+ "locator",
827
+ "credential_reference_ids",
828
+ "data_classification",
829
+ "rights_claim",
830
+ "retention_policy",
831
+ "connector",
832
+ ),
833
+ )
834
+ locator = _object(entry["locator"], name, f"{path}.locator")
835
+ _keys(
836
+ locator,
837
+ name,
838
+ f"{path}.locator",
839
+ ("kind", "display_locator"),
840
+ ("connector_contract_version",),
841
+ )
842
+ checked_locator = {
843
+ "kind": _enum(locator["kind"], name, f"{path}.locator.kind", _LOCATOR_KINDS),
844
+ "display_locator": _string(
845
+ locator["display_locator"], name, f"{path}.locator.display_locator", maximum=2048
846
+ ),
847
+ }
848
+ if "connector_contract_version" in locator:
849
+ checked_locator["connector_contract_version"] = _string(
850
+ locator["connector_contract_version"],
851
+ name,
852
+ f"{path}.locator.connector_contract_version",
853
+ pattern=_SEMVER,
854
+ what="a version like 1.0.0",
855
+ )
856
+ if entry["credential_reference_ids"] != []:
857
+ raise ProposeError(
858
+ "THIN_PROPOSE_CONNECTOR_UNSUPPORTED",
859
+ f"{name} {path}.credential_reference_ids must be empty: this lane registers "
860
+ "credential-free sources only",
861
+ )
862
+ claim = _object(entry["rights_claim"], name, f"{path}.rights_claim")
863
+ _keys(
864
+ claim,
865
+ name,
866
+ f"{path}.rights_claim",
867
+ ("claimed_basis", "claim_evidence_digest"),
868
+ ("claim_note",),
869
+ )
870
+ checked_claim = {
871
+ "claimed_basis": _enum(
872
+ claim["claimed_basis"], name, f"{path}.rights_claim.claimed_basis", _CLAIMED_BASES
873
+ ),
874
+ "claim_evidence_digest": _digest(
875
+ claim["claim_evidence_digest"], name, f"{path}.rights_claim.claim_evidence_digest"
876
+ ),
877
+ }
878
+ if "claim_note" in claim:
879
+ checked_claim["claim_note"] = _string(
880
+ claim["claim_note"], name, f"{path}.rights_claim.claim_note", maximum=2000
881
+ )
882
+ retention = _object(entry["retention_policy"], name, f"{path}.retention_policy")
883
+ _keys(
884
+ retention,
885
+ name,
886
+ f"{path}.retention_policy",
887
+ ("raw_days", "derived_days", "tombstone_required"),
888
+ )
889
+ entries.append(
890
+ {
891
+ "name": _string(entry["name"], name, f"{path}.name", maximum=200),
892
+ "source_class": _enum(
893
+ entry["source_class"], name, f"{path}.source_class", _SOURCE_CLASSES
894
+ ),
895
+ "locator": checked_locator,
896
+ "credential_reference_ids": [],
897
+ "data_classification": _enum(
898
+ entry["data_classification"],
899
+ name,
900
+ f"{path}.data_classification",
901
+ _CLASSIFICATIONS,
902
+ ),
903
+ "rights_claim": checked_claim,
904
+ "retention_policy": {
905
+ "raw_days": _integer(
906
+ retention["raw_days"],
907
+ name,
908
+ f"{path}.retention_policy.raw_days",
909
+ minimum=0,
910
+ maximum=3650,
911
+ ),
912
+ "derived_days": _integer(
913
+ retention["derived_days"],
914
+ name,
915
+ f"{path}.retention_policy.derived_days",
916
+ minimum=0,
917
+ maximum=3650,
918
+ ),
919
+ "tombstone_required": _boolean(
920
+ retention["tombstone_required"],
921
+ name,
922
+ f"{path}.retention_policy.tombstone_required",
923
+ ),
924
+ },
925
+ "connector": _check_connector(entry["connector"], name, f"{path}.connector"),
926
+ }
927
+ )
928
+ names = [entry["name"] for entry in entries]
929
+ if len(set(names)) != len(names):
930
+ raise _invalid(
931
+ name,
932
+ "",
933
+ "must give each entry a distinct 'name'; the name is how the registered source id is "
934
+ "reported back for recipe.json to reference",
935
+ )
936
+ return entries
937
+
938
+
939
+ def check_backfill_policy(value: Any, name: str, path: str) -> dict[str, Any]:
940
+ policy = _object(value, name, path)
941
+ _keys(
942
+ policy,
943
+ name,
944
+ path,
945
+ (
946
+ "schema_version",
947
+ "coverage_envelope",
948
+ "max_window_seconds",
949
+ "overlap_policy",
950
+ "partition_scope",
951
+ ),
952
+ )
953
+ if policy["schema_version"] != _BACKFILL_POLICY_SCHEMA:
954
+ raise _invalid(name, f"{path}.schema_version", f"must be {_BACKFILL_POLICY_SCHEMA}")
955
+ scope = _object(policy["partition_scope"], name, f"{path}.partition_scope")
956
+ _keys(
957
+ scope,
958
+ name,
959
+ f"{path}.partition_scope",
960
+ ("fields", "allowed_values", "max_partitions_per_run"),
961
+ )
962
+ allowed = _object(scope["allowed_values"], name, f"{path}.partition_scope.allowed_values")
963
+ if len(allowed) > 16:
964
+ raise _invalid(
965
+ name, f"{path}.partition_scope.allowed_values", "must name at most 16 fields"
966
+ )
967
+ checked_allowed: dict[str, list[str]] = {}
968
+ for field, values in allowed.items():
969
+ where = f"{path}.partition_scope.allowed_values.{field}"
970
+ if not isinstance(values, list) or not 1 <= len(values) <= 10_000:
971
+ raise _invalid(name, where, "must be a list of 1 to 10000 values")
972
+ items = [
973
+ _string(item, name, f"{where}[{index}]", maximum=256)
974
+ for index, item in enumerate(values)
975
+ ]
976
+ if len(set(items)) != len(items):
977
+ raise _invalid(name, where, "must not repeat a value")
978
+ checked_allowed[field] = items
979
+ return {
980
+ "schema_version": _BACKFILL_POLICY_SCHEMA,
981
+ "coverage_envelope": _time_range(
982
+ policy["coverage_envelope"], name, f"{path}.coverage_envelope"
983
+ ),
984
+ "max_window_seconds": _integer(
985
+ policy["max_window_seconds"],
986
+ name,
987
+ f"{path}.max_window_seconds",
988
+ minimum=1,
989
+ maximum=315_360_000,
990
+ ),
991
+ "overlap_policy": _enum(
992
+ policy["overlap_policy"], name, f"{path}.overlap_policy", _OVERLAP_POLICIES
993
+ ),
994
+ "partition_scope": {
995
+ "fields": _column_list(
996
+ scope["fields"],
997
+ name,
998
+ f"{path}.partition_scope.fields",
999
+ minimum_items=0,
1000
+ maximum_items=16,
1001
+ ),
1002
+ "allowed_values": checked_allowed,
1003
+ "max_partitions_per_run": _integer(
1004
+ scope["max_partitions_per_run"],
1005
+ name,
1006
+ f"{path}.partition_scope.max_partitions_per_run",
1007
+ minimum=1,
1008
+ maximum=10_000,
1009
+ ),
1010
+ },
1011
+ }
1012
+
1013
+
1014
+ def check_table_plan(document: Any) -> dict[str, Any]:
1015
+ """``table_plan.json``: the authored plan content, or the id of a plan to adopt.
1016
+
1017
+ The three digests are optional: each is derived from ``recipe.json`` when left out, and held
1018
+ to that derivation when written, so an author never has to compute one and cannot get one
1019
+ wrong. ``approved_backfill_policy`` defaults to the recipe's own ``backfill_policy``.
1020
+ """
1021
+
1022
+ name = "table_plan.json"
1023
+ value = _object(document, name)
1024
+ if "table_plan_id" in value:
1025
+ _keys(value, name, "", ("table_plan_id",))
1026
+ return {"table_plan_id": _uuid_text(value["table_plan_id"], name, "table_plan_id")}
1027
+ _keys(
1028
+ value,
1029
+ name,
1030
+ "",
1031
+ ("output_grain", "transformation_contract_version"),
1032
+ (
1033
+ "execution_plan_digest",
1034
+ "validation_policy_digest",
1035
+ "approved_backfill_policy",
1036
+ "backfill_policy_digest",
1037
+ ),
1038
+ )
1039
+ checked: dict[str, Any] = {
1040
+ "output_grain": _column_list(value["output_grain"], name, "output_grain"),
1041
+ "transformation_contract_version": _string(
1042
+ value["transformation_contract_version"],
1043
+ name,
1044
+ "transformation_contract_version",
1045
+ pattern=_SEMVER,
1046
+ what="a version like 1.0.0",
1047
+ ),
1048
+ }
1049
+ for field in ("execution_plan_digest", "validation_policy_digest", "backfill_policy_digest"):
1050
+ if field in value:
1051
+ checked[field] = _digest(value[field], name, field)
1052
+ if "approved_backfill_policy" in value:
1053
+ checked["approved_backfill_policy"] = check_backfill_policy(
1054
+ value["approved_backfill_policy"], name, "approved_backfill_policy"
1055
+ )
1056
+ return checked
1057
+
1058
+
1059
+ def check_activation(document: Any) -> dict[str, Any]:
1060
+ """``activation.json``: the Table name and the refresh schedule, defaults filled in."""
1061
+
1062
+ name = "activation.json"
1063
+ value = _object(document, name)
1064
+ _keys(value, name, "", ("table_name",), ("schedule",))
1065
+ table_name = _string(value["table_name"], name, "table_name", maximum=200)
1066
+ if not table_name.strip():
1067
+ raise _invalid(name, "table_name", "must not be blank")
1068
+ schedule = dict(DEFAULT_SCHEDULE)
1069
+ if "schedule" in value:
1070
+ authored = _object(value["schedule"], name, "schedule")
1071
+ _keys(authored, name, "schedule", (), tuple(DEFAULT_SCHEDULE))
1072
+ schedule.update(authored)
1073
+ checked_schedule = {
1074
+ "cron": _string(schedule["cron"], name, "schedule.cron", minimum=9, maximum=100),
1075
+ "timezone": _enum(schedule["timezone"], name, "schedule.timezone", ("UTC",)),
1076
+ "mode": _enum(schedule["mode"], name, "schedule.mode", ("incremental_refresh",)),
1077
+ "max_concurrent_runs": _integer(
1078
+ schedule["max_concurrent_runs"],
1079
+ name,
1080
+ "schedule.max_concurrent_runs",
1081
+ minimum=1,
1082
+ maximum=10,
1083
+ ),
1084
+ }
1085
+ return {"table_name": table_name, "schedule": checked_schedule}
1086
+
1087
+
1088
+ def check_recipe(document: Any) -> tuple[dict[str, Any], bytes, str]:
1089
+ """``recipe.json``: its coordinates, its canonical bytes and their digest.
1090
+
1091
+ Only what the proposal restates is checked -- the schema version, the first-recipe coordinate,
1092
+ the source inventory's shape, and the two members the plan's digests derive from. The transform
1093
+ rules, the Reader pins and the unit semantics are NOT checked here: that is the local lane's
1094
+ job and imports the modules this profile forbids, and the hosted worker refuses a recipe its
1095
+ engine cannot run.
1096
+ """
1097
+
1098
+ name = "recipe.json"
1099
+ recipe = _object(document, name)
1100
+ for field in (
1101
+ "schema_version",
1102
+ "recipe_id",
1103
+ "recipe_version",
1104
+ "predecessor_recipe_digest",
1105
+ "sources",
1106
+ "transform_plan",
1107
+ "validation",
1108
+ ):
1109
+ if field not in recipe:
1110
+ raise _invalid(name, "", f"is missing {field!r}")
1111
+ _enum(recipe["schema_version"], name, "schema_version", RECIPE_SCHEMA_VERSIONS)
1112
+ _string(
1113
+ recipe["recipe_id"],
1114
+ name,
1115
+ "recipe_id",
1116
+ pattern=_IDENTIFIER,
1117
+ what="a recipe identifier (letters, digits, '_', '.', '-')",
1118
+ )
1119
+ if recipe["recipe_version"] != 1:
1120
+ raise ProposeError(
1121
+ "THIN_PROPOSE_RECIPE_COORDINATE",
1122
+ "a hosted-first recipe is a first recipe: recipe.json must carry recipe_version 1 "
1123
+ f"(it carries {recipe['recipe_version']!r})",
1124
+ )
1125
+ if recipe["predecessor_recipe_digest"] is not None:
1126
+ raise ProposeError(
1127
+ "THIN_PROPOSE_RECIPE_COORDINATE",
1128
+ "a hosted-first recipe has no predecessor: recipe.json predecessor_recipe_digest "
1129
+ "must be null",
1130
+ )
1131
+ sources = recipe["sources"]
1132
+ if not isinstance(sources, list) or not sources:
1133
+ raise _invalid(name, "sources", "must be a non-empty list")
1134
+ seen: set[str] = set()
1135
+ for index, item in enumerate(sources):
1136
+ source = _object(item, name, f"sources[{index}]")
1137
+ source_id = _string(
1138
+ source.get("source_id"),
1139
+ name,
1140
+ f"sources[{index}].source_id",
1141
+ pattern=_IDENTIFIER,
1142
+ what="a source identifier",
1143
+ )
1144
+ _string(
1145
+ source.get("adapter_id"),
1146
+ name,
1147
+ f"sources[{index}].adapter_id",
1148
+ pattern=_IDENTIFIER,
1149
+ what="an adapter identifier",
1150
+ )
1151
+ if source_id in seen:
1152
+ raise _invalid(
1153
+ name,
1154
+ f"sources[{index}].source_id",
1155
+ f"repeats {source_id!r}; a recipe names each source once",
1156
+ )
1157
+ seen.add(source_id)
1158
+ _object(recipe["transform_plan"], name, "transform_plan")
1159
+ validation = _object(recipe["validation"], name, "validation")
1160
+ _string(
1161
+ validation.get("policy_digest"),
1162
+ name,
1163
+ "validation.policy_digest",
1164
+ pattern=_BARE_DIGEST,
1165
+ what="a bare lowercase sha256 digest",
1166
+ )
1167
+ if "backfill_policy" in recipe:
1168
+ check_backfill_policy(recipe["backfill_policy"], name, "backfill_policy")
1169
+ try:
1170
+ canonical_bytes = canonical_json_bytes(recipe)
1171
+ except CanonicalJSONError as error:
1172
+ raise ProposeError(
1173
+ "THIN_PROPOSE_RECIPE_NONCANONICAL",
1174
+ f"recipe.json cannot be canonicalised at {error.path}: {error.detail}",
1175
+ ) from error
1176
+ if len(canonical_bytes) > MAX_CANONICAL_RECIPE_BYTES:
1177
+ raise ProposeError(
1178
+ "THIN_PROPOSE_RECIPE_TOO_LARGE",
1179
+ f"recipe.json is {len(canonical_bytes)} canonical bytes, over the "
1180
+ f"{MAX_CANONICAL_RECIPE_BYTES}-byte contract ceiling",
1181
+ )
1182
+ return recipe, canonical_bytes, hashlib.sha256(canonical_bytes).hexdigest()
1183
+
1184
+
1185
+ def harness_version() -> str:
1186
+ """The X.Y.Z this client records as ``harness_version``, refused when there is none.
1187
+
1188
+ Checked before the first request rather than at the proposal stage: a checkout that reports
1189
+ ``uninstalled`` would otherwise register four stages of resources and then stop.
1190
+ """
1191
+
1192
+ version = package_version()
1193
+ if _HARNESS_VERSION.fullmatch(version) is None:
1194
+ raise ProposeError(
1195
+ "THIN_PROPOSE_HARNESS_VERSION",
1196
+ f"this client reports version {version!r}, which is not the X.Y.Z a hosted bootstrap "
1197
+ "records; run propose from an installed release",
1198
+ )
1199
+ return version
1200
+
1201
+
1202
+ # -- the derived plan digests -----------------------------------------------------------------
1203
+
1204
+
1205
+ def execution_plan_digest(recipe: Mapping[str, Any]) -> str:
1206
+ """The digest of the recipe's transform plan as the hosted worker seals it.
1207
+
1208
+ The worker writes ``plan.json`` as the plan's sorted-key compact JSON terminated by one
1209
+ newline and binds that member's digest; for a plan in its normalized form those bytes are the
1210
+ canonical bytes plus the newline, so the plan's ``execution_plan_digest`` can describe the
1211
+ recipe exactly without this lane carrying the engine.
1212
+ """
1213
+
1214
+ return (
1215
+ "sha256:"
1216
+ + hashlib.sha256(canonical_json_bytes(recipe["transform_plan"]) + b"\n").hexdigest()
1217
+ )
1218
+
1219
+
1220
+ def validation_policy_digest(recipe: Mapping[str, Any]) -> str:
1221
+ return "sha256:" + str(recipe["validation"]["policy_digest"])
1222
+
1223
+
1224
+ def backfill_policy_digest(policy: Mapping[str, Any]) -> str:
1225
+ return "sha256:" + canonical_sha256(dict(policy))
1226
+
1227
+
1228
+ def plan_content(plan: Mapping[str, Any], recipe: Mapping[str, Any]) -> dict[str, Any]:
1229
+ """The table plan's authored content, its digests derived from the recipe and held to it."""
1230
+
1231
+ name = "table_plan.json"
1232
+ derived = {
1233
+ "execution_plan_digest": execution_plan_digest(recipe),
1234
+ "validation_policy_digest": validation_policy_digest(recipe),
1235
+ }
1236
+ policy = plan.get("approved_backfill_policy")
1237
+ if policy is None:
1238
+ policy = recipe.get("backfill_policy")
1239
+ if not isinstance(policy, dict):
1240
+ raise _invalid(
1241
+ name,
1242
+ "approved_backfill_policy",
1243
+ "is required when recipe.json carries no backfill_policy",
1244
+ )
1245
+ policy = check_backfill_policy(policy, "recipe.json", "backfill_policy")
1246
+ derived["backfill_policy_digest"] = backfill_policy_digest(policy)
1247
+ for field, expected in derived.items():
1248
+ written = plan.get(field)
1249
+ if written is not None and written != expected:
1250
+ raise ProposeError(
1251
+ "THIN_PROPOSE_PLAN_DIGEST",
1252
+ f"table_plan.json {field} is {written}, but recipe.json derives {expected}; leave "
1253
+ "the field out to have it derived, or correct it",
1254
+ )
1255
+ return {
1256
+ "output_grain": plan["output_grain"],
1257
+ "transformation_contract_version": plan["transformation_contract_version"],
1258
+ "approved_backfill_policy": policy,
1259
+ **derived,
1260
+ }
1261
+
1262
+
1263
+ # --------------------------------------------------------------------------------------------
1264
+ # The journal
1265
+ # --------------------------------------------------------------------------------------------
1266
+
1267
+
1268
+ class Journal:
1269
+ """The record of one bundle's authoring, in ``BUNDLE_DIR/.mr-data``.
1270
+
1271
+ It carries a nonce minted when it is created -- the idempotency keys are derived from it, so
1272
+ two bundles with byte-identical documents can never replay each other's resources -- the
1273
+ digest of every document pinned at the moment the first resource built from it completed, and
1274
+ one entry per created or adopted resource: the request digest it was created from, the
1275
+ idempotency key that created it, and the resource Studio answered with. A re-run replays every
1276
+ completed resource from here without rebuilding it; a pinned document that changed is refused
1277
+ with the document named.
1278
+ """
1279
+
1280
+ def __init__(self, bundle: Path) -> None:
1281
+ self._dir = bundle / JOURNAL_DIR
1282
+ self._path = self._dir / JOURNAL_FILE
1283
+ self.state: dict[str, Any] = {}
1284
+
1285
+ @property
1286
+ def path(self) -> Path:
1287
+ return self._path
1288
+
1289
+ def load_or_start(self, *, workspace_id: UUID) -> None:
1290
+ if self._path.is_file():
1291
+ self.state = read_journal(self._path)
1292
+ if self.state.get("workspace_id") != str(workspace_id):
1293
+ raise ProposeError(
1294
+ "THIN_PROPOSE_JOURNAL_SCOPE",
1295
+ "this bundle was proposed under a different workspace; a bundle belongs to the "
1296
+ "workspace that first proposed it",
1297
+ )
1298
+ return
1299
+ nonce = secrets.token_hex(16)
1300
+ self.state = {
1301
+ "schema_version": JOURNAL_SCHEMA,
1302
+ "workspace_id": str(workspace_id),
1303
+ "nonce": nonce,
1304
+ "identity": canonical_sha256({"workspace_id": str(workspace_id), "nonce": nonce}),
1305
+ "documents": {},
1306
+ "operations": {},
1307
+ "resources": {},
1308
+ "etags": {},
1309
+ "ids": {},
1310
+ "through": None,
1311
+ }
1312
+ self.save()
1313
+
1314
+ def assert_pinned_documents_unchanged(self, digests: Mapping[str, str]) -> None:
1315
+ pinned = self.state.get("documents", {})
1316
+ drifted = sorted(
1317
+ name for name, digest in digests.items() if name in pinned and pinned[name] != digest
1318
+ )
1319
+ if drifted:
1320
+ raise ProposeError(
1321
+ "THIN_PROPOSE_BUNDLE_DRIFTED",
1322
+ f"{', '.join(drifted)} changed after the resource built from it was created, so "
1323
+ "the resource Studio holds no longer describes the document; restore the document "
1324
+ "to resume, or start a new bundle that adopts the created resources by id",
1325
+ )
1326
+
1327
+ def pin(self, names: Sequence[str], digests: Mapping[str, str]) -> None:
1328
+ for name in names:
1329
+ if name in digests:
1330
+ self.state.setdefault("documents", {})[name] = digests[name]
1331
+
1332
+ def save(self) -> None:
1333
+ """Write the journal whole, through a temp file and one rename, mode 0600."""
1334
+
1335
+ self._dir.mkdir(mode=0o700, exist_ok=True)
1336
+ rendered = json.dumps(self.state, ensure_ascii=False, sort_keys=True).encode("utf-8")
1337
+ staged = self._path.with_name(self._path.name + ".partial")
1338
+ descriptor = os.open(staged, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
1339
+ try:
1340
+ os.write(descriptor, rendered)
1341
+ os.fsync(descriptor)
1342
+ finally:
1343
+ os.close(descriptor)
1344
+ os.chmod(staged, 0o600)
1345
+ os.replace(staged, self._path)
1346
+
1347
+ def record_id(self, key: str, value: Any) -> None:
1348
+ self.state.setdefault("ids", {})[key] = value
1349
+
1350
+ def record_stage(self, stage: str) -> None:
1351
+ """Record the furthest stage this bundle reached. A shorter re-run never winds it back."""
1352
+
1353
+ reached = self.state.get("through")
1354
+ if reached is None or STAGES.index(stage) > STAGES.index(reached):
1355
+ self.state["through"] = stage
1356
+ self.save()
1357
+
1358
+
1359
+ def read_journal(path: Path) -> dict[str, Any]:
1360
+ """One propose journal, read under the plain-file rule, refused unless this client wrote it."""
1361
+
1362
+ try:
1363
+ raw = read_plain_file(path, max_bytes=MAX_JOURNAL_BYTES)
1364
+ parsed = parse_json(raw)
1365
+ except (PlainFileRefusal, OSError, CanonicalJSONError) as error:
1366
+ raise ProposeError(
1367
+ "THIN_PROPOSE_JOURNAL_UNREADABLE",
1368
+ f"the propose journal at {path} could not be read; move it aside to start a fresh "
1369
+ "bundle",
1370
+ ) from error
1371
+ if not isinstance(parsed, dict) or parsed.get("schema_version") != JOURNAL_SCHEMA:
1372
+ raise ProposeError(
1373
+ "THIN_PROPOSE_JOURNAL_UNREADABLE",
1374
+ f"the file at {path} is not a propose journal this client wrote",
1375
+ )
1376
+ return parsed
1377
+
1378
+
1379
+ def journal_ids(bundle: Path | str) -> dict[str, Any]:
1380
+ """The ids one bundle's journal recorded, for ``research-open --bundle`` to read.
1381
+
1382
+ A bundle with no journal has registered nothing yet, and the refusal names the command that
1383
+ registers it.
1384
+ """
1385
+
1386
+ directory = Path(str(bundle)).expanduser()
1387
+ path = directory / JOURNAL_DIR / JOURNAL_FILE
1388
+ if not path.is_file():
1389
+ raise ProposeError(
1390
+ "THIN_PROPOSE_JOURNAL_UNREADABLE",
1391
+ f"--bundle names {directory}, which has no propose journal at "
1392
+ f"{JOURNAL_DIR}/{JOURNAL_FILE}; run mr-data propose --through sources against it first",
1393
+ )
1394
+ ids = read_journal(path).get("ids")
1395
+ return dict(ids) if isinstance(ids, dict) else {}
1396
+
1397
+
1398
+ # --------------------------------------------------------------------------------------------
1399
+ # One journaled mutation, and one adoption
1400
+ # --------------------------------------------------------------------------------------------
1401
+
1402
+
1403
+ def _idempotency_key(identity: str, operation: str, request_digest: str) -> str:
1404
+ """A deterministic idempotency key for one request of one bundle.
1405
+
1406
+ Derived from the bundle's identity -- which carries the journal's nonce, so no other bundle
1407
+ can derive it -- the operation, and the request's own digest, the way ``hosted_deploy``
1408
+ folds the whole request into its deployment identity. A crash and resume re-sends the
1409
+ identical request under the identical key and Studio replays its first result; a corrected
1410
+ request after a refusal gets a key of its own.
1411
+ """
1412
+
1413
+ digest = hashlib.sha256(f"{operation}:{identity}:{request_digest}".encode()).hexdigest()
1414
+ return f"mr-data.propose.v1:{digest}"
1415
+
1416
+
1417
+ def _drifted_fields(journaled: Mapping[str, Any], fields: Mapping[str, str]) -> str:
1418
+ recorded = journaled.get("request_fields")
1419
+ if not isinstance(recorded, Mapping):
1420
+ return "the request"
1421
+ drifted = sorted(
1422
+ field for field in {*recorded, *fields} if recorded.get(field) != fields.get(field)
1423
+ )
1424
+ return ", ".join(drifted) if drifted else "the request"
1425
+
1426
+
1427
+ def _mutate(
1428
+ journal: Journal,
1429
+ client: StudioProposeClient,
1430
+ *,
1431
+ name: str,
1432
+ path: str,
1433
+ build: Callable[[], Mapping[str, Any]],
1434
+ expected: int,
1435
+ document: str | None,
1436
+ method: str = "POST",
1437
+ if_match: str | None = None,
1438
+ lost_answer_recoverable: bool = False,
1439
+ refusal_remedy: Callable[[ThinLaneError], str | None] | None = None,
1440
+ ) -> dict[str, Any]:
1441
+ """Create one resource, or replay the one this bundle already holds.
1442
+
1443
+ ``document`` names the bundle document the request is built from, or ``None`` for the two
1444
+ requests built from Studio's own answers (the proposal, the approval). A completed operation
1445
+ built from a document is REBUILT on replay and held to the request that created it: the pin
1446
+ on the document catches a change once its stage completed, but a stage that did not complete
1447
+ -- one source registered, the next refused -- has pinned nothing, and a replay that trusted
1448
+ the journal's position would hand the first source's id to whichever entry now sits there.
1449
+ A mismatch is the document drift it is, named by document and operation. The two requests
1450
+ built from Studio's answers are replayed without rebuilding: this client's own version and
1451
+ Studio's fleet selection legitimately move between runs and are nobody's drift.
1452
+
1453
+ A planned operation whose answer was lost is re-sent under its own key, so Studio replays the
1454
+ first result rather than creating a second resource; if the request it rebuilds is no longer
1455
+ the one that was sent, it is refused, because that first request may have been applied --
1456
+ unless ``lost_answer_recoverable`` says the caller already learned the outcome from Studio, in
1457
+ which case a changed re-send is safe and goes out under its own key.
1458
+ """
1459
+
1460
+ operations = journal.state["operations"]
1461
+ resources = journal.state["resources"]
1462
+ existing = resources.get(name)
1463
+ if isinstance(existing, dict) and document is None:
1464
+ return existing
1465
+ try:
1466
+ body = dict(build())
1467
+ request_digest = canonical_sha256(body)
1468
+ fields = {field: canonical_sha256(value) for field, value in body.items()}
1469
+ except CanonicalJSONError as error:
1470
+ raise ProposeError(
1471
+ "THIN_PROPOSE_DOCUMENT_INVALID",
1472
+ f"the {name} request cannot be canonicalised at {error.path}: {error.detail}",
1473
+ ) from error
1474
+ journaled = operations.get(name)
1475
+ if isinstance(existing, dict):
1476
+ recorded = journaled.get("request_digest") if isinstance(journaled, dict) else None
1477
+ if recorded != request_digest:
1478
+ raise ProposeError(
1479
+ "THIN_PROPOSE_BUNDLE_DRIFTED",
1480
+ f"{document} changed after {name} was created from it "
1481
+ f"({_drifted_fields(journaled or {}, fields)} no longer match), so the resource "
1482
+ "Studio holds no longer describes the document; restore the document to resume, "
1483
+ "or start a new bundle that adopts the created resources by id",
1484
+ )
1485
+ return existing
1486
+ if (
1487
+ isinstance(journaled, dict)
1488
+ and journaled.get("request_digest") != request_digest
1489
+ and not lost_answer_recoverable
1490
+ ):
1491
+ raise ProposeError(
1492
+ "THIN_PROPOSE_RESUME_UNCERTAIN",
1493
+ f"{name} was sent before and its answer was lost, and "
1494
+ f"{_drifted_fields(journaled, fields)} here no longer matches what was sent; that "
1495
+ "request may have been applied, so restore "
1496
+ f"{document or 'the request'} to resume it, or start a new bundle",
1497
+ )
1498
+ idempotency_key = _idempotency_key(journal.state["identity"], name, request_digest)
1499
+ operations[name] = {
1500
+ "operation": name,
1501
+ "document": document,
1502
+ "request_digest": request_digest,
1503
+ "request_fields": fields,
1504
+ "idempotency_key": idempotency_key,
1505
+ "state": "planned",
1506
+ }
1507
+ journal.save()
1508
+ captured: dict[str, str] = {}
1509
+ try:
1510
+ result = client.create(
1511
+ path,
1512
+ body,
1513
+ idempotency_key=idempotency_key,
1514
+ expected=expected,
1515
+ method=method,
1516
+ if_match=if_match,
1517
+ response_headers=captured,
1518
+ )
1519
+ except ThinLaneError as refusal:
1520
+ if _nothing_was_applied(refusal):
1521
+ # Studio answered and applied nothing, so the plan is withdrawn and the document that
1522
+ # produced it stays editable: the corrected request goes out under its own key.
1523
+ del operations[name]
1524
+ journal.save()
1525
+ remedy = refusal_remedy(refusal) if refusal_remedy is not None else None
1526
+ if remedy is not None:
1527
+ refusal.detail = f"{refusal.detail}; {remedy}"
1528
+ refusal.args = (f"{refusal.code}: {refusal.detail}",)
1529
+ raise
1530
+ resources[name] = result
1531
+ etag = captured.get("etag")
1532
+ if isinstance(etag, str) and etag:
1533
+ journal.state.setdefault("etags", {})[name] = etag
1534
+ operations[name] = {**operations[name], "state": "completed"}
1535
+ journal.save()
1536
+ return result
1537
+
1538
+
1539
+ def _adopt(
1540
+ journal: Journal,
1541
+ client: StudioProposeClient,
1542
+ *,
1543
+ name: str,
1544
+ path: str,
1545
+ check: Callable[[Mapping[str, Any]], None],
1546
+ missing: str | None = None,
1547
+ ) -> dict[str, Any]:
1548
+ """Read one resource that already exists and hold it the way a created one is held.
1549
+
1550
+ ``missing`` is the sentence a 404 becomes, in the document's words, when the id named does
1551
+ not resolve -- an adopted question whose requirements were never written, say -- so the
1552
+ refusal names what to author rather than a route.
1553
+ """
1554
+
1555
+ resources = journal.state["resources"]
1556
+ existing = resources.get(name)
1557
+ if isinstance(existing, dict):
1558
+ journaled = journal.state["operations"].get(name, {})
1559
+ if journaled.get("state") == "adopted" and journaled.get("path") != path:
1560
+ raise ProposeError(
1561
+ "THIN_PROPOSE_BUNDLE_DRIFTED",
1562
+ f"the id {name} adopts changed after it was adopted; restore it, or start a new "
1563
+ "bundle",
1564
+ )
1565
+ return existing
1566
+ captured: dict[str, str] = {}
1567
+ try:
1568
+ record = client.read(path, response_headers=captured)
1569
+ except ThinLaneError as refusal:
1570
+ if refusal.code == "THIN_NOT_FOUND" and missing is not None:
1571
+ raise ProposeError("THIN_PROPOSE_ADOPT_SCOPE", missing) from refusal
1572
+ raise
1573
+ if record.get("workspace_id") != journal.state["workspace_id"]:
1574
+ raise ProposeError(
1575
+ "THIN_PROPOSE_ADOPT_SCOPE",
1576
+ f"the resource {name} names belongs to another workspace than this bundle's",
1577
+ )
1578
+ check(record)
1579
+ resources[name] = record
1580
+ etag = captured.get("etag")
1581
+ if isinstance(etag, str) and etag:
1582
+ journal.state.setdefault("etags", {})[name] = etag
1583
+ # The path the resource was adopted from is journaled beside it, so a later run that adopts a
1584
+ # different id under the same name is a drift rather than a silent substitution.
1585
+ journal.state["operations"][name] = {"operation": name, "state": "adopted", "path": path}
1586
+ journal.save()
1587
+ return record
1588
+
1589
+
1590
+ # --------------------------------------------------------------------------------------------
1591
+ # The bodies, with the contract applied here
1592
+ # --------------------------------------------------------------------------------------------
1593
+
1594
+
1595
+ def _uuid_response(record: Mapping[str, Any], field: str, name: str) -> str:
1596
+ value = record.get(field)
1597
+ if not isinstance(value, str) or _UUID_TEXT.fullmatch(value) is None:
1598
+ raise ProposeError(
1599
+ "THIN_PROPOSE_RESPONSE_INVALID", f"Studio answered {name} without a usable {field}"
1600
+ )
1601
+ return value
1602
+
1603
+
1604
+ def _text_response(record: Mapping[str, Any], field: str, name: str) -> str:
1605
+ value = record.get(field)
1606
+ if not isinstance(value, str) or not value:
1607
+ raise ProposeError(
1608
+ "THIN_PROPOSE_RESPONSE_INVALID", f"Studio answered {name} without a usable {field}"
1609
+ )
1610
+ return value
1611
+
1612
+
1613
+ def dataset_body(dataset: Mapping[str, Any], *, workspace_id: UUID) -> dict[str, Any]:
1614
+ return {"workspace_id": str(workspace_id), **dataset}
1615
+
1616
+
1617
+ def question_body(
1618
+ question: Mapping[str, Any], *, workspace_id: UUID, dataset_id: str
1619
+ ) -> dict[str, Any]:
1620
+ return {
1621
+ "workspace_id": str(workspace_id),
1622
+ "dataset_id": dataset_id,
1623
+ "question": question["question"],
1624
+ }
1625
+
1626
+
1627
+ def requirements_body(question: Mapping[str, Any], *, workspace_id: UUID) -> dict[str, Any]:
1628
+ return {
1629
+ "schema_version": STUDIO_SCHEMA_VERSION,
1630
+ "workspace_id": str(workspace_id),
1631
+ **question["requirements"],
1632
+ }
1633
+
1634
+
1635
+ def source_body(entry: Mapping[str, Any], *, workspace_id: UUID, dataset_id: str) -> dict[str, Any]:
1636
+ return {
1637
+ "schema_version": STUDIO_SCHEMA_VERSION,
1638
+ "workspace_id": str(workspace_id),
1639
+ "dataset_id": dataset_id,
1640
+ **{key: value for key, value in entry.items() if key != "connector"},
1641
+ }
1642
+
1643
+
1644
+ def connector_body(
1645
+ entry: Mapping[str, Any], *, workspace_id: UUID, source_id: str
1646
+ ) -> dict[str, Any]:
1647
+ connector = entry["connector"]
1648
+ body: dict[str, Any] = {
1649
+ "schema_version": STUDIO_SCHEMA_VERSION,
1650
+ "workspace_id": str(workspace_id),
1651
+ "source_id": source_id,
1652
+ "adapter_id": connector["adapter_id"],
1653
+ "credential_mode": "none",
1654
+ "crawler_egress_policy_attestation": CONNECTOR_EGRESS_ATTESTATION[connector["adapter_id"]],
1655
+ }
1656
+ if "query" in connector:
1657
+ body["query"] = connector["query"]
1658
+ return body
1659
+
1660
+
1661
+ def table_plan_body(
1662
+ content: Mapping[str, Any],
1663
+ *,
1664
+ workspace_id: UUID,
1665
+ dataset_id: str,
1666
+ question_id: str,
1667
+ requirements_id: str,
1668
+ source_ids: Sequence[str],
1669
+ ) -> dict[str, Any]:
1670
+ return {
1671
+ "schema_version": STUDIO_SCHEMA_VERSION,
1672
+ "workspace_id": str(workspace_id),
1673
+ "dataset_id": dataset_id,
1674
+ "question_id": question_id,
1675
+ "requirements_id": requirements_id,
1676
+ "source_ids": list(source_ids),
1677
+ **content,
1678
+ }
1679
+
1680
+
1681
+ def bootstrap_evidence(
1682
+ *, version: str, research_session_id: str | None, research_run_id: str | None
1683
+ ) -> dict[str, Any]:
1684
+ evidence: dict[str, Any] = {"origin": HOSTED_BOOTSTRAP_ORIGIN, "harness_version": version}
1685
+ if research_session_id is not None:
1686
+ evidence["research_session_id"] = research_session_id
1687
+ if research_run_id is not None:
1688
+ evidence["research_run_id"] = research_run_id
1689
+ return evidence
1690
+
1691
+
1692
+ def connector_authority(*, source_id: str, configuration: Mapping[str, Any]) -> dict[str, Any]:
1693
+ """One connector source-authority binding, its digest derived exactly as Studio derives it.
1694
+
1695
+ Built from the connector configuration Studio holds -- adapter, digest, attestation -- so an
1696
+ adopted configuration binds the same way a registered one does. ``source_authority_digest``
1697
+ is the bare SHA-256 over the canonical binding with that field removed, the identical rule
1698
+ ``hosted_deploy._source_authorities`` uses and ``_assert_recipe_source_authority_current``
1699
+ recomputes on Studio's side, so the binding is the one the hosted worker is handed and checks
1700
+ the recipe against at build time.
1701
+ """
1702
+
1703
+ authority: dict[str, Any] = {
1704
+ "source_id": source_id,
1705
+ "authority_kind": "connector",
1706
+ "adapter_id": _text_response(configuration, "adapter_id", "the connector configuration"),
1707
+ "connector_configuration_id": _uuid_response(
1708
+ configuration, "connector_configuration_id", "the connector configuration"
1709
+ ),
1710
+ "connector_configuration_digest": _text_response(
1711
+ configuration, "configuration_digest", "the connector configuration"
1712
+ ),
1713
+ "credential_mode": "none",
1714
+ "crawler_egress_policy_attestation": _text_response(
1715
+ configuration, "crawler_egress_policy_attestation", "the connector configuration"
1716
+ ),
1717
+ }
1718
+ authority["source_authority_digest"] = canonical_sha256(dict(authority))
1719
+ return authority
1720
+
1721
+
1722
+ def proposal_body(
1723
+ *,
1724
+ recipe: Mapping[str, Any],
1725
+ canonical_bytes: bytes,
1726
+ recipe_digest: str,
1727
+ activation: Mapping[str, Any],
1728
+ workspace_id: UUID,
1729
+ dataset_id: str,
1730
+ table_plan_id: str,
1731
+ source_authority_bindings: Sequence[Mapping[str, Any]],
1732
+ evidence: Mapping[str, Any],
1733
+ ) -> dict[str, Any]:
1734
+ """The ``TableRecipeProposalCommand``, every coordinate read straight off the recipe."""
1735
+
1736
+ inventory_digest = canonical_sha256(
1737
+ [
1738
+ {
1739
+ "source_id": item["source_id"],
1740
+ "source_authority_digest": item["source_authority_digest"],
1741
+ }
1742
+ for item in source_authority_bindings
1743
+ ]
1744
+ )
1745
+ return {
1746
+ "schema_version": STUDIO_SCHEMA_VERSION,
1747
+ "workspace_id": str(workspace_id),
1748
+ "dataset_id": dataset_id,
1749
+ "table_plan_id": table_plan_id,
1750
+ "recipe_media_type": RECIPE_MEDIA_TYPE,
1751
+ "recipe_schema_version": recipe["schema_version"],
1752
+ "table_recipe_id": recipe["recipe_id"],
1753
+ "recipe_version": 1,
1754
+ "predecessor_recipe_digest": None,
1755
+ "canonical_recipe_json": canonical_bytes.decode("utf-8"),
1756
+ "recipe_digest": recipe_digest,
1757
+ "source_authority_bindings": [dict(item) for item in source_authority_bindings],
1758
+ "source_authority_bindings_digest": inventory_digest,
1759
+ "bootstrap_evidence": dict(evidence),
1760
+ "activation_intent": {
1761
+ "table_name": activation["table_name"],
1762
+ "schedule": dict(activation["schedule"]),
1763
+ },
1764
+ }
1765
+
1766
+
1767
+ def preflight_body(
1768
+ proposal: Mapping[str, Any], *, workspace_id: UUID, recipe_proposal_id: str
1769
+ ) -> dict[str, Any]:
1770
+ return {
1771
+ "workspace_id": str(workspace_id),
1772
+ "dataset_id": proposal["dataset_id"],
1773
+ "table_plan_id": proposal["table_plan_id"],
1774
+ "table_recipe_id": proposal["table_recipe_id"],
1775
+ "recipe_version": proposal["recipe_version"],
1776
+ "recipe_digest": proposal["recipe_digest"],
1777
+ "recipe_proposal_id": recipe_proposal_id,
1778
+ }
1779
+
1780
+
1781
+ def approval_body(
1782
+ *,
1783
+ workspace_id: UUID,
1784
+ recipe_proposal_id: str,
1785
+ recipe_digest: str,
1786
+ preflight: Mapping[str, Any],
1787
+ ) -> dict[str, Any]:
1788
+ return {
1789
+ "schema_version": STUDIO_SCHEMA_VERSION,
1790
+ "workspace_id": str(workspace_id),
1791
+ "recipe_proposal_id": recipe_proposal_id,
1792
+ "recipe_digest": recipe_digest,
1793
+ "worker_policy_preflight": dict(preflight),
1794
+ }
1795
+
1796
+
1797
+ # --------------------------------------------------------------------------------------------
1798
+ # The command
1799
+ # --------------------------------------------------------------------------------------------
1800
+
1801
+
1802
+ def _session(args: argparse.Namespace) -> StudioSession:
1803
+ return open_studio_session(resolve_cloud_credentials())
1804
+
1805
+
1806
+ def _client(args: argparse.Namespace) -> StudioProposeClient:
1807
+ return StudioProposeClient(_session(args))
1808
+
1809
+
1810
+ def _through(args: argparse.Namespace) -> str:
1811
+ stage = getattr(args, "through", None) or "approval"
1812
+ if stage not in STAGES:
1813
+ allowed = ", ".join(STAGES)
1814
+ raise ProposeError("THIN_PROPOSE_STAGE_INVALID", f"--through is one of: {allowed}")
1815
+ return stage
1816
+
1817
+
1818
+ def _research_session(args: argparse.Namespace, *, stage: str) -> str | None:
1819
+ named = getattr(args, "research_session", None)
1820
+ if named is None:
1821
+ return None
1822
+ try:
1823
+ session_id = str(UUID(str(named)))
1824
+ except (ValueError, AttributeError) as error:
1825
+ raise ProposeError(
1826
+ "THIN_PROPOSE_REQUEST_INVALID", f"{named!r} is not a research session identifier"
1827
+ ) from error
1828
+ if STAGES.index(stage) < STAGES.index("proposal"):
1829
+ raise ProposeError(
1830
+ "THIN_PROPOSE_REQUEST_INVALID",
1831
+ "--research-session is sealed into the proposal, and --through "
1832
+ f"{stage} stops before one exists; drop the flag, or run through the proposal",
1833
+ )
1834
+ return session_id
1835
+
1836
+
1837
+ class _Bundle:
1838
+ """Every document a run needs, read and checked before a credential is resolved."""
1839
+
1840
+ def __init__(self, directory: Path, *, stage: str) -> None:
1841
+ self.directory = directory
1842
+ self.stage = stage
1843
+ self.digests: dict[str, str] = {}
1844
+ raw: dict[str, Any] = {}
1845
+ for step, names in DOCUMENTS.items():
1846
+ if STAGES.index(step) > STAGES.index(stage):
1847
+ continue
1848
+ for name in names:
1849
+ raw[name], self.digests[name] = _read_document(directory, name)
1850
+ self.dataset = check_dataset(raw["dataset.json"])
1851
+ self.question = check_question(raw["question.json"]) if "question.json" in raw else None
1852
+ self.sources = check_sources(raw["sources.json"]) if "sources.json" in raw else None
1853
+ self.plan = check_table_plan(raw["table_plan.json"]) if "table_plan.json" in raw else None
1854
+ self.recipe: dict[str, Any] | None = None
1855
+ self.canonical_recipe = b""
1856
+ self.recipe_digest = ""
1857
+ if "recipe.json" in raw:
1858
+ self.recipe, self.canonical_recipe, self.recipe_digest = check_recipe(
1859
+ raw["recipe.json"]
1860
+ )
1861
+ self.activation = (
1862
+ check_activation(raw["activation.json"]) if "activation.json" in raw else None
1863
+ )
1864
+ self.plan_content = (
1865
+ plan_content(self.plan, self.recipe)
1866
+ if self.plan is not None
1867
+ and "table_plan_id" not in self.plan
1868
+ and self.recipe is not None
1869
+ else None
1870
+ )
1871
+
1872
+ def reaches(self, step: str) -> bool:
1873
+ return STAGES.index(self.stage) >= STAGES.index(step)
1874
+
1875
+
1876
+ def propose(
1877
+ args: argparse.Namespace,
1878
+ *,
1879
+ client: StudioProposeClient | None = None,
1880
+ ) -> dict[str, Any]:
1881
+ """``mr-data propose``: author one hosted recipe proposal from a bundle, and open its approval.
1882
+
1883
+ Everything a client can settle on its own -- the bundle directory, the stage, the research
1884
+ coordinate, every document each reached stage needs and its shape, this client's own version
1885
+ -- is settled before a credential is resolved, so a mistyped path or a malformed document costs
1886
+ a sentence rather than a token and a 422 several routes in. The journal makes every stage
1887
+ idempotent: a re-run replays what was created, and a pinned document that changed is refused
1888
+ with the document named.
1889
+ """
1890
+
1891
+ directory = _bundle_directory(getattr(args, "bundle_dir", None))
1892
+ stage = _through(args)
1893
+ research_session_id = _research_session(args, stage=stage)
1894
+ announce = getattr(args, "json", False) is False
1895
+ bundle = _Bundle(directory, stage=stage)
1896
+ version = harness_version() if bundle.reaches("proposal") else None
1897
+
1898
+ selected = client or _client(args)
1899
+ workspace_id = selected.session.workspace_id
1900
+ journal = Journal(directory)
1901
+ journal.load_or_start(workspace_id=workspace_id)
1902
+ journal.assert_pinned_documents_unchanged(bundle.digests)
1903
+
1904
+ def note(line: str) -> None:
1905
+ if announce:
1906
+ print(line, flush=True)
1907
+
1908
+ # dataset -------------------------------------------------------------------------------
1909
+ adopted_dataset = "dataset_id" in bundle.dataset
1910
+ if adopted_dataset:
1911
+ dataset = _adopt(
1912
+ journal,
1913
+ selected,
1914
+ name="dataset",
1915
+ path=GET_DATASET_PATH.format(dataset_id=bundle.dataset["dataset_id"]),
1916
+ check=lambda record: None,
1917
+ missing="dataset.json adopts a Dataset this workspace does not hold",
1918
+ )
1919
+ else:
1920
+ dataset = _mutate(
1921
+ journal,
1922
+ selected,
1923
+ name="dataset",
1924
+ path=CREATE_DATASET_PATH,
1925
+ build=lambda: dataset_body(bundle.dataset, workspace_id=workspace_id),
1926
+ expected=201,
1927
+ document="dataset.json",
1928
+ refusal_remedy=_dataset_conflict_remedy,
1929
+ )
1930
+ dataset_id = _uuid_response(dataset, "dataset_id", "the dataset")
1931
+ journal.record_id("dataset_id", dataset_id)
1932
+ journal.pin(PINS["dataset"], bundle.digests)
1933
+ journal.record_stage("dataset")
1934
+ note(f"dataset {dataset_id} {'adopted' if adopted_dataset else 'registered'}")
1935
+ if stage == "dataset":
1936
+ return _stage_receipt(selected, journal, stage=stage)
1937
+
1938
+ # question + requirements ---------------------------------------------------------------
1939
+ question_document = bundle.question
1940
+ assert question_document is not None
1941
+ if "question_id" in question_document:
1942
+
1943
+ def _same_dataset(record: Mapping[str, Any]) -> None:
1944
+ if record.get("dataset_id") != dataset_id:
1945
+ raise ProposeError(
1946
+ "THIN_PROPOSE_ADOPT_SCOPE",
1947
+ "question.json adopts a question that belongs to another Dataset than this "
1948
+ "bundle's",
1949
+ )
1950
+
1951
+ question = _adopt(
1952
+ journal,
1953
+ selected,
1954
+ name="question",
1955
+ path=GET_QUESTION_PATH.format(question_id=question_document["question_id"]),
1956
+ check=_same_dataset,
1957
+ missing="question.json adopts a question this workspace does not hold",
1958
+ )
1959
+ question_id = _uuid_response(question, "question_id", "the question")
1960
+ requirements = _adopt(
1961
+ journal,
1962
+ selected,
1963
+ name="requirements",
1964
+ path=GET_REQUIREMENTS_PATH.format(question_id=question_id),
1965
+ check=lambda record: None,
1966
+ missing=(
1967
+ "question.json adopts a question whose requirements were never written; author "
1968
+ "the question and its requirements in question.json instead, or write the "
1969
+ "requirements for that question first"
1970
+ ),
1971
+ )
1972
+ else:
1973
+ if adopted_dataset and "question" not in journal.state["resources"]:
1974
+ _refuse_a_question_the_dataset_already_holds(
1975
+ selected, dataset_id=dataset_id, text=question_document["question"]
1976
+ )
1977
+ question = _mutate(
1978
+ journal,
1979
+ selected,
1980
+ name="question",
1981
+ path=CREATE_QUESTION_PATH,
1982
+ build=lambda: question_body(
1983
+ question_document, workspace_id=workspace_id, dataset_id=dataset_id
1984
+ ),
1985
+ expected=201,
1986
+ document="question.json",
1987
+ )
1988
+ question_id = _uuid_response(question, "question_id", "the question")
1989
+ requirements = _mutate(
1990
+ journal,
1991
+ selected,
1992
+ name="requirements",
1993
+ path=PUT_REQUIREMENTS_PATH.format(question_id=question_id),
1994
+ build=lambda: requirements_body(question_document, workspace_id=workspace_id),
1995
+ expected=200,
1996
+ document="question.json",
1997
+ method="PUT",
1998
+ if_match="*",
1999
+ )
2000
+ requirements_id = _uuid_response(requirements, "requirements_id", "the requirements")
2001
+ journal.record_id("question_id", question_id)
2002
+ journal.record_id("requirements_id", requirements_id)
2003
+ journal.pin(PINS["question"], bundle.digests)
2004
+ journal.record_stage("question")
2005
+ note(f"question {question_id} recorded, with its requirements")
2006
+ if stage == "question":
2007
+ return _stage_receipt(selected, journal, stage=stage)
2008
+
2009
+ # sources + connectors ------------------------------------------------------------------
2010
+ entries = bundle.sources
2011
+ assert entries is not None
2012
+ _refuse_a_registered_source_the_bundle_no_longer_names(journal, entries)
2013
+ if adopted_dataset:
2014
+ _refuse_sources_the_dataset_already_holds(
2015
+ selected, journal, dataset_id=dataset_id, entries=entries
2016
+ )
2017
+ source_ids: dict[str, str] = {}
2018
+ configurations: dict[str, dict[str, Any]] = {}
2019
+ for entry in entries:
2020
+ name = entry["name"]
2021
+ if "source_id" in entry:
2022
+
2023
+ def _in_dataset(record: Mapping[str, Any], *, name: str = name) -> None:
2024
+ if record.get("dataset_id") != dataset_id:
2025
+ raise ProposeError(
2026
+ "THIN_PROPOSE_ADOPT_SCOPE",
2027
+ f"sources.json {name!r} adopts a source registered under another Dataset "
2028
+ "than this bundle's",
2029
+ )
2030
+
2031
+ source = _adopt(
2032
+ journal,
2033
+ selected,
2034
+ name=f"source:{name}",
2035
+ path=GET_SOURCE_PATH.format(source_id=entry["source_id"]),
2036
+ check=_in_dataset,
2037
+ missing=f"sources.json {name!r} adopts a source this workspace does not hold",
2038
+ )
2039
+ source_id = _uuid_response(source, "source_id", f"the source {name!r}")
2040
+
2041
+ def _for_source(
2042
+ record: Mapping[str, Any], *, expected: str = source_id, name: str = name
2043
+ ) -> None:
2044
+ if record.get("source_id") != expected:
2045
+ raise ProposeError(
2046
+ "THIN_PROPOSE_ADOPT_SCOPE",
2047
+ f"sources.json {name!r} adopts a connector configuration registered for "
2048
+ "another source",
2049
+ )
2050
+
2051
+ configuration = _adopt(
2052
+ journal,
2053
+ selected,
2054
+ name=f"connector:{name}",
2055
+ path=GET_CONNECTOR_PATH.format(
2056
+ connector_configuration_id=entry["connector_configuration_id"]
2057
+ ),
2058
+ check=_for_source,
2059
+ missing=(
2060
+ f"sources.json {name!r} adopts a connector configuration this workspace "
2061
+ "does not hold"
2062
+ ),
2063
+ )
2064
+ else:
2065
+ source = _mutate(
2066
+ journal,
2067
+ selected,
2068
+ name=f"source:{name}",
2069
+ path=REGISTER_SOURCE_PATH,
2070
+ build=lambda entry=entry: source_body(
2071
+ entry, workspace_id=workspace_id, dataset_id=dataset_id
2072
+ ),
2073
+ expected=201,
2074
+ document="sources.json",
2075
+ )
2076
+ source_id = _uuid_response(source, "source_id", f"the source {name!r}")
2077
+ configuration = _mutate(
2078
+ journal,
2079
+ selected,
2080
+ name=f"connector:{name}",
2081
+ path=REGISTER_CONNECTOR_PATH,
2082
+ build=lambda entry=entry, source_id=source_id: connector_body(
2083
+ entry, workspace_id=workspace_id, source_id=source_id
2084
+ ),
2085
+ expected=201,
2086
+ document="sources.json",
2087
+ )
2088
+ source_ids[name] = source_id
2089
+ configurations[source_id] = configuration
2090
+ note(f"source {name!r} is {source_id}, with its connector")
2091
+ journal.record_id("sources", source_ids)
2092
+ journal.record_id("source_ids", list(source_ids.values()))
2093
+ journal.pin(PINS["sources"], bundle.digests)
2094
+ journal.record_stage("sources")
2095
+ if stage == "sources":
2096
+ return _stage_receipt(selected, journal, stage=stage)
2097
+
2098
+ # the recipe against the registered sources, and the session it was explored in ---------
2099
+ recipe = bundle.recipe
2100
+ assert recipe is not None
2101
+ _check_recipe_sources(recipe, source_ids=source_ids, configurations=configurations)
2102
+ research_run_id: str | None = None
2103
+ if research_session_id is not None:
2104
+ record = _lookup_session(
2105
+ selected, research_session_id, dataset_id=dataset_id, question_id=question_id
2106
+ )
2107
+ research_run_id = _uuid_response(record, "run_id", "the research session")
2108
+
2109
+ # plan ----------------------------------------------------------------------------------
2110
+ plan_document = bundle.plan
2111
+ assert plan_document is not None
2112
+ if "table_plan_id" in plan_document:
2113
+
2114
+ def _binds_this_bundle(record: Mapping[str, Any]) -> None:
2115
+ if record.get("dataset_id") != dataset_id or record.get("question_id") != question_id:
2116
+ raise ProposeError(
2117
+ "THIN_PROPOSE_ADOPT_SCOPE",
2118
+ "table_plan.json adopts a plan that binds another Dataset or question than "
2119
+ "this bundle's",
2120
+ )
2121
+ if record.get("requirements_id") != requirements_id:
2122
+ raise ProposeError(
2123
+ "THIN_PROPOSE_ADOPT_SCOPE",
2124
+ "table_plan.json adopts a plan bound to other requirements than this "
2125
+ "bundle's question carries",
2126
+ )
2127
+ planned_sources = record.get("source_ids")
2128
+ if not isinstance(planned_sources, list) or set(planned_sources) != set(
2129
+ source_ids.values()
2130
+ ):
2131
+ raise ProposeError(
2132
+ "THIN_PROPOSE_ADOPT_SCOPE",
2133
+ "table_plan.json adopts a plan whose sources are not the sources this bundle "
2134
+ "registered or adopted",
2135
+ )
2136
+
2137
+ plan = _adopt(
2138
+ journal,
2139
+ selected,
2140
+ name="plan",
2141
+ path=GET_TABLE_PLAN_PATH.format(table_plan_id=plan_document["table_plan_id"]),
2142
+ check=_binds_this_bundle,
2143
+ missing="table_plan.json adopts a plan this workspace does not hold",
2144
+ )
2145
+ else:
2146
+ content = bundle.plan_content
2147
+ assert content is not None
2148
+ plan = _mutate(
2149
+ journal,
2150
+ selected,
2151
+ name="plan",
2152
+ path=CREATE_TABLE_PLAN_PATH,
2153
+ build=lambda: table_plan_body(
2154
+ content,
2155
+ workspace_id=workspace_id,
2156
+ dataset_id=dataset_id,
2157
+ question_id=question_id,
2158
+ requirements_id=requirements_id,
2159
+ source_ids=list(source_ids.values()),
2160
+ ),
2161
+ expected=201,
2162
+ # The plan's digests derive from the recipe, so a recipe that moved after the plan
2163
+ # was created is a drift of this operation too, and the sentence names both.
2164
+ document="table_plan.json or recipe.json",
2165
+ )
2166
+ table_plan_id = _uuid_response(plan, "table_plan_id", "the table plan")
2167
+ journal.record_id("table_plan_id", table_plan_id)
2168
+ journal.pin(PINS["plan"], bundle.digests)
2169
+ journal.record_stage("plan")
2170
+ note(f"table plan {table_plan_id} in place")
2171
+ if stage == "plan":
2172
+ return _stage_receipt(selected, journal, stage=stage)
2173
+
2174
+ # proposal ------------------------------------------------------------------------------
2175
+ activation = bundle.activation
2176
+ assert activation is not None and version is not None
2177
+ _check_plan_describes_recipe(plan, recipe)
2178
+ bindings = sorted(
2179
+ (
2180
+ connector_authority(source_id=source_id, configuration=configurations[source_id])
2181
+ for source_id in source_ids.values()
2182
+ ),
2183
+ key=lambda item: item["source_id"],
2184
+ )
2185
+ command = proposal_body(
2186
+ recipe=recipe,
2187
+ canonical_bytes=bundle.canonical_recipe,
2188
+ recipe_digest=bundle.recipe_digest,
2189
+ activation=activation,
2190
+ workspace_id=workspace_id,
2191
+ dataset_id=dataset_id,
2192
+ table_plan_id=table_plan_id,
2193
+ source_authority_bindings=bindings,
2194
+ evidence=bootstrap_evidence(
2195
+ version=version,
2196
+ research_session_id=research_session_id,
2197
+ research_run_id=research_run_id,
2198
+ ),
2199
+ )
2200
+ proposal = _mutate(
2201
+ journal,
2202
+ selected,
2203
+ name="proposal",
2204
+ path=CREATE_RECIPE_PROPOSAL_PATH,
2205
+ build=lambda: command,
2206
+ expected=201,
2207
+ document=None,
2208
+ )
2209
+ recipe_proposal_id = _uuid_response(proposal, "recipe_proposal_id", "the recipe proposal")
2210
+ recipe_digest = _text_response(proposal, "recipe_digest", "the recipe proposal")
2211
+ if recipe_digest != bundle.recipe_digest and "recipe_digest" not in journal.state["ids"]:
2212
+ raise ProposeError(
2213
+ "THIN_PROPOSE_RESPONSE_INVALID",
2214
+ "Studio stored a recipe digest other than the one computed from recipe.json",
2215
+ )
2216
+ journal.record_id("recipe_proposal_id", recipe_proposal_id)
2217
+ journal.record_id("recipe_digest", recipe_digest)
2218
+ journal.pin(PINS["proposal"], bundle.digests)
2219
+ journal.record_stage("proposal")
2220
+ note(f"recipe proposal {recipe_proposal_id} holds recipe digest {recipe_digest}")
2221
+ if stage == "proposal":
2222
+ return _proposal_receipt(selected, journal, stage=stage)
2223
+
2224
+ # approval ------------------------------------------------------------------------------
2225
+ # An approval this bundle already opened is replayed from the journal. Otherwise the proposal
2226
+ # is read back first -- for the ETag the ask is pinned to, which the create never carries,
2227
+ # and because the proposal is where Studio records an approval whose answer was lost: one it
2228
+ # already holds is adopted rather than asked for again. The preflight is a stateless resolve
2229
+ # that Studio's fleet can move between runs, so it is asked for only when the ask is still to
2230
+ # be made, and a re-send after a lost answer is safe because the outcome was just read.
2231
+ if not isinstance(journal.state["resources"].get("approval"), dict):
2232
+ proposal_etag, held = _proposal_state(selected, journal, recipe_proposal_id, recipe_digest)
2233
+ if held is not None:
2234
+ _adopt_approval(journal, selected, held)
2235
+ else:
2236
+ preflight = selected.worker_policy_preflight(
2237
+ preflight_body(
2238
+ command, workspace_id=workspace_id, recipe_proposal_id=recipe_proposal_id
2239
+ )
2240
+ )
2241
+ _mutate(
2242
+ journal,
2243
+ selected,
2244
+ name="approval",
2245
+ path=REQUEST_RECIPE_APPROVAL_PATH.format(recipe_proposal_id=recipe_proposal_id),
2246
+ build=lambda: approval_body(
2247
+ workspace_id=workspace_id,
2248
+ recipe_proposal_id=recipe_proposal_id,
2249
+ recipe_digest=recipe_digest,
2250
+ preflight=preflight,
2251
+ ),
2252
+ expected=201,
2253
+ document=None,
2254
+ if_match=proposal_etag,
2255
+ lost_answer_recoverable=True,
2256
+ )
2257
+ approval = journal.state["resources"]["approval"]
2258
+ approval_request_id = _uuid_response(approval, "approval_request_id", "the approval request")
2259
+ journal.record_id("approval_request_id", approval_request_id)
2260
+ journal.record_stage("approval")
2261
+ note(f"approval request {approval_request_id} open")
2262
+ return _proposal_receipt(selected, journal, stage="approval")
2263
+
2264
+
2265
+ def _dataset_conflict_remedy(refusal: ThinLaneError) -> str | None:
2266
+ """What to do when Studio will not register a second Dataset of this name.
2267
+
2268
+ The commonest way to meet this is a bundle whose journal is gone: the Dataset was registered
2269
+ by an earlier run and nothing remembers it. The remedy is the id, not a new name.
2270
+ """
2271
+
2272
+ if getattr(refusal, "http_status", None) != 409:
2273
+ return None
2274
+ return (
2275
+ "a Dataset of this name already exists in the workspace; if an earlier run of this bundle "
2276
+ 'registered it, adopt it by id in dataset.json ({"dataset_id": "..."}) -- '
2277
+ "mr-data list names the workspace's datasets -- or choose another name"
2278
+ )
2279
+
2280
+
2281
+ def _refuse_a_question_the_dataset_already_holds(
2282
+ client: StudioProposeClient, *, dataset_id: str, text: str
2283
+ ) -> None:
2284
+ """Under an adopted Dataset, a question whose text the Dataset already holds is adopted, not
2285
+ registered again.
2286
+
2287
+ This is the lost-journal case one step on: the Dataset was adopted by id, and the question
2288
+ that went with it would be registered a second time in silence. The list is one page of the
2289
+ Dataset's own questions -- the bound Studio's list route has -- which is every question a
2290
+ bundle could have registered.
2291
+ """
2292
+
2293
+ for question in client._call_list(
2294
+ "/v3/questions", query={"dataset_id": dataset_id, "limit": 200}
2295
+ ):
2296
+ if question.get("text") == text and isinstance(question.get("question_id"), str):
2297
+ raise ProposeError(
2298
+ "THIN_PROPOSE_ADOPT_REQUIRED",
2299
+ f"the adopted Dataset already holds this question as {question['question_id']}; "
2300
+ 'adopt it in question.json ({"question_id": "..."}) rather than registering it '
2301
+ "again",
2302
+ )
2303
+
2304
+
2305
+ def _refuse_sources_the_dataset_already_holds(
2306
+ client: StudioProposeClient,
2307
+ journal: Journal,
2308
+ *,
2309
+ dataset_id: str,
2310
+ entries: Sequence[Mapping[str, Any]],
2311
+ ) -> None:
2312
+ """Under an adopted Dataset, a source whose name the Dataset already holds is adopted, not
2313
+ registered again -- the same lost-journal case, for the sources."""
2314
+
2315
+ authored = {
2316
+ entry["name"]
2317
+ for entry in entries
2318
+ if "source_id" not in entry and f"source:{entry['name']}" not in journal.state["resources"]
2319
+ }
2320
+ if not authored:
2321
+ return
2322
+ for source in client._call_list(REGISTER_SOURCE_PATH, query={"limit": 200}):
2323
+ if source.get("dataset_id") != dataset_id or source.get("name") not in authored:
2324
+ continue
2325
+ source_id = source.get("source_id")
2326
+ raise ProposeError(
2327
+ "THIN_PROPOSE_ADOPT_REQUIRED",
2328
+ f"the adopted Dataset already holds a source named {source.get('name')!r} as "
2329
+ f'{source_id}; adopt it in sources.json ({{"name": ..., "source_id": '
2330
+ f'"{source_id}", "connector_configuration_id": "..."}}) -- '
2331
+ f"GET /v3/connector-configurations?source_id={source_id} names its configuration -- "
2332
+ "rather than registering it again",
2333
+ )
2334
+
2335
+
2336
+ def _refuse_a_registered_source_the_bundle_no_longer_names(
2337
+ journal: Journal, entries: Sequence[Mapping[str, Any]]
2338
+ ) -> None:
2339
+ """A source this bundle registered and then stopped naming is a drift, not an orphan.
2340
+
2341
+ Sources are journaled by the name the author gave them, so reordering the list is harmless
2342
+ and renaming an entry after its source was registered is not: the registered source would be
2343
+ left behind and a second one registered under the new name. Naming the missing one is the
2344
+ whole of the remedy.
2345
+ """
2346
+
2347
+ named = {entry["name"] for entry in entries}
2348
+ registered = sorted(
2349
+ operation[len("source:") :]
2350
+ for operation, journaled in journal.state["operations"].items()
2351
+ if operation.startswith("source:") and journaled.get("state") in {"completed", "adopted"}
2352
+ )
2353
+ missing = [name for name in registered if name not in named]
2354
+ if missing:
2355
+ raise ProposeError(
2356
+ "THIN_PROPOSE_BUNDLE_DRIFTED",
2357
+ f"sources.json no longer names {', '.join(repr(name) for name in missing)}, which this "
2358
+ "bundle already registered; restore the entry under that name, or start a new bundle",
2359
+ )
2360
+
2361
+
2362
+ def _check_recipe_sources(
2363
+ recipe: Mapping[str, Any],
2364
+ *,
2365
+ source_ids: Mapping[str, str],
2366
+ configurations: Mapping[str, Mapping[str, Any]],
2367
+ ) -> None:
2368
+ """The recipe names exactly the registered sources, each under its connector's adapter."""
2369
+
2370
+ registered = set(source_ids.values())
2371
+ named = {item["source_id"]: item for item in recipe["sources"]}
2372
+ if set(named) != registered:
2373
+ raise ProposeError(
2374
+ "THIN_PROPOSE_SOURCE_INVENTORY",
2375
+ "recipe.json sources must reference exactly the registered source ids "
2376
+ f"({', '.join(sorted(registered))}); author recipe.json after "
2377
+ "`mr-data propose --through sources` and reference the ids the journal recorded",
2378
+ )
2379
+ for source_id, source in named.items():
2380
+ adapter_id = configurations[source_id].get("adapter_id")
2381
+ if source.get("adapter_id") != adapter_id:
2382
+ raise ProposeError(
2383
+ "THIN_PROPOSE_SOURCE_INVENTORY",
2384
+ f"recipe.json names source {source_id} under adapter {source.get('adapter_id')!r}, "
2385
+ f"but its connector is registered as {adapter_id!r}",
2386
+ )
2387
+
2388
+
2389
+ def _check_plan_describes_recipe(plan: Mapping[str, Any], recipe: Mapping[str, Any]) -> None:
2390
+ """The plan Studio holds -- created earlier, or adopted -- still describes this recipe."""
2391
+
2392
+ expected = {
2393
+ "plan_digest": execution_plan_digest(recipe),
2394
+ "validation_policy_digest": validation_policy_digest(recipe),
2395
+ }
2396
+ stale = sorted(field for field, value in expected.items() if plan.get(field) != value)
2397
+ if stale:
2398
+ raise ProposeError(
2399
+ "THIN_PROPOSE_PLAN_DIGEST",
2400
+ f"the table plan's {', '.join(stale)} no longer describes recipe.json: the plan was "
2401
+ "built from an earlier recipe; restore that recipe, or start a new bundle that adopts "
2402
+ "the Dataset, question and sources by id and creates a plan for this one",
2403
+ )
2404
+
2405
+
2406
+ def _lookup_session(
2407
+ client: StudioProposeClient, session_id: str, *, dataset_id: str, question_id: str
2408
+ ) -> dict[str, Any]:
2409
+ """The research session named as evidence, held to what Studio will hold it to.
2410
+
2411
+ Read before the plan exists rather than at the proposal, so a session that explored another
2412
+ Dataset, that failed, or that probed nothing is a sentence here and not a 409 after the plan
2413
+ has been created -- every one of these is a check Studio makes on the proposal.
2414
+ """
2415
+
2416
+ record = client.read(GET_SESSION_PATH.format(session_id=session_id))
2417
+ if record.get("session_id") != session_id:
2418
+ raise ProposeError(
2419
+ "THIN_PROPOSE_REQUEST_INVALID",
2420
+ "Studio answered about a different research session than the one named as evidence",
2421
+ )
2422
+ context = record.get("context")
2423
+ if not isinstance(context, Mapping) or context.get("dataset_id") != dataset_id:
2424
+ raise ProposeError(
2425
+ "THIN_PROPOSE_RESEARCH_SESSION",
2426
+ "--research-session names a session that explored another Dataset than this bundle's",
2427
+ )
2428
+ if context.get("question_id") != question_id:
2429
+ raise ProposeError(
2430
+ "THIN_PROPOSE_RESEARCH_SESSION",
2431
+ "--research-session names a session that explored another question than this bundle's",
2432
+ )
2433
+ if record.get("state") == "failed":
2434
+ raise ProposeError(
2435
+ "THIN_PROPOSE_RESEARCH_SESSION",
2436
+ "--research-session names a session that failed; its run holds no probe results",
2437
+ )
2438
+ submitted = record.get("probes_submitted")
2439
+ if type(submitted) is not int or submitted < 1:
2440
+ raise ProposeError(
2441
+ "THIN_PROPOSE_RESEARCH_SESSION",
2442
+ "--research-session names a session that submitted no probe; it explored nothing "
2443
+ "this recipe could rest on",
2444
+ )
2445
+ return record
2446
+
2447
+
2448
+ def _proposal_state(
2449
+ client: StudioProposeClient, journal: Journal, recipe_proposal_id: str, recipe_digest: str
2450
+ ) -> tuple[str, dict[str, Any] | None]:
2451
+ """The proposal's current ETag, and the approval Studio already holds for it, if any.
2452
+
2453
+ Read EVERY time the approval is still to be opened, never from the journal: the proposal's
2454
+ version moves when an approval is requested, so an ETag journaled before a lost answer is
2455
+ exactly the stale one a 412 would refuse forever. And the proposal is where Studio records the
2456
+ approval a lost ``POST`` did open -- ``approval_request_id`` on the proposal -- so reading it
2457
+ is how that outcome is learned rather than asked for twice.
2458
+ """
2459
+
2460
+ captured: dict[str, str] = {}
2461
+ record = client.recipe_proposal(recipe_proposal_id, response_headers=captured)
2462
+ if record.get("recipe_proposal_id") != recipe_proposal_id:
2463
+ raise ProposeError(
2464
+ "THIN_PROPOSE_RESPONSE_INVALID", "Studio answered about a different recipe proposal"
2465
+ )
2466
+ if record.get("recipe_digest") != recipe_digest:
2467
+ raise ProposeError(
2468
+ "THIN_PROPOSE_RESPONSE_INVALID",
2469
+ "the recipe proposal Studio holds no longer carries the recipe digest this bundle "
2470
+ "proposed",
2471
+ )
2472
+ etag = captured.get("etag")
2473
+ if not isinstance(etag, str) or not etag:
2474
+ raise ProposeError(
2475
+ "THIN_PROPOSE_RESPONSE_INVALID",
2476
+ "Studio answered the proposal without an ETag, and the approval ask must be pinned "
2477
+ "to the exact proposal version it was made against",
2478
+ )
2479
+ journal.state.setdefault("etags", {})["proposal"] = etag
2480
+ journal.save()
2481
+ held = record.get("approval_request_id")
2482
+ if isinstance(held, str) and _UUID_TEXT.fullmatch(held) is not None:
2483
+ return etag, client.read(GET_APPROVAL_PATH.format(approval_request_id=held))
2484
+ return etag, None
2485
+
2486
+
2487
+ def _adopt_approval(
2488
+ journal: Journal, client: StudioProposeClient, record: Mapping[str, Any]
2489
+ ) -> None:
2490
+ """Hold the approval Studio already opened for this proposal as this bundle's own."""
2491
+
2492
+ approval_request_id = _uuid_response(record, "approval_request_id", "the approval request")
2493
+ journal.state["resources"]["approval"] = dict(record)
2494
+ journal.state["operations"].pop("approval", None)
2495
+ journal.state["operations"]["approval"] = {
2496
+ "operation": "approval",
2497
+ "state": "adopted",
2498
+ "path": GET_APPROVAL_PATH.format(approval_request_id=approval_request_id),
2499
+ }
2500
+ journal.save()
2501
+
2502
+
2503
+ # --------------------------------------------------------------------------------------------
2504
+ # The receipts
2505
+ # --------------------------------------------------------------------------------------------
2506
+
2507
+
2508
+ def _common(client: StudioProposeClient, journal: Journal, *, stage: str) -> dict[str, Any]:
2509
+ ids = dict(journal.state.get("ids", {}))
2510
+ payload = {
2511
+ "schema_version": PROPOSE_SCHEMA,
2512
+ "lane": "hosted",
2513
+ "bundle": str(journal.path.parent.parent),
2514
+ "journal": str(journal.path),
2515
+ "through": stage,
2516
+ "workspace_id": journal.state["workspace_id"],
2517
+ "ids": ids,
2518
+ }
2519
+ dataset_id = ids.get("dataset_id")
2520
+ if isinstance(dataset_id, str):
2521
+ payload["dataset_dashboard_url"] = dataset_dashboard_url(
2522
+ client.session.cloud_url, dataset_id
2523
+ )
2524
+ return payload
2525
+
2526
+
2527
+ def _stage_receipt(client: StudioProposeClient, journal: Journal, *, stage: str) -> dict[str, Any]:
2528
+ """The receipt for a ``--through`` stop before the proposal exists.
2529
+
2530
+ Its status is ``authoring_stage_complete`` for every one of the four early stages, and the
2531
+ payload names which. After ``--through sources`` it carries the registered source ids under
2532
+ ``ids.sources`` so the agent can write ``recipe.json`` against them; that is the whole reason
2533
+ the stage exists.
2534
+ """
2535
+
2536
+ payload = _common(client, journal, stage=stage)
2537
+ payload["status"] = "authoring_stage_complete"
2538
+ next_stage = STAGES[STAGES.index(stage) + 1]
2539
+ payload["note"] = (
2540
+ f"Stopped after the {stage} stage. Re-run mr-data propose to continue through the "
2541
+ f"{next_stage} stage"
2542
+ + (
2543
+ "; the registered source ids are under ids.sources for recipe.json to reference"
2544
+ if stage == "sources"
2545
+ else ""
2546
+ )
2547
+ + "."
2548
+ )
2549
+ return payload
2550
+
2551
+
2552
+ def _proposal_receipt(
2553
+ client: StudioProposeClient, journal: Journal, *, stage: str
2554
+ ) -> dict[str, Any]:
2555
+ """The receipt once the proposal exists: the proposal, its digest, and what settles it.
2556
+
2557
+ When the approval was opened too, the receipt reads the approval back and says which of four
2558
+ things is true, because each is a different next step: the approval is still PENDING, and a
2559
+ person settles it at the dashboard address (exit 2); it is APPROVED and no Build has started,
2560
+ and ``mr-data build --hosted`` queues one; it is approved and the confirmation STARTED the
2561
+ Build in the same act -- Studio #124 -- and the run is named here because ``mr-data build``
2562
+ would refuse it rather than replay it; or it was declined (exit 2). Both commands carry
2563
+ ``--hosted``, because on a full install the bare names are the local lane's. Nothing here
2564
+ decides anything.
2565
+ """
2566
+
2567
+ payload = _common(client, journal, stage=stage)
2568
+ ids = journal.state.get("ids", {})
2569
+ recipe_proposal_id = ids.get("recipe_proposal_id")
2570
+ recipe_digest = ids.get("recipe_digest")
2571
+ payload["recipe_proposal_id"] = recipe_proposal_id
2572
+ payload["recipe_digest"] = recipe_digest
2573
+ build_command = (
2574
+ f"mr-data build --recipe-proposal {recipe_proposal_id} --recipe-digest {recipe_digest} "
2575
+ "--hosted"
2576
+ )
2577
+ if stage == "proposal":
2578
+ payload["status"] = "recipe_proposed"
2579
+ payload["build_command"] = build_command
2580
+ payload["note"] = (
2581
+ "The recipe proposal exists and nothing has approved it. Re-run mr-data propose to "
2582
+ "open its approval; then a person settles it and you build."
2583
+ )
2584
+ return payload
2585
+ approval_request_id = str(ids.get("approval_request_id"))
2586
+ payload["approval_request_id"] = approval_request_id
2587
+ payload["dashboard_url"] = approval_dashboard_url(client.session.cloud_url, approval_request_id)
2588
+ payload["never_decides"] = (
2589
+ "this command opened an approval; it did not grant one. A Table recipe is confirmed by a "
2590
+ "signed-in editor in the dashboard, which the command line cannot be."
2591
+ )
2592
+ approval = client.read(GET_APPROVAL_PATH.format(approval_request_id=approval_request_id))
2593
+ approval_status = approval.get("status")
2594
+ payload["approval_status"] = approval_status
2595
+ if approval_status == "pending":
2596
+ payload["status"] = "recipe_proposal_approval_requested"
2597
+ payload["approve_command"] = (
2598
+ f"mr-data recipe-approve --recipe {approval_request_id} "
2599
+ '--reason "why you are asking" --hosted'
2600
+ )
2601
+ payload["build_command"] = build_command
2602
+ payload["note"] = (
2603
+ "The proposal is open and waiting on a person. Open the dashboard address above, or "
2604
+ "run the recipe-approve command, so a human settles it; then build with the build "
2605
+ "command -- unless their confirmation started the Build, which re-running this "
2606
+ "command reports."
2607
+ )
2608
+ return payload
2609
+ if approval_status != "approved":
2610
+ payload["status"] = "recipe_proposal_not_approved"
2611
+ payload["note"] = (
2612
+ f"The approval settled as {approval_status}; nothing will build from this proposal. "
2613
+ "Author the recipe again in a new bundle that adopts the Dataset, question and "
2614
+ "sources by id."
2615
+ )
2616
+ return payload
2617
+ started = client.approval_start(approval_request_id)
2618
+ if started is None:
2619
+ payload["status"] = "recipe_proposal_approved"
2620
+ payload["build_command"] = build_command
2621
+ payload["note"] = (
2622
+ "A person approved the recipe and no Build has been started for it; queue the first "
2623
+ "hosted Build with the build command. It replays to the same run if one is started "
2624
+ "meanwhile."
2625
+ )
2626
+ return payload
2627
+ run = started.get("run") if isinstance(started.get("run"), Mapping) else {}
2628
+ run_id = run.get("run_id")
2629
+ if not isinstance(run_id, str):
2630
+ raise ProposeError(
2631
+ "THIN_PROPOSE_RESPONSE_INVALID", "Studio named a started Build without a run"
2632
+ )
2633
+ payload["status"] = "recipe_build_queued"
2634
+ payload["run_id"] = run_id
2635
+ payload["run_dashboard_url"] = run_dashboard_url(client.session.cloud_url, run_id)
2636
+ payload["watch_command"] = f"mr-data watch {run_id} --hosted"
2637
+ payload["note"] = (
2638
+ "The confirmation started the first hosted Build in the same act; there is nothing left "
2639
+ "to queue, and mr-data build would refuse rather than repeat it. Follow this run."
2640
+ )
2641
+ return payload
2642
+
2643
+
2644
+ def propose_exit_code(payload: Mapping[str, Any]) -> int:
2645
+ """``0`` when the requested work is done, ``2`` when a person still has to decide, or did not.
2646
+
2647
+ A ``--through`` stop before the approval, a proposal created without opening its approval, an
2648
+ approved proposal, and a Build the approval already started all did exactly what was asked
2649
+ and exit 0. An approval a human has not settled exits 2 and names the dashboard, the way
2650
+ ``mr-data approve`` exits 2 while an ask is pending -- because a script gating on this must
2651
+ not read "asked" as "approved" -- and so does one a human declined.
2652
+ """
2653
+
2654
+ if payload.get("status") in {
2655
+ "recipe_proposal_approval_requested",
2656
+ "recipe_proposal_not_approved",
2657
+ }:
2658
+ return 2
2659
+ return 0
2660
+
2661
+
2662
+ # --------------------------------------------------------------------------------------------
2663
+ # The command line, declared once for both parsers
2664
+ # --------------------------------------------------------------------------------------------
2665
+
2666
+ COMMAND = "propose"
2667
+ COMMAND_HELP = "author one hosted recipe proposal from a bundle and open its approval"
2668
+
2669
+
2670
+ def declare_arguments(parser: argparse.ArgumentParser) -> None:
2671
+ """Add ``propose``'s arguments to ``parser``.
2672
+
2673
+ Called by both front doors -- the thin router's parser and ``cli``'s -- so the argument surface
2674
+ is declared once and the two cannot disagree, the property ``note`` and the research commands
2675
+ have by the same construction.
2676
+ """
2677
+
2678
+ parser.add_argument(
2679
+ "bundle_dir",
2680
+ help="the directory holding the documents to author from; its .mr-data journal is written "
2681
+ "here",
2682
+ )
2683
+ parser.add_argument(
2684
+ "--through",
2685
+ choices=STAGES,
2686
+ help="stop after this stage instead of opening the approval; "
2687
+ "--through sources registers the sources first so recipe.json can reference their ids",
2688
+ )
2689
+ parser.add_argument(
2690
+ "--research-session",
2691
+ dest="research_session",
2692
+ metavar="UUID",
2693
+ help="the research session this recipe was explored in; its id and run are sealed into "
2694
+ "the hosted bootstrap evidence",
2695
+ )
2696
+
2697
+
2698
+ __all__ = [
2699
+ "COMMAND",
2700
+ "COMMAND_HELP",
2701
+ "CREATE_DATASET_PATH",
2702
+ "CREATE_QUESTION_PATH",
2703
+ "CREATE_RECIPE_PROPOSAL_PATH",
2704
+ "CREATE_TABLE_PLAN_PATH",
2705
+ "DEFAULT_SCHEDULE",
2706
+ "GET_CONNECTOR_PATH",
2707
+ "GET_DATASET_PATH",
2708
+ "GET_QUESTION_PATH",
2709
+ "GET_RECIPE_PROPOSAL_PATH",
2710
+ "GET_REQUIREMENTS_PATH",
2711
+ "GET_SOURCE_PATH",
2712
+ "GET_TABLE_PLAN_PATH",
2713
+ "HOSTED_BOOTSTRAP_ORIGIN",
2714
+ "JOURNAL_DIR",
2715
+ "JOURNAL_FILE",
2716
+ "JOURNAL_SCHEMA",
2717
+ "MAX_DOCUMENT_BYTES",
2718
+ "MAX_JOURNAL_BYTES",
2719
+ "PINS",
2720
+ "PROPOSE_SCHEMA",
2721
+ "PUT_REQUIREMENTS_PATH",
2722
+ "RECIPE_MEDIA_TYPE",
2723
+ "RECIPE_SCHEMA_VERSIONS",
2724
+ "REGISTER_CONNECTOR_PATH",
2725
+ "REGISTER_SOURCE_PATH",
2726
+ "REQUEST_RECIPE_APPROVAL_PATH",
2727
+ "STAGES",
2728
+ "WORKER_POLICY_PREFLIGHT_PATH",
2729
+ "Journal",
2730
+ "ProposeError",
2731
+ "StudioProposeClient",
2732
+ "approval_body",
2733
+ "backfill_policy_digest",
2734
+ "bootstrap_evidence",
2735
+ "check_activation",
2736
+ "check_dataset",
2737
+ "check_question",
2738
+ "check_recipe",
2739
+ "check_sources",
2740
+ "check_table_plan",
2741
+ "connector_authority",
2742
+ "connector_body",
2743
+ "dataset_body",
2744
+ "declare_arguments",
2745
+ "execution_plan_digest",
2746
+ "harness_version",
2747
+ "journal_ids",
2748
+ "plan_content",
2749
+ "preflight_body",
2750
+ "proposal_body",
2751
+ "propose",
2752
+ "propose_exit_code",
2753
+ "question_body",
2754
+ "read_journal",
2755
+ "requirements_body",
2756
+ "source_body",
2757
+ "table_plan_body",
2758
+ "validation_policy_digest",
2759
+ ]