mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,1663 @@
1
+ """``mr-data research-open``, ``research-probe``, ``research-status`` and ``research-close``.
2
+
3
+ WHAT WAS MISSING. ADR 0021 puts interactive source probing on the backend, and the backend has it:
4
+ Studio opens a research session, starts a warm session worker behind it, queues probes for that
5
+ worker, and appends every probe's progress to the session's own Run. The harness has the worker
6
+ (:mod:`mostlyright.data_harness.hosted_session_worker`) and the probe vocabulary
7
+ (:mod:`mostlyright.data_harness.session_probes`). What nothing had was the client: no command
8
+ opened a session, submitted a probe, or closed one, so the skill told the agent that this profile
9
+ submits no probe -- true while it was true, and no longer.
10
+
11
+ THE SESSION ROUTES, AND THE ONE STREAM.
12
+
13
+ 1. ``POST /v3/sessions`` opens a session against one Dataset and one question. The answer is fast
14
+ and the cold start is VISIBLE: a freshly opened session reports ``warming`` until its worker
15
+ takes the lease, and is never reported ready before one exists. ``research-open`` polls the
16
+ session and prints each state as it changes, because ``warming`` is a product state rather than
17
+ a stall, and a person watching a blank terminal for twenty seconds cannot tell the two apart.
18
+ 2. ``POST /v3/sessions/{id}/probes`` queues one probe. The body under ``probe_kind`` is opaque to
19
+ Studio and specified by this repository, so it is checked here, in the vocabulary's own words,
20
+ before an idempotency key is spent on it.
21
+ 3. The probe's progress rides the session run's ordinary event stream --
22
+ ``GET /v3/runs/{run_id}/events:stream``, the one ``mr-data watch`` already follows -- as a
23
+ ``source_probe_started`` / ``source_probe_settled`` pair whose facts carry Studio's own
24
+ ``ordinal``. That counter is how a frame on the stream is joined to the probe this command
25
+ submitted, and the settled frame is what ends the wait.
26
+ 4. ``GET /v3/sessions/{id}/probes/{probe_id}`` is the authoritative answer, read once the stream
27
+ says the probe settled. The stream is an unsealed progress adjunct and says a probe ENDED; the
28
+ resource says how, and carries the result.
29
+ 5. ``POST /v3/sessions/{id}:close`` ends the session, cancels every probe still queued, and takes
30
+ the session run terminal so the streams a client holds close cleanly.
31
+ 6. ``GET /v3/runs/{run_id}/session`` and ``POST /v3/runs/{run_id}/session:close`` let a caller
32
+ who only has the Run identifier printed by list, fleet, or status resolve (and, when requested,
33
+ close) its owning research session. Studio performs that ownership check directly; this client
34
+ never lists or scans sessions to infer it.
35
+
36
+ ⚠ WHAT A PROBE ANSWER IS, AND IS NOT. It is exploration. A probe result is opaque to Studio, is
37
+ never attestation evidence, produces no observation a Build could cite, and is not re-checked by
38
+ anything. This lane prints it so a person or an agent can decide what to build from; it cannot be
39
+ quoted by a receipt, and ``not_evidence`` says so in every answer that carries one.
40
+
41
+ ⚠ A 429 IS REPORTED, NEVER SLEPT THROUGH. A workspace holds a bounded number of live sessions,
42
+ and an open past the bound is ``AGENT_BUDGET_EXCEEDED`` with a ``Retry-After`` header -- a
43
+ refusal, not a queue. It is surfaced here as a typed refusal naming the window Studio stated, so
44
+ the caller decides what to do with the wait rather than finding this command holding a terminal
45
+ for the length of somebody else's session.
46
+ """
47
+
48
+ from __future__ import annotations
49
+
50
+ import argparse
51
+ import os
52
+ import re
53
+ import time
54
+ from collections.abc import Callable, Mapping, Sequence
55
+ from pathlib import Path
56
+ from typing import Any
57
+ from uuid import UUID
58
+
59
+ from mostlyright.data_harness import session_probes
60
+ from mostlyright.data_harness.canonical import CanonicalJSONError, parse_json
61
+ from mostlyright.data_harness.session_probes import ProbeVocabularyError, validate_probe_request
62
+ from mostlyright.data_harness.thin import THIN_SCHEMA_PREFIX
63
+ from mostlyright.data_harness.thin.runs import StudioApiClient, new_idempotency_key
64
+ from mostlyright.data_harness.thin.session import (
65
+ StudioSession,
66
+ dataset_dashboard_url,
67
+ dataset_research_dashboard_url,
68
+ open_studio_session,
69
+ run_dashboard_url,
70
+ )
71
+ from mostlyright.data_harness.thin.stream import (
72
+ RUN_PROGRESS_FRAME,
73
+ RunEventStream,
74
+ progress_facts,
75
+ render_frame,
76
+ )
77
+ from mostlyright.data_harness.thin.transport import ThinLaneError
78
+ from mostlyright.data_harness.ux.credentials import resolve_cloud_credentials
79
+ from mostlyright.data_harness.ux.plain_file import PlainFileRefusal, open_plain_file, read_bounded
80
+
81
+ RESEARCH_OPEN_SCHEMA = f"{THIN_SCHEMA_PREFIX}-research-open.v1"
82
+ RESEARCH_PROBE_SCHEMA = f"{THIN_SCHEMA_PREFIX}-research-probe.v1"
83
+ RESEARCH_STATUS_SCHEMA = f"{THIN_SCHEMA_PREFIX}-research-status.v1"
84
+ RESEARCH_CLOSE_SCHEMA = f"{THIN_SCHEMA_PREFIX}-research-close.v1"
85
+
86
+ #: The routes. Three of the five write: they open a session, queue a probe, and close a session.
87
+ #: None of them can cancel or retry the session's run, release anything, or record a decision --
88
+ #: Studio refuses ``:cancel`` and ``:retry`` on a session run by design, and nothing here reaches
89
+ #: either.
90
+ OPEN_SESSION_PATH = "/v3/sessions"
91
+ GET_SESSION_PATH = "/v3/sessions/{session_id}"
92
+ SUBMIT_PROBE_PATH = "/v3/sessions/{session_id}/probes"
93
+ GET_PROBE_PATH = "/v3/sessions/{session_id}/probes/{probe_id}"
94
+ CLOSE_SESSION_PATH = "/v3/sessions/{session_id}:close"
95
+ GET_SESSION_BY_RUN_PATH = "/v3/runs/{run_id}/session"
96
+ CLOSE_SESSION_BY_RUN_PATH = "/v3/runs/{run_id}/session:close"
97
+
98
+ #: ``research-session.schema.json#/$defs/session_state``, in the order a live session walks it.
99
+ SESSION_STATES = ("opening", "warming", "ready", "busy", "idle", "closed", "failed")
100
+
101
+ #: The states in which a worker holds the lease and a probe would be answered. ``idle`` is
102
+ #: ``ready`` that has been quiet for ``idle_after_seconds`` and is counting down to its close; it
103
+ #: is still a session a probe reaches, and submitting one is exactly what resets that countdown.
104
+ LIVE_SESSION_STATES = frozenset({"ready", "busy", "idle"})
105
+
106
+ #: The two terminal states. A session in either will never answer a probe again.
107
+ TERMINAL_SESSION_STATES = frozenset({"closed", "failed"})
108
+
109
+ #: ``research-session.schema.json#/$defs/probe_state``. The last four settle the probe; only the
110
+ #: third of them is an answer.
111
+ PROBE_STATES = ("queued", "claimed", "succeeded", "failed", "expired", "cancelled")
112
+ SETTLED_PROBE_STATES = frozenset({"succeeded", "failed", "expired", "cancelled"})
113
+
114
+ #: Studio's closed probe-kind enum, taken from the vocabulary module rather than restated, so the
115
+ #: command line's ``--kind`` choices cannot drift from what the worker answers.
116
+ PROBE_KINDS = session_probes.PROBE_KINDS
117
+
118
+ #: The one sentence every answer carrying a probe result says about it, for the reason
119
+ #: ``approvals`` says ``never_decides`` and ``narrative`` says ``not_evidence`` in every one of
120
+ #: theirs: the property has to survive being read by somebody who only sees one payload.
121
+ NOT_EVIDENCE = (
122
+ "A probe answer is exploration: opaque to Studio, re-checked by nothing, quoted by no receipt, "
123
+ "and never attestation evidence. It says what a warm worker saw; it does not prove anything "
124
+ "about a Build."
125
+ )
126
+
127
+ #: ``research-session.schema.json``: ``label`` is 1..200 characters and a close ``reason`` is
128
+ #: 1..512. Both are free text a person types, and both land in somebody else's session record, so
129
+ #: control characters are refused under the same rule ``approvals`` applies to ``--reason``.
130
+ MAX_LABEL_CHARS = 200
131
+ MAX_CLOSE_REASON_CHARS = 512
132
+ #: ``session_context.source_ids`` admits at most this many.
133
+ MAX_SOURCE_IDS = 32
134
+
135
+ #: How often ``research-open`` asks again while a session warms, and how long it keeps asking.
136
+ #: One second is the cadence the worker itself polls at, and nothing in a warming session changes
137
+ #: faster than that. The ceiling covers Studio's own policy with room to spare: a session may sit
138
+ #: in ``warming`` for ``warm_timeout_seconds`` (180 by default) before the five-minute sweep fails
139
+ #: it, so a session still warming at ten minutes is one Studio has already given up on, and the
140
+ #: honest thing is to stop asking and say which state it was really in.
141
+ OPEN_POLL_SECONDS = 1.0
142
+ MAX_OPEN_WAIT_SECONDS = 600.0
143
+
144
+ #: How long ``research-probe`` waits for its answer when the session states no usable idle
145
+ #: timeout, and the ceiling it will wait whatever the session states. The session's own
146
+ #: ``idle_timeout_seconds`` is the ordinary bound: a probe that has not settled when the session
147
+ #: would have closed for inactivity is not going to settle, and the close is what cancels it.
148
+ DEFAULT_PROBE_WAIT_SECONDS = 600.0
149
+ MAX_PROBE_WAIT_SECONDS = 3600.0
150
+
151
+ #: How often, at most, ``research-probe`` reads the probe resource while it follows the stream.
152
+ #: The stream is asked between frames and on every heartbeat, so on a quiet session the resource
153
+ #: is read about once per heartbeat (Studio's are at most fifteen seconds apart) and never more
154
+ #: often than this.
155
+ PROBE_POLL_SECONDS = 5.0
156
+
157
+ #: The one line of ``--request`` this lane reads whole. A probe request is bounded by the
158
+ #: vocabulary at 64 KiB of canonical JSON; a file four times that is refused before it is parsed.
159
+ MAX_REQUEST_FILE_BYTES = 4 * session_probes.PROBE_REQUEST_MAX_BYTES
160
+
161
+ #: Studio's ``Retry-After`` is delta-seconds in this contract. Bounded the way the token exchange
162
+ #: bounds its own: a window this lane does not understand is reported as no window rather than as
163
+ #: a number nobody checked.
164
+ MAX_RETRY_AFTER_SECONDS = 3600
165
+
166
+ #: The code Studio answers a full workspace with, as this lane's translation of it names it.
167
+ BUDGET_EXCEEDED_CODE = "THIN_STUDIO_AGENT_BUDGET_EXCEEDED"
168
+
169
+ _CONTROL = re.compile(r"[\x00-\x1f\x7f]")
170
+ _FAILURE_CODE = re.compile(r"^[A-Z][A-Z0-9_]{2,127}$")
171
+
172
+ #: ``common.schema.json#/$defs/uuid``, which is narrower than what :class:`uuid.UUID` parses: a
173
+ #: version nibble of 1..8 and an RFC 4122 variant. An identifier the contract would refuse with a
174
+ #: 422 is refused here with a sentence, before a credential is resolved.
175
+ _CONTRACT_UUID = re.compile(
176
+ r"^[0-9a-f]{8}-[0-9a-f]{4}-[1-8][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$"
177
+ )
178
+
179
+
180
+ class RetryAfterRefusal(ThinLaneError):
181
+ """A refusal Studio stamped with a retry window. Reported to the caller, never slept through.
182
+
183
+ Distinct from a plain :class:`ThinLaneError` in one attribute, ``retry_after_seconds``, which
184
+ the router carries into the error object so a script reads the window as a number rather than
185
+ parsing it out of a sentence. The sentence names it too, for the person.
186
+ """
187
+
188
+ def __init__(self, code: str, detail: str, *, retry_after_seconds: int | None) -> None:
189
+ super().__init__(code, detail)
190
+ self.retry_after_seconds = retry_after_seconds
191
+
192
+
193
+ def _failure_code(value: Any, fallback: str) -> str:
194
+ """Keep one exact terminal worker code without rendering untrusted diagnostics."""
195
+
196
+ return value if isinstance(value, str) and _FAILURE_CODE.fullmatch(value) else fallback
197
+
198
+
199
+ class StudioResearchClient(StudioApiClient):
200
+ """The research-session route family, including direct Run-to-session resolution.
201
+
202
+ Three write. ``open_session`` causes a session and a worker, ``submit_probe`` causes one row
203
+ a worker will answer, and the close calls end both. The two Run-coordinate calls are only
204
+ Studio's direct ownership resolver and close operation, never a client-side session scan.
205
+ What none of them can do is steer the session's Run: Studio refuses ``:cancel`` and ``:retry``
206
+ on a run of kind ``research_session`` so that taking the run terminal behind its session's back
207
+ is impossible, and this class formats neither path.
208
+ """
209
+
210
+ def open_session(
211
+ self, body: Mapping[str, Any], *, idempotency_key: str | None = None
212
+ ) -> dict[str, Any]:
213
+ """Open one session and return it as Studio recorded it, ``warming`` or not yet."""
214
+
215
+ headers: dict[str, str] = {}
216
+ try:
217
+ return self._call(
218
+ "POST",
219
+ OPEN_SESSION_PATH,
220
+ body=body,
221
+ extra_headers={
222
+ "Idempotency-Key": idempotency_key or new_idempotency_key("mr-data-research")
223
+ },
224
+ expected=(201,),
225
+ response_headers=headers,
226
+ )
227
+ except ThinLaneError as refusal:
228
+ if refusal.code != BUDGET_EXCEEDED_CODE:
229
+ raise
230
+ raise _budget_refusal(refusal, headers, undone="opened") from refusal
231
+
232
+ def get_session(self, session_id: str) -> dict[str, Any]:
233
+ return self._call("GET", GET_SESSION_PATH.format(session_id=session_id), expected=(200,))
234
+
235
+ def get_session_by_run(self, run_id: str) -> dict[str, Any]:
236
+ """Resolve exactly one research session through Studio's Run ownership check."""
237
+
238
+ return self._call("GET", GET_SESSION_BY_RUN_PATH.format(run_id=run_id), expected=(200,))
239
+
240
+ def submit_probe(
241
+ self,
242
+ session_id: str,
243
+ body: Mapping[str, Any],
244
+ *,
245
+ idempotency_key: str | None = None,
246
+ ) -> dict[str, Any]:
247
+ """Queue one probe. Studio answers 202 with the ResearchProbe, durable from that moment."""
248
+
249
+ headers: dict[str, str] = {}
250
+ try:
251
+ return self._call(
252
+ "POST",
253
+ SUBMIT_PROBE_PATH.format(session_id=session_id),
254
+ body=body,
255
+ extra_headers={
256
+ "Idempotency-Key": idempotency_key or new_idempotency_key("mr-data-probe")
257
+ },
258
+ expected=(202,),
259
+ response_headers=headers,
260
+ )
261
+ except ThinLaneError as refusal:
262
+ if refusal.code != BUDGET_EXCEEDED_CODE:
263
+ raise
264
+ raise _budget_refusal(refusal, headers, undone="submitted") from refusal
265
+
266
+ def get_probe(self, session_id: str, probe_id: str) -> dict[str, Any]:
267
+ return self._call(
268
+ "GET",
269
+ GET_PROBE_PATH.format(session_id=session_id, probe_id=probe_id),
270
+ expected=(200,),
271
+ )
272
+
273
+ def close_session(
274
+ self,
275
+ session_id: str,
276
+ body: Mapping[str, Any],
277
+ *,
278
+ idempotency_key: str | None = None,
279
+ ) -> dict[str, Any]:
280
+ """End one session. Closing a closed session is the same answer, not a conflict."""
281
+
282
+ return self._call(
283
+ "POST",
284
+ CLOSE_SESSION_PATH.format(session_id=session_id),
285
+ body=body,
286
+ extra_headers={
287
+ "Idempotency-Key": idempotency_key or new_idempotency_key("mr-data-research-close")
288
+ },
289
+ expected=(200,),
290
+ )
291
+
292
+ def close_session_by_run(
293
+ self,
294
+ run_id: str,
295
+ body: Mapping[str, Any],
296
+ *,
297
+ idempotency_key: str | None = None,
298
+ ) -> dict[str, Any]:
299
+ """Ask Studio to resolve and atomically close the session owned by this displayed Run."""
300
+
301
+ return self._call(
302
+ "POST",
303
+ CLOSE_SESSION_BY_RUN_PATH.format(run_id=run_id),
304
+ body=body,
305
+ extra_headers={
306
+ "Idempotency-Key": idempotency_key
307
+ or new_idempotency_key("mr-data-research-close-by-run")
308
+ },
309
+ expected=(200,),
310
+ )
311
+
312
+
313
+ def _budget_refusal(
314
+ refusal: ThinLaneError, headers: Mapping[str, str], *, undone: str
315
+ ) -> RetryAfterRefusal:
316
+ """Studio's capacity refusal, with the window it stated carried rather than slept on.
317
+
318
+ ``undone`` names what this call would have caused and did not -- a session, a probe -- because
319
+ the one sentence a caller reads after a 429 is the one saying nothing happened.
320
+ """
321
+
322
+ raw = headers.get("retry-after", "")
323
+ window = int(raw) if raw.isdecimal() and 1 <= int(raw) <= MAX_RETRY_AFTER_SECONDS else None
324
+ said = (
325
+ f"; Studio asked for a retry after {window} seconds"
326
+ if window is not None
327
+ else "; Studio stated no usable retry window"
328
+ )
329
+ return RetryAfterRefusal(
330
+ refusal.code,
331
+ f"{refusal.detail}{said}. Nothing was {undone}, and this command did not wait",
332
+ retry_after_seconds=window,
333
+ )
334
+
335
+
336
+ # --------------------------------------------------------------------------------------------
337
+ # The bodies this lane sends, with the contract applied here
338
+ # --------------------------------------------------------------------------------------------
339
+
340
+
341
+ def _identifier(value: Any, *, what: str, flag: str) -> str:
342
+ """Refuse anything that is not an identifier the contract admits, before it reaches the wire.
343
+
344
+ Checked before a credential is resolved and a token minted, the ordering every other lane here
345
+ establishes: a typo costs a sentence, not a round trip. The check is the contract's own
346
+ pattern rather than :class:`uuid.UUID`, which also parses the nil UUID, braces, and URN
347
+ prefixes that Studio would answer with a 422.
348
+ """
349
+
350
+ try:
351
+ normalized = str(UUID(str(value)))
352
+ except (ValueError, AttributeError) as error:
353
+ raise ThinLaneError(
354
+ "THIN_REQUEST_INVALID", f"{value!r} is not {what}; {flag} names one by its identifier"
355
+ ) from error
356
+ if not isinstance(value, str) or _CONTRACT_UUID.fullmatch(value) is None:
357
+ raise ThinLaneError(
358
+ "THIN_REQUEST_INVALID",
359
+ f"{value!r} is not {what}; {flag} names one by its identifier, as lowercase "
360
+ "hyphenated hex with a version of 1 to 8",
361
+ )
362
+ return normalized
363
+
364
+
365
+ def _free_text(value: Any, *, flag: str, maximum: int) -> str:
366
+ if not isinstance(value, str) or not 1 <= len(value) <= maximum or _CONTROL.search(value):
367
+ raise ThinLaneError(
368
+ "THIN_REQUEST_INVALID",
369
+ f"{flag} is one line of at most {maximum} characters, with no control characters",
370
+ )
371
+ return value
372
+
373
+
374
+ def session_context(
375
+ *,
376
+ dataset_id: Any,
377
+ question_id: Any,
378
+ table_id: Any = None,
379
+ source_ids: Sequence[Any] | None = None,
380
+ ) -> dict[str, Any]:
381
+ """The ``session_context`` a session is opened against, checked before anything is minted.
382
+
383
+ A session binds the Dataset and the question being explored -- and, optionally, a Table
384
+ already under construction and the sources it may probe -- and never a recipe activation.
385
+ Probing is what a recipe gets written FROM, so there is nothing here that could name one.
386
+ """
387
+
388
+ context: dict[str, Any] = {
389
+ "dataset_id": _identifier(dataset_id, what="a Dataset identifier", flag="--dataset"),
390
+ "question_id": _identifier(question_id, what="a question identifier", flag="--question"),
391
+ }
392
+ if table_id is not None:
393
+ context["table_id"] = _identifier(table_id, what="a Table identifier", flag="--table")
394
+ if not source_ids:
395
+ # ⚠ REQUIRED HERE, THOUGH THE CONTRACT MAKES IT OPTIONAL. `context.source_ids` is the
396
+ # permission: a probe naming a source outside it is refused `RESEARCH_SOURCE_FORBIDDEN`,
397
+ # so a session opened with none is a warm worker that can answer nothing, billed for
398
+ # every second it waits. Refusing the open is cheaper than the session it would produce.
399
+ raise ThinLaneError(
400
+ "THIN_REQUEST_INVALID",
401
+ "--source names at least one source the session may probe; a session opened with "
402
+ "none refuses every probe, because the sources named at the open are the permission",
403
+ )
404
+ # Repeats are folded rather than refused: `--source X --source X` is one source twice, and
405
+ # the contract's `uniqueItems` would otherwise turn a duplicated shell variable into a 422
406
+ # about a list the person cannot see.
407
+ selected = list(
408
+ dict.fromkeys(
409
+ _identifier(item, what="a source identifier", flag="--source") for item in source_ids
410
+ )
411
+ )
412
+ if len(selected) > MAX_SOURCE_IDS:
413
+ raise ThinLaneError(
414
+ "THIN_REQUEST_INVALID",
415
+ f"a session may name at most {MAX_SOURCE_IDS} sources; {len(selected)} were given",
416
+ )
417
+ context["source_ids"] = selected
418
+ return context
419
+
420
+
421
+ def bundle_session_context(value: Any) -> dict[str, Any]:
422
+ """The Dataset, question and sources a bundle's propose journal recorded, for ``--bundle``.
423
+
424
+ ``propose --through sources`` registers those three and records their ids, so opening a session
425
+ to probe them is naming the bundle rather than repeating what it already holds. The journal is
426
+ read through the authoring lane's own reader -- the path, the byte ceiling and the schema check
427
+ are its -- imported here rather than at module scope because that lane imports this one.
428
+ A bundle whose proposal has not reached the sources stage has nothing to probe yet, and the
429
+ refusal says so.
430
+ """
431
+
432
+ from mostlyright.data_harness.thin import propose
433
+
434
+ ids = propose.journal_ids(value)
435
+ if ids.get("dataset_id") is None or ids.get("question_id") is None:
436
+ raise ThinLaneError(
437
+ "THIN_REQUEST_INVALID",
438
+ "--bundle's propose journal records no Dataset and question yet; run "
439
+ "mr-data propose --through question against it first",
440
+ )
441
+ source_ids = ids.get("source_ids")
442
+ if not isinstance(source_ids, list) or not source_ids:
443
+ # A session with no sources can answer no probe, so a bundle whose sources are not yet
444
+ # registered has nothing to open a session against. Refuse it here, naming the stage that
445
+ # registers them, rather than opening a warm worker that would refuse every probe.
446
+ raise ThinLaneError(
447
+ "THIN_REQUEST_INVALID",
448
+ "--bundle's propose journal records no registered sources yet; run "
449
+ "mr-data propose --through sources against it first, then probe what it registered",
450
+ )
451
+ return {
452
+ "dataset_id": ids.get("dataset_id"),
453
+ "question_id": ids.get("question_id"),
454
+ "source_ids": list(source_ids),
455
+ }
456
+
457
+
458
+ def session_label(value: Any) -> str | None:
459
+ """The display label, or ``None`` when none was given. Never an identity."""
460
+
461
+ if value is None:
462
+ return None
463
+ return _free_text(value, flag="--label", maximum=MAX_LABEL_CHARS)
464
+
465
+
466
+ def open_session_body(
467
+ *,
468
+ workspace_id: UUID,
469
+ dataset_id: Any,
470
+ question_id: Any,
471
+ table_id: Any = None,
472
+ source_ids: Sequence[Any] | None = None,
473
+ label: Any = None,
474
+ ) -> dict[str, Any]:
475
+ """The ``OpenResearchSessionCommand`` this lane sends."""
476
+
477
+ body: dict[str, Any] = {
478
+ "workspace_id": str(workspace_id),
479
+ "context": session_context(
480
+ dataset_id=dataset_id,
481
+ question_id=question_id,
482
+ table_id=table_id,
483
+ source_ids=source_ids,
484
+ ),
485
+ }
486
+ labelled = session_label(label)
487
+ if labelled is not None:
488
+ body["label"] = labelled
489
+ return body
490
+
491
+
492
+ def probe_request(probe_kind: Any, request: Any) -> tuple[str, dict[str, Any]]:
493
+ """The kind and the normalized body, checked against the vocabulary before anything is sent.
494
+
495
+ ⚠ THE REQUEST IS CHECKED BEFORE IT IS SENT, AND WHAT IS SENT IS THE NORMALIZED FORM. Studio
496
+ stores a probe body uninterpreted and digests it; the worker then refuses one outside the
497
+ vocabulary as a ``failed`` probe with a code. Checking here turns that into a sentence at the
498
+ terminal, before a credential is resolved, a session's idempotency key is spent and a worker's
499
+ time is taken. The body that goes over the wire is the one :func:`validate_probe_request`
500
+ returns -- defaults filled in -- so two callers asking the same question get the same
501
+ ``request_digest`` whether or not one of them spelled out ``scan_rows``.
502
+ """
503
+
504
+ try:
505
+ kind = session_probes.probe_kind(probe_kind)
506
+ normalized = validate_probe_request(kind, request)
507
+ except ProbeVocabularyError as refusal:
508
+ raise ThinLaneError(
509
+ "THIN_REQUEST_INVALID", f"the probe request is outside the vocabulary: {refusal.detail}"
510
+ ) from refusal
511
+ except CanonicalJSONError as refusal:
512
+ raise ThinLaneError(
513
+ "THIN_REQUEST_INVALID",
514
+ f"the probe request is not canonical JSON: {refusal.detail}; a probe carries integers, "
515
+ "strings, booleans and null, never a fractional number",
516
+ ) from refusal
517
+ return kind, normalized
518
+
519
+
520
+ def submit_probe_body(
521
+ *,
522
+ workspace_id: UUID,
523
+ session_id: str,
524
+ probe_kind: Any,
525
+ request: Any,
526
+ ) -> dict[str, Any]:
527
+ """The ``SubmitResearchProbeCommand`` this lane sends, with the vocabulary applied."""
528
+
529
+ kind, normalized = probe_request(probe_kind, request)
530
+ return {
531
+ "workspace_id": str(workspace_id),
532
+ "session_id": session_id,
533
+ "probe_kind": kind,
534
+ "request": normalized,
535
+ }
536
+
537
+
538
+ def close_reason(value: Any) -> str | None:
539
+ """The close reason, or ``None`` when none was given."""
540
+
541
+ if value is None:
542
+ return None
543
+ return _free_text(value, flag="--reason", maximum=MAX_CLOSE_REASON_CHARS)
544
+
545
+
546
+ def close_session_body(
547
+ *, workspace_id: UUID, session_id: str, reason: Any = None
548
+ ) -> dict[str, Any]:
549
+ """The ``CloseResearchSessionCommand`` this lane sends."""
550
+
551
+ body: dict[str, Any] = {"workspace_id": str(workspace_id), "session_id": session_id}
552
+ stated = close_reason(reason)
553
+ if stated is not None:
554
+ body["reason"] = stated
555
+ return body
556
+
557
+
558
+ # --------------------------------------------------------------------------------------------
559
+ # What the answers carry
560
+ # --------------------------------------------------------------------------------------------
561
+
562
+
563
+ def session_receipt(record: Mapping[str, Any]) -> dict[str, Any]:
564
+ """The ResearchSession fields a person reads, exactly as Studio reported them.
565
+
566
+ The three lease coordinates -- ``attempt_id``, ``worker_generation``, ``lease_nonce_digest``
567
+ -- and the image digest are left to ``--json`` readers of the raw record: they are facts about
568
+ the worker's fence, and a person deciding whether to probe needs the state, the clock and the
569
+ context, not the fence.
570
+ """
571
+
572
+ receipt = {
573
+ key: record[key]
574
+ for key in (
575
+ "session_id",
576
+ "run_id",
577
+ "state",
578
+ "context",
579
+ "label",
580
+ "warming_since",
581
+ "ready_at",
582
+ "idle_after_seconds",
583
+ "idle_timeout_seconds",
584
+ "probes_submitted",
585
+ "created_at",
586
+ "last_activity_at",
587
+ "last_worker_contact_at",
588
+ "closed_at",
589
+ "close_reason",
590
+ "failure_code",
591
+ )
592
+ if key in record
593
+ }
594
+ if "failure_code" in receipt:
595
+ receipt["failure_code"] = _failure_code(receipt["failure_code"], "RESEARCH_SESSION_FAILED")
596
+ return receipt
597
+
598
+
599
+ def probe_receipt(record: Mapping[str, Any]) -> dict[str, Any]:
600
+ """The ResearchProbe coordinates, without the answer, which the payload carries on its own."""
601
+
602
+ receipt = {
603
+ key: record[key]
604
+ for key in (
605
+ "probe_id",
606
+ "session_id",
607
+ "run_id",
608
+ "ordinal",
609
+ "probe_kind",
610
+ "request",
611
+ "request_digest",
612
+ "state",
613
+ "result_digest",
614
+ "failure_code",
615
+ "submitted_at",
616
+ "claimed_at",
617
+ "settled_at",
618
+ )
619
+ if key in record
620
+ }
621
+ if "failure_code" in receipt:
622
+ receipt["failure_code"] = _failure_code(receipt["failure_code"], "PROBE_FAILED")
623
+ return receipt
624
+
625
+
626
+ def _session(args: argparse.Namespace) -> StudioSession:
627
+ return open_studio_session(resolve_cloud_credentials())
628
+
629
+
630
+ def _client(args: argparse.Namespace) -> StudioResearchClient:
631
+ return StudioResearchClient(_session(args))
632
+
633
+
634
+ def _session_identifier(record: Mapping[str, Any]) -> str:
635
+ value = record.get("session_id")
636
+ try:
637
+ return str(UUID(str(value)))
638
+ except (ValueError, AttributeError) as error:
639
+ raise ThinLaneError(
640
+ "THIN_RESPONSE_INVALID", "Studio answered with a session that has no identifier"
641
+ ) from error
642
+
643
+
644
+ def _run_identifier(record: Mapping[str, Any]) -> str:
645
+ value = record.get("run_id")
646
+ try:
647
+ return str(UUID(str(value)))
648
+ except (ValueError, AttributeError) as error:
649
+ raise ThinLaneError(
650
+ "THIN_RESPONSE_INVALID", "Studio answered with a session that names no run"
651
+ ) from error
652
+
653
+
654
+ def _dataset_identifier(record: Mapping[str, Any]) -> str:
655
+ context = record.get("context")
656
+ value = context.get("dataset_id") if isinstance(context, Mapping) else None
657
+ try:
658
+ return str(UUID(str(value)))
659
+ except (ValueError, AttributeError) as error:
660
+ raise ThinLaneError(
661
+ "THIN_RESPONSE_INVALID", "Studio answered with a session that names no Dataset"
662
+ ) from error
663
+
664
+
665
+ def _dashboard_urls(cloud_url: str, record: Mapping[str, Any], run_id: str) -> dict[str, str]:
666
+ """All visual coordinates for one research session, derived from Studio's own binding."""
667
+
668
+ dataset_id = _dataset_identifier(record)
669
+ return {
670
+ "dashboard_url": run_dashboard_url(cloud_url, run_id),
671
+ "dataset_dashboard_url": dataset_dashboard_url(cloud_url, dataset_id),
672
+ "research_dashboard_url": dataset_research_dashboard_url(cloud_url, dataset_id, run_id),
673
+ }
674
+
675
+
676
+ def _bind_session(record: Mapping[str, Any], *, session_id: str) -> None:
677
+ """Refuse a session record that is not the session this command is about.
678
+
679
+ Compared on every poll, the way the acquisition lane compares its immutable coordinates: a
680
+ record whose identifier moved under a reading is a different session, and printing its state
681
+ as this one's would be reporting somebody else's worker.
682
+ """
683
+
684
+ if record.get("session_id") != session_id:
685
+ raise ThinLaneError(
686
+ "THIN_SESSION_BINDING", "Studio answered about a different research session"
687
+ )
688
+
689
+
690
+ def _refusal_receipt(refusal: ThinLaneError) -> dict[str, Any]:
691
+ """One typed refusal, as a member of a payload that still carries what was done before it."""
692
+
693
+ receipt: dict[str, Any] = {"code": refusal.code, "detail": refusal.detail}
694
+ window = getattr(refusal, "retry_after_seconds", None)
695
+ if window is not None:
696
+ receipt["retry_after_seconds"] = window
697
+ return receipt
698
+
699
+
700
+ def _state_line(session_id: str, record: Mapping[str, Any]) -> str:
701
+ """One line a person reads when the session's state changes, and what the state means."""
702
+
703
+ state = str(record.get("state"))
704
+ if state == "opening":
705
+ return f"session {session_id} opening: the record exists; the worker is not started yet"
706
+ if state == "warming":
707
+ return (
708
+ f"session {session_id} warming: the worker is starting. This is the visible cold "
709
+ "start, not a stall; it ends when the worker takes its lease"
710
+ )
711
+ if state == "ready":
712
+ return f"session {session_id} ready: a worker holds the lease and no probe is waiting"
713
+ if state == "busy":
714
+ return f"session {session_id} busy: a probe is queued or being answered"
715
+ if state == "idle":
716
+ timeout = record.get("idle_timeout_seconds")
717
+ return (
718
+ f"session {session_id} idle: ready, and quiet; it closes on its own after "
719
+ f"{timeout} seconds without a probe"
720
+ )
721
+ if state == "failed":
722
+ code = _failure_code(record.get("failure_code"), "RESEARCH_SESSION_FAILED")
723
+ return f"session {session_id} failed: {code}"
724
+ if state == "closed":
725
+ reason = record.get("close_reason") or "no reason recorded"
726
+ return f"session {session_id} closed: {reason}"
727
+ return f"session {session_id} {state}: a state this client has no sentence for"
728
+
729
+
730
+ # --------------------------------------------------------------------------------------------
731
+ # research-open
732
+ # --------------------------------------------------------------------------------------------
733
+
734
+
735
+ def research_open(
736
+ args: argparse.Namespace,
737
+ *,
738
+ client: StudioResearchClient | None = None,
739
+ sleep: Callable[[float], None] = time.sleep,
740
+ clock: Callable[[], float] = time.monotonic,
741
+ ) -> dict[str, Any]:
742
+ """``mr-data research-open``: open one session and watch it warm.
743
+
744
+ Every transition is printed as it happens -- ``opening``, ``warming``, ``ready`` -- so the
745
+ cold start reads as what it is. ``--no-wait`` returns as soon as Studio has recorded the
746
+ session, which is the shape a script that polls ``research-status`` itself wants.
747
+ """
748
+
749
+ announce = getattr(args, "json", False) is False
750
+ # Everything a client can settle on its own is settled before a credential is resolved: a
751
+ # mistyped identifier costs a sentence, not a token and a 422.
752
+ bundle = getattr(args, "bundle", None)
753
+ dataset = getattr(args, "dataset", None)
754
+ question = getattr(args, "question", None)
755
+ if bundle is not None and (dataset is not None or question is not None):
756
+ # Refused rather than silently preferring one: a bundle's journal and a flag that names a
757
+ # different Dataset are two answers to one question, and a session opened against the
758
+ # wrong one would probe sources the recipe will never name.
759
+ raise ThinLaneError(
760
+ "THIN_REQUEST_INVALID",
761
+ "--bundle names the Dataset and question from the bundle's journal; give it alone, "
762
+ "or give --dataset and --question without it",
763
+ )
764
+ if bundle is not None:
765
+ from_bundle = bundle_session_context(bundle)
766
+ dataset = from_bundle["dataset_id"]
767
+ question = from_bundle["question_id"]
768
+ else:
769
+ from_bundle = {}
770
+ if dataset is None or question is None:
771
+ raise ThinLaneError(
772
+ "THIN_REQUEST_INVALID",
773
+ "research-open needs a Dataset and a question: give --dataset and --question, or "
774
+ "--bundle whose propose journal records them",
775
+ )
776
+ # A bundle's sources and any named explicitly are one allowlist; `session_context` folds
777
+ # duplicates, so naming a source the bundle already carries is harmless.
778
+ sources = [*from_bundle.get("source_ids", []), *(getattr(args, "source", None) or [])]
779
+ context = session_context(
780
+ dataset_id=dataset,
781
+ question_id=question,
782
+ table_id=getattr(args, "table", None),
783
+ source_ids=sources or None,
784
+ )
785
+ label = session_label(getattr(args, "label", None))
786
+ selected = client or _client(args)
787
+ body: dict[str, Any] = {"workspace_id": str(selected.session.workspace_id), "context": context}
788
+ if label is not None:
789
+ body["label"] = label
790
+ record = selected.open_session(body)
791
+ session_id = _session_identifier(record)
792
+ run_id = _run_identifier(record)
793
+ waited = not getattr(args, "no_wait", False)
794
+ if announce:
795
+ print(_state_line(session_id, record), flush=True)
796
+ refusal: ThinLaneError | None = None
797
+ if waited:
798
+ try:
799
+ record = _await_live(
800
+ selected,
801
+ session_id,
802
+ record,
803
+ sleep=sleep,
804
+ clock=clock,
805
+ on_state=(lambda line: print(line, flush=True)) if announce else None,
806
+ )
807
+ except ThinLaneError as interrupted:
808
+ # The session exists whatever stopped the polling, and its identifier must not go
809
+ # down with the refusal: the record last read stands, the refusal rides beside it.
810
+ refusal = interrupted
811
+ state = record.get("state")
812
+ ready = state in LIVE_SESSION_STATES
813
+ if state in TERMINAL_SESSION_STATES:
814
+ status = "research_session_ended"
815
+ elif ready:
816
+ status = "research_session_ready"
817
+ else:
818
+ status = "research_session_opened"
819
+ payload: dict[str, Any] = {
820
+ "schema_version": RESEARCH_OPEN_SCHEMA,
821
+ "status": status,
822
+ "lane": "hosted",
823
+ "session_id": session_id,
824
+ "run_id": run_id,
825
+ "ready": ready,
826
+ "waited": waited,
827
+ "watch_command": f"mr-data watch {run_id} --hosted",
828
+ **_dashboard_urls(selected.session.cloud_url, record, run_id),
829
+ "session": session_receipt(record),
830
+ "studio": selected.session.to_receipt(),
831
+ }
832
+ if refusal is not None:
833
+ payload["refusal"] = _refusal_receipt(refusal)
834
+ payload["note"] = (
835
+ f"The session was opened and then this command could not keep reading it: "
836
+ f"{refusal.detail}. The session is unaffected; read it with mr-data research-status "
837
+ f"{session_id}."
838
+ )
839
+ elif status == "research_session_opened":
840
+ payload["note"] = f"The session is {state} and no worker has taken it yet. " + (
841
+ "Read it again with mr-data research-status; a probe submitted now waits in its "
842
+ "queue until the worker arrives."
843
+ if not waited
844
+ else f"Nothing answered within {int(MAX_OPEN_WAIT_SECONDS)} seconds, which is "
845
+ "past the warm timeout Studio enforces; read it again with mr-data "
846
+ "research-status, and open a new session if it has been failed."
847
+ )
848
+ elif status == "research_session_ended":
849
+ payload["note"] = (
850
+ "The session ended before a worker answered. It will not answer a probe; open a new "
851
+ "session."
852
+ )
853
+ return payload
854
+
855
+
856
+ def _await_live(
857
+ client: StudioResearchClient,
858
+ session_id: str,
859
+ record: Mapping[str, Any],
860
+ *,
861
+ sleep: Callable[[float], None],
862
+ clock: Callable[[], float],
863
+ on_state: Callable[[str], None] | None,
864
+ ) -> dict[str, Any]:
865
+ """Poll the session until a worker holds it, it ends, or the ceiling arrives.
866
+
867
+ The ceiling is reported rather than raised: a session still warming at the deadline is a fact
868
+ about the session, and the payload says which state it was really in. Reading the session is
869
+ not activity -- ``last_activity_at`` moves on open, submission and settlement only -- so
870
+ polling cannot keep a session alive, and nothing here needs to avoid doing so.
871
+ """
872
+
873
+ observed = dict(record)
874
+ deadline = clock() + MAX_OPEN_WAIT_SECONDS
875
+ last_state = observed.get("state")
876
+ while True:
877
+ state = observed.get("state")
878
+ if state in LIVE_SESSION_STATES or state in TERMINAL_SESSION_STATES:
879
+ return observed
880
+ if clock() >= deadline:
881
+ return observed
882
+ sleep(OPEN_POLL_SECONDS)
883
+ observed = client.get_session(session_id)
884
+ _bind_session(observed, session_id=session_id)
885
+ if observed.get("state") != last_state:
886
+ last_state = observed.get("state")
887
+ if on_state is not None:
888
+ on_state(_state_line(session_id, observed))
889
+
890
+
891
+ def research_open_exit_code(payload: Mapping[str, Any]) -> int:
892
+ """``0`` when a worker holds the session, or when the caller asked not to wait for one.
893
+
894
+ A session that ended, and a wait that ran out while it was still warming, both exit 2: the
895
+ caller asked for a session it could probe and did not get one, and the payload says which of
896
+ the two it was.
897
+ """
898
+
899
+ if payload.get("ready") is True:
900
+ return 0
901
+ if payload.get("status") == "research_session_opened" and payload.get("waited") is False:
902
+ return 0
903
+ return 2
904
+
905
+
906
+ # --------------------------------------------------------------------------------------------
907
+ # research-probe
908
+ # --------------------------------------------------------------------------------------------
909
+
910
+
911
+ def _read_request(args: argparse.Namespace) -> Any:
912
+ """The probe request, from ``--request`` or from ``--request-file``, and exactly one of them.
913
+
914
+ Parsed with the strict reader rather than the foreign one: the body goes out as canonical
915
+ JSON, which refuses a fractional number, so a request that carries one is refused here with a
916
+ sentence rather than three calls later with a traceback out of the encoder.
917
+ """
918
+
919
+ inline = getattr(args, "request", None)
920
+ named = getattr(args, "request_file", None)
921
+ if (inline is None) == (named is None):
922
+ raise ThinLaneError(
923
+ "THIN_REQUEST_INVALID",
924
+ "a probe carries one request body: give it inline with --request, or name a file "
925
+ "with --request-file, and not both",
926
+ )
927
+ if inline is not None:
928
+ raw: bytes | str = inline
929
+ else:
930
+ target = Path(str(named)).expanduser()
931
+ try:
932
+ descriptor, size = open_plain_file(target)
933
+ except PlainFileRefusal as refusal:
934
+ raise ThinLaneError(
935
+ "THIN_REQUEST_INVALID",
936
+ f"--request-file did not name one readable file: {refusal.path}",
937
+ ) from refusal
938
+ except OSError as error:
939
+ raise ThinLaneError(
940
+ "THIN_REQUEST_INVALID", f"--request-file could not be opened: {target}"
941
+ ) from error
942
+ try:
943
+ if size > MAX_REQUEST_FILE_BYTES:
944
+ raise ThinLaneError(
945
+ "THIN_REQUEST_INVALID",
946
+ f"--request-file holds {size} bytes; a probe request is at most "
947
+ f"{session_probes.PROBE_REQUEST_MAX_BYTES} bytes of canonical JSON",
948
+ )
949
+ raw = read_bounded(descriptor, max_bytes=MAX_REQUEST_FILE_BYTES + 1)
950
+ finally:
951
+ os.close(descriptor)
952
+ if len(raw) > MAX_REQUEST_FILE_BYTES:
953
+ raise ThinLaneError(
954
+ "THIN_REQUEST_INVALID",
955
+ f"--request-file grew past {MAX_REQUEST_FILE_BYTES} bytes while it was being read",
956
+ )
957
+ try:
958
+ parsed = parse_json(raw)
959
+ except CanonicalJSONError as refusal:
960
+ raise ThinLaneError(
961
+ "THIN_REQUEST_INVALID",
962
+ f"the probe request is not strict JSON: {refusal.detail}; a probe carries integers, "
963
+ "strings, booleans and null, never a fractional number",
964
+ ) from refusal
965
+ if not isinstance(parsed, dict):
966
+ raise ThinLaneError("THIN_REQUEST_INVALID", "the probe request must be a JSON object")
967
+ return parsed
968
+
969
+
970
+ def _probe_coordinates(record: Mapping[str, Any]) -> tuple[str, int]:
971
+ """The probe's identifier and its ordinal, which is how the stream names it."""
972
+
973
+ try:
974
+ probe_id = str(UUID(str(record.get("probe_id"))))
975
+ except (ValueError, AttributeError) as error:
976
+ raise ThinLaneError(
977
+ "THIN_RESPONSE_INVALID", "Studio queued a probe without an identifier"
978
+ ) from error
979
+ ordinal = record.get("ordinal")
980
+ if type(ordinal) is not int or ordinal < 1:
981
+ raise ThinLaneError(
982
+ "THIN_RESPONSE_INVALID",
983
+ "Studio queued a probe without the ordinal the stream would name it by",
984
+ )
985
+ return probe_id, ordinal
986
+
987
+
988
+ def _probe_wait_seconds(record: Mapping[str, Any]) -> float:
989
+ declared = record.get("idle_timeout_seconds")
990
+ if type(declared) is int and declared >= 1:
991
+ return float(min(declared, MAX_PROBE_WAIT_SECONDS))
992
+ return DEFAULT_PROBE_WAIT_SECONDS
993
+
994
+
995
+ def _session_sources(record: Mapping[str, Any]) -> list[str]:
996
+ context = record.get("context")
997
+ sources = context.get("source_ids") if isinstance(context, Mapping) else None
998
+ return [item for item in sources if isinstance(item, str)] if isinstance(sources, list) else []
999
+
1000
+
1001
+ def _require_session_source(record: Mapping[str, Any], source_id: Any) -> None:
1002
+ """Refuse a probe naming a source the session was not opened against, before it is queued.
1003
+
1004
+ The session's ``context.source_ids`` IS the permission: Studio's worker door refuses any other
1005
+ source as ``RESEARCH_SOURCE_FORBIDDEN``, after the probe has been queued, claimed and given a
1006
+ worker's time. The session record is already in hand here, so the refusal costs a sentence.
1007
+ """
1008
+
1009
+ allowed = _session_sources(record)
1010
+ if source_id in allowed:
1011
+ return
1012
+ raise ThinLaneError(
1013
+ "THIN_RESEARCH_SOURCE_FORBIDDEN",
1014
+ f"source {source_id} is not one this session may probe; the session was opened against "
1015
+ + (", ".join(allowed) if allowed else "no source at all")
1016
+ + ". Open a session naming the source with --source, or probe one of those",
1017
+ )
1018
+
1019
+
1020
+ #: What the worker's own ``outcome`` on the settled frame means for the caller, in the words a
1021
+ #: caller acts on. The probe resource settles as ``failed`` for four of the five; the unsealed
1022
+ #: frame is where the refinement lives, and it is the difference between "ask again in a moment"
1023
+ #: and "never ask this again".
1024
+ _OUTCOME_ADVICE = {
1025
+ "refused": (
1026
+ "the request was refused rather than answered, and re-asking the same probe refuses the "
1027
+ "same way; change the probe, or the source it names"
1028
+ ),
1029
+ "timed_out": (
1030
+ "the worker did not finish inside its ceiling; the same probe against a smaller scan may "
1031
+ "succeed"
1032
+ ),
1033
+ "failed": "the worker attempted it and broke; the code above is the worker's own",
1034
+ }
1035
+
1036
+ _SESSION_ACQUISITION_THROTTLED = "SESSION_ACQUISITION_THROTTLED"
1037
+ _SESSION_ACQUISITION_THROTTLE_WIRE_CODE = "AGENT_BUDGET_EXCEEDED"
1038
+
1039
+
1040
+ def _worker_outcome_receipt(facts: Mapping[str, Any]) -> dict[str, Any]:
1041
+ """Keep a bounded user-facing display code beside, never instead of, Studio's wire code."""
1042
+
1043
+ outcome = facts.get("outcome")
1044
+ succeeded = outcome == "succeeded" and facts.get("code") == "OK"
1045
+ wire_code = "OK" if succeeded else _failure_code(facts.get("code"), "PROBE_FAILED")
1046
+ display_code = (
1047
+ "OK"
1048
+ if succeeded and facts.get("display_code") == "OK"
1049
+ else _failure_code(facts.get("display_code"), wire_code)
1050
+ )
1051
+ receipt: dict[str, Any] = {
1052
+ "outcome": outcome,
1053
+ "code": display_code,
1054
+ "wire_code": wire_code,
1055
+ "bytes": facts.get("bytes"),
1056
+ }
1057
+ retry_after_seconds = facts.get("retry_after_seconds")
1058
+ if wire_code == _SESSION_ACQUISITION_THROTTLE_WIRE_CODE:
1059
+ receipt["code"] = _SESSION_ACQUISITION_THROTTLED
1060
+ if (
1061
+ receipt["code"] == _SESSION_ACQUISITION_THROTTLED
1062
+ and wire_code == _SESSION_ACQUISITION_THROTTLE_WIRE_CODE
1063
+ and type(retry_after_seconds) is int
1064
+ and 1 <= retry_after_seconds <= MAX_RETRY_AFTER_SECONDS
1065
+ ):
1066
+ receipt["retry_after_seconds"] = retry_after_seconds
1067
+ return receipt
1068
+
1069
+
1070
+ def _settles(
1071
+ ordinal: int, announce: Callable[[str], None] | None, seen: dict[str, Any]
1072
+ ) -> Callable[[Mapping[str, Any], str], bool]:
1073
+ """The observer that ends the wait: the ``source_probe_settled`` frame carrying this ordinal.
1074
+
1075
+ Every frame on the session run passes through here -- replayed ones included, because the
1076
+ stream is read from its first record -- and only the two about this probe are shown. Matching
1077
+ on the ordinal rather than on a probe identifier is the design: the progress facts carry
1078
+ Studio's counter and no identifier, and the counter is unique within the session, so a replay
1079
+ of an earlier probe's settlement cannot be mistaken for this one's. The settled frame's facts
1080
+ are kept in ``seen``: they carry the worker's refined ``outcome``, which the probe resource
1081
+ collapses to ``failed`` and which is the fact a caller acts on.
1082
+ """
1083
+
1084
+ def observe(body: Mapping[str, Any], frame_name: str) -> bool:
1085
+ named = progress_facts(body)
1086
+ if named is None:
1087
+ return False
1088
+ name, facts = named
1089
+ if name not in {"source_probe_started", "source_probe_settled"}:
1090
+ return False
1091
+ if facts.get("ordinal") != ordinal:
1092
+ return False
1093
+ if announce is not None:
1094
+ announce(render_frame(body, frame_name or RUN_PROGRESS_FRAME))
1095
+ if name == "source_probe_settled":
1096
+ seen.update(facts)
1097
+ return True
1098
+ return False
1099
+
1100
+ return observe
1101
+
1102
+
1103
+ def _polls_the_resource(
1104
+ client: StudioResearchClient,
1105
+ session_id: str,
1106
+ probe_id: str,
1107
+ *,
1108
+ clock: Callable[[], float],
1109
+ latest: dict[str, Any],
1110
+ ) -> Callable[[], bool]:
1111
+ """The check the stream asks between frames: has the probe settled on the resource itself?
1112
+
1113
+ ⚠ THE FRAME IS THE FAST PATH, NOT THE ONLY PATH. The worker writes a probe's result to Studio
1114
+ BEFORE it emits the settled frame, and its emitter is best effort: it disarms itself after a
1115
+ bounded number of progress records per attempt, and an append the callback refused is dropped
1116
+ rather than retried. A wait that ended only on the frame would, on a long session, wait out
1117
+ its whole ceiling for an answer Studio had already recorded. So the resource is read on a
1118
+ bounded cadence -- at most once every :data:`PROBE_POLL_SECONDS`, asked between frames, on
1119
+ every heartbeat of a quiet stream, and before every reconnect -- and the wait ends as soon as
1120
+ either the frame or the resource says the probe settled. The last record read is kept in
1121
+ ``latest`` so the caller does not read it twice.
1122
+ """
1123
+
1124
+ asked_at = clock()
1125
+
1126
+ def stop() -> bool:
1127
+ nonlocal asked_at
1128
+ now = clock()
1129
+ if now - asked_at < PROBE_POLL_SECONDS:
1130
+ return False
1131
+ asked_at = now
1132
+ try:
1133
+ record = client.get_probe(session_id, probe_id)
1134
+ except ThinLaneError as refusal:
1135
+ if refusal.code == "THIN_STUDIO_UNREACHABLE":
1136
+ # One poll that did not get through is not a reason to stop listening to a
1137
+ # stream that is still delivering; the next cadence asks again, and a Studio that
1138
+ # stays unreachable is met at the read that follows the wait, as a refusal the
1139
+ # payload carries.
1140
+ return False
1141
+ raise
1142
+ if record.get("probe_id") != probe_id:
1143
+ raise ThinLaneError("THIN_SESSION_BINDING", "Studio answered about a different probe")
1144
+ latest.clear()
1145
+ latest.update(record)
1146
+ return record.get("state") in SETTLED_PROBE_STATES
1147
+
1148
+ return stop
1149
+
1150
+
1151
+ def research_probe(
1152
+ args: argparse.Namespace,
1153
+ *,
1154
+ client: StudioResearchClient | None = None,
1155
+ sleep: Callable[[float], None] = time.sleep,
1156
+ clock: Callable[[], float] = time.monotonic,
1157
+ ) -> dict[str, Any]:
1158
+ """``mr-data research-probe``: ask the warm worker one question and print its answer.
1159
+
1160
+ The request is checked against the vocabulary before a credential is resolved, the session is
1161
+ read so the run to follow, the wait ceiling and the sources the probe may name are Studio's
1162
+ rather than guessed, the probe is queued, and then the session run's stream is followed from
1163
+ its first record until either the frame that says this probe settled or the probe resource,
1164
+ read between frames, says so. The answer is then read from the probe resource itself.
1165
+
1166
+ ⚠ WHAT ``succeeded`` MEANS. It means the worker answered, and ``result`` is what it answered.
1167
+ Every other settled state -- ``failed``, ``expired``, ``cancelled`` -- exits 2 with the state
1168
+ named, the way ``approve`` treats every decision that is not the one asked for: a script
1169
+ gating on an answer must not read the absence of one as a yes.
1170
+
1171
+ ⚠ A REFUSAL AFTER THE 202 KEEPS THE PROBE'S COORDINATES. Once Studio has queued the probe, a
1172
+ stream this credential may not read, a stream that could not be re-established, or a frame
1173
+ the parser refused is not a reason to lose the identifier of a probe that is still running:
1174
+ the payload is emitted with the refusal under ``refusal`` and exits 2, and the probe is read
1175
+ back with ``research-status``.
1176
+ """
1177
+
1178
+ session_id = _identifier(
1179
+ getattr(args, "session_id", None), what="a research session identifier", flag="SESSION_ID"
1180
+ )
1181
+ kind, normalized = probe_request(getattr(args, "kind", None), _read_request(args))
1182
+ announce = getattr(args, "json", False) is False
1183
+ selected = client or _client(args)
1184
+ body = {
1185
+ "workspace_id": str(selected.session.workspace_id),
1186
+ "session_id": session_id,
1187
+ "probe_kind": kind,
1188
+ "request": normalized,
1189
+ }
1190
+ session_record = selected.get_session(session_id)
1191
+ _bind_session(session_record, session_id=session_id)
1192
+ run_id = _run_identifier(session_record)
1193
+ if session_record.get("state") in TERMINAL_SESSION_STATES:
1194
+ raise ThinLaneError(
1195
+ "THIN_RESEARCH_SESSION_ENDED",
1196
+ f"session {session_id} is {session_record.get('state')} and will not answer a probe; "
1197
+ "open a new session with mr-data research-open",
1198
+ )
1199
+ _require_session_source(session_record, normalized.get("source_id"))
1200
+ probe = selected.submit_probe(session_id, body)
1201
+ probe_id, ordinal = _probe_coordinates(probe)
1202
+ waited = not getattr(args, "no_wait", False)
1203
+ if announce:
1204
+ print(
1205
+ f"probe {ordinal} ({body['probe_kind']}) queued on session {session_id} as {probe_id}",
1206
+ flush=True,
1207
+ )
1208
+ payload: dict[str, Any] = {
1209
+ "schema_version": RESEARCH_PROBE_SCHEMA,
1210
+ "status": "research_probe_submitted",
1211
+ "lane": "hosted",
1212
+ "session_id": session_id,
1213
+ "run_id": run_id,
1214
+ "probe_id": probe_id,
1215
+ "waited": waited,
1216
+ "settled": False,
1217
+ "watch_command": f"mr-data watch {run_id} --hosted",
1218
+ **_dashboard_urls(selected.session.cloud_url, session_record, run_id),
1219
+ "probe": probe_receipt(probe),
1220
+ "result": None,
1221
+ "not_evidence": NOT_EVIDENCE,
1222
+ "studio": selected.session.to_receipt(),
1223
+ }
1224
+ if not waited:
1225
+ payload["note"] = (
1226
+ "The probe is queued and this command did not wait for it. Read its state with "
1227
+ f"mr-data research-status {session_id} --probe {probe_id}."
1228
+ )
1229
+ return payload
1230
+ stream = RunEventStream(selected.session, run_id, transport=selected.transport, sleep=sleep)
1231
+ settled_facts: dict[str, Any] = {}
1232
+ polled: dict[str, Any] = {}
1233
+ try:
1234
+ progress = stream.follow(
1235
+ from_seq=0,
1236
+ observe=_settles(
1237
+ ordinal, (lambda line: print(line, flush=True)) if announce else None, settled_facts
1238
+ ),
1239
+ stop=_polls_the_resource(selected, session_id, probe_id, clock=clock, latest=polled),
1240
+ deadline=clock() + _probe_wait_seconds(session_record),
1241
+ clock=clock,
1242
+ )
1243
+ if polled.get("state") in SETTLED_PROBE_STATES:
1244
+ # The poll that ended the wait already holds the settled record; reading it again
1245
+ # would be one more round trip for the same bytes.
1246
+ settled_probe = polled
1247
+ else:
1248
+ settled_probe = selected.get_probe(session_id, probe_id)
1249
+ if settled_probe.get("probe_id") != probe_id:
1250
+ raise ThinLaneError(
1251
+ "THIN_SESSION_BINDING", "Studio answered about a different probe"
1252
+ )
1253
+ except ThinLaneError as refusal:
1254
+ payload["refusal"] = _refusal_receipt(refusal)
1255
+ payload["note"] = (
1256
+ f"The probe was queued and then this command could not follow it: {refusal.detail}. "
1257
+ f"Nothing about the probe changed; read it with mr-data research-status {session_id} "
1258
+ f"--probe {probe_id}."
1259
+ )
1260
+ return payload
1261
+ state = settled_probe.get("state")
1262
+ payload["probe"] = probe_receipt(settled_probe)
1263
+ payload["stream"] = progress.to_receipt()
1264
+ # The worker's own reading of how the probe ended, from the unsealed frame that ended the
1265
+ # wait. Reported beside the resource rather than instead of it: the resource is the authority
1266
+ # on WHETHER the probe settled, and the frame is the only place that says HOW.
1267
+ worker_outcome = _worker_outcome_receipt(settled_facts) if settled_facts else None
1268
+ # The terminal resource is authoritative and already passes through the
1269
+ # bounded-code gate above. This is an unsealed SSE observation used only
1270
+ # to explain *how* the wait ended, so it gets the same treatment before it
1271
+ # enters a CLI receipt; otherwise a malicious worker event could render a
1272
+ # token or multiline diagnostic beside a perfectly safe resource record.
1273
+ payload["worker_outcome"] = worker_outcome
1274
+ outcome = settled_facts.get("outcome")
1275
+ if state in SETTLED_PROBE_STATES:
1276
+ payload["settled"] = True
1277
+ # Which of the two said so first. The frame is the fast path; the resource is the one
1278
+ # that is always there.
1279
+ payload["settled_by"] = "stream" if settled_facts else "resource"
1280
+ if state == "succeeded":
1281
+ payload["status"] = "research_probe_answered"
1282
+ payload["result"] = settled_probe.get("result")
1283
+ elif state in SETTLED_PROBE_STATES:
1284
+ payload["status"] = "research_probe_settled"
1285
+ wire_failure_code = _failure_code(settled_probe.get("failure_code"), "PROBE_FAILED")
1286
+ display_failure_code = wire_failure_code
1287
+ retry_after_seconds: int | None = None
1288
+ if worker_outcome is not None and worker_outcome.get("wire_code") == wire_failure_code:
1289
+ display_failure_code = worker_outcome["code"]
1290
+ retry_after_seconds = worker_outcome.get("retry_after_seconds")
1291
+ payload["failure_code"] = display_failure_code
1292
+ if display_failure_code != wire_failure_code:
1293
+ payload["wire_failure_code"] = wire_failure_code
1294
+ if retry_after_seconds is not None:
1295
+ payload["retry_after_seconds"] = retry_after_seconds
1296
+ payload["note"] = f"The probe settled as {state} and has no answer. " + _settled_why(
1297
+ state, outcome, display_failure_code, retry_after_seconds
1298
+ )
1299
+ else:
1300
+ payload["note"] = (
1301
+ f"The probe is still {state}. "
1302
+ + _unsettled_why(progress, bool(settled_facts))
1303
+ + f"; read it again with mr-data research-status {session_id} --probe {probe_id}."
1304
+ )
1305
+ return payload
1306
+
1307
+
1308
+ def _settled_why(
1309
+ state: str,
1310
+ outcome: Any,
1311
+ display_code: str | None = None,
1312
+ retry_after_seconds: int | None = None,
1313
+ ) -> str:
1314
+ """Why a settled probe has no answer, following the RESOURCE where the two disagree.
1315
+
1316
+ The resource is the authority on the state; the frame refines a ``failed`` into which kind of
1317
+ failure. A frame that says ``succeeded`` beside a resource that says ``cancelled`` is a probe
1318
+ whose session ended between the worker answering and Studio recording it, and the record wins.
1319
+ """
1320
+
1321
+ if state in {"cancelled", "expired"}:
1322
+ said = (
1323
+ "The session ended, or its worker's lease lapsed, before an answer was recorded; "
1324
+ "open a new session."
1325
+ )
1326
+ if outcome == "succeeded":
1327
+ said += (
1328
+ " The worker's own frame says it answered, but Studio recorded no result, and "
1329
+ "the record is the authority."
1330
+ )
1331
+ return said
1332
+ if display_code == _SESSION_ACQUISITION_THROTTLED:
1333
+ window = (
1334
+ f"Retry after {retry_after_seconds} seconds."
1335
+ if retry_after_seconds is not None
1336
+ else "Studio supplied no usable retry window."
1337
+ )
1338
+ return (
1339
+ f"Studio held this session's acquisition admission slot. {window} This is unrelated to "
1340
+ "model tokens, billing, authentication, or source credentials."
1341
+ )
1342
+ if display_code == "PROBE_SOURCE_ACQUISITION_TIMEOUT":
1343
+ return (
1344
+ "Studio did not obtain the source before the acquisition timeout. Retrying later may "
1345
+ "help; "
1346
+ "choosing a smaller scan will not make the remote acquisition settle."
1347
+ )
1348
+ if display_code == "SANDBOX_TIMEOUT":
1349
+ return (
1350
+ "The source arrived, but the Clean room did not finish scanning it in time; the same "
1351
+ "probe against a smaller scan may succeed."
1352
+ )
1353
+ if isinstance(outcome, str) and outcome in _OUTCOME_ADVICE:
1354
+ return f"The worker reported it {outcome}: {_OUTCOME_ADVICE[outcome]}."
1355
+ if outcome == "succeeded":
1356
+ return (
1357
+ "The worker's own frame says it answered, but Studio recorded a failure, and the "
1358
+ "record is the authority; the code above is what Studio holds."
1359
+ )
1360
+ return (
1361
+ "The code above is the worker's own; the settled line on the session's stream says "
1362
+ "whether it was refused, throttled, or timed out."
1363
+ )
1364
+
1365
+
1366
+ def _unsettled_why(progress: Any, frame_seen: bool) -> str:
1367
+ """Why the wait ended with the probe still queued or claimed, by what actually ended it."""
1368
+
1369
+ if progress.expired:
1370
+ return "The wait ran out before it settled"
1371
+ if progress.end_reason is not None or progress.status is not None:
1372
+ return "The session's run ended before it settled"
1373
+ if frame_seen:
1374
+ return "The stream said it settled, but Studio's record had not caught up when it was read"
1375
+ return "The stream stopped before it settled"
1376
+
1377
+
1378
+ def research_probe_exit_code(payload: Mapping[str, Any]) -> int:
1379
+ """``0`` only for an answered probe, or for a submission the caller chose not to wait on."""
1380
+
1381
+ if payload.get("status") == "research_probe_answered":
1382
+ return 0
1383
+ if payload.get("status") == "research_probe_submitted" and payload.get("waited") is False:
1384
+ return 0
1385
+ return 2
1386
+
1387
+
1388
+ # --------------------------------------------------------------------------------------------
1389
+ # research-status and research-close
1390
+ # --------------------------------------------------------------------------------------------
1391
+
1392
+
1393
+ def research_status(
1394
+ args: argparse.Namespace, *, client: StudioResearchClient | None = None
1395
+ ) -> dict[str, Any]:
1396
+ """``mr-data research-status``: Studio's own reading of one session, and of one probe."""
1397
+
1398
+ session_id, requested_run_id = _session_or_run(args)
1399
+ named = getattr(args, "probe", None)
1400
+ probe_id = (
1401
+ _identifier(named, what="a probe identifier", flag="--probe") if named is not None else None
1402
+ )
1403
+ selected = client or _client(args)
1404
+ record = (
1405
+ selected.get_session_by_run(requested_run_id)
1406
+ if requested_run_id is not None
1407
+ else selected.get_session(session_id)
1408
+ )
1409
+ if requested_run_id is not None:
1410
+ session_id = _session_identifier(record)
1411
+ _bind_session(record, session_id=session_id)
1412
+ run_id = _run_identifier(record)
1413
+ if requested_run_id is not None and run_id != requested_run_id:
1414
+ raise ThinLaneError("THIN_SESSION_BINDING", "Studio answered about a different session Run")
1415
+ payload: dict[str, Any] = {
1416
+ "schema_version": RESEARCH_STATUS_SCHEMA,
1417
+ "status": "research_session_reported",
1418
+ "lane": "hosted",
1419
+ "session_id": session_id,
1420
+ "run_id": run_id,
1421
+ "watch_command": f"mr-data watch {run_id} --hosted",
1422
+ **_dashboard_urls(selected.session.cloud_url, record, run_id),
1423
+ "session": session_receipt(record),
1424
+ "studio": selected.session.to_receipt(),
1425
+ }
1426
+ if probe_id is not None:
1427
+ probe = selected.get_probe(session_id, probe_id)
1428
+ if probe.get("probe_id") != probe_id:
1429
+ raise ThinLaneError("THIN_SESSION_BINDING", "Studio answered about a different probe")
1430
+ payload["probe"] = probe_receipt(probe)
1431
+ payload["result"] = probe.get("result") if probe.get("state") == "succeeded" else None
1432
+ payload["not_evidence"] = NOT_EVIDENCE
1433
+ return payload
1434
+
1435
+
1436
+ def research_close(
1437
+ args: argparse.Namespace, *, client: StudioResearchClient | None = None
1438
+ ) -> dict[str, Any]:
1439
+ """``mr-data research-close``: end the session, its queued probes, and its worker."""
1440
+
1441
+ session_id, requested_run_id = _session_or_run(args)
1442
+ # Checked once, and before the credential: the body below is composed from the checked value
1443
+ # rather than run through the check a second time.
1444
+ reason = close_reason(getattr(args, "reason", None))
1445
+ selected = client or _client(args)
1446
+ body: dict[str, Any] = {"workspace_id": str(selected.session.workspace_id)}
1447
+ if requested_run_id is None:
1448
+ body["session_id"] = session_id
1449
+ else:
1450
+ body["run_id"] = requested_run_id
1451
+ if reason is not None:
1452
+ body["reason"] = reason
1453
+ record = (
1454
+ selected.close_session_by_run(requested_run_id, body)
1455
+ if requested_run_id is not None
1456
+ else selected.close_session(session_id, body)
1457
+ )
1458
+ if requested_run_id is not None:
1459
+ session_id = _session_identifier(record)
1460
+ _bind_session(record, session_id=session_id)
1461
+ run_id = _run_identifier(record)
1462
+ if requested_run_id is not None and run_id != requested_run_id:
1463
+ raise ThinLaneError("THIN_SESSION_BINDING", "Studio answered about a different session Run")
1464
+ return {
1465
+ "schema_version": RESEARCH_CLOSE_SCHEMA,
1466
+ "status": "research_session_closed",
1467
+ "lane": "hosted",
1468
+ "session_id": session_id,
1469
+ "run_id": run_id,
1470
+ **_dashboard_urls(selected.session.cloud_url, record, run_id),
1471
+ "session": session_receipt(record),
1472
+ "studio": selected.session.to_receipt(),
1473
+ }
1474
+
1475
+
1476
+ # --------------------------------------------------------------------------------------------
1477
+ # The command line, declared once for both parsers
1478
+ # --------------------------------------------------------------------------------------------
1479
+
1480
+ #: The four names, in the order a session is used.
1481
+ COMMANDS = ("research-open", "research-probe", "research-status", "research-close")
1482
+
1483
+ #: One help line per command. Read by the thin help and by the local parser alike, so the two
1484
+ #: cannot describe one command two ways.
1485
+ COMMAND_HELP = dict(
1486
+ (
1487
+ ("research-open", "open one hosted research session and watch its worker warm up"),
1488
+ ("research-probe", "ask a research session's worker one question and print its answer"),
1489
+ ("research-status", "report Studio's own reading of one research session, or one probe"),
1490
+ ("research-close", "close one research session; its worker exits"),
1491
+ )
1492
+ )
1493
+
1494
+ _SESSION_POSITIONAL = "the research session, by the identifier research-open printed"
1495
+
1496
+
1497
+ def _session_or_run(args: argparse.Namespace) -> tuple[str, str | None]:
1498
+ """Choose one direct Studio coordinate; resolving a Run by listing sessions is forbidden."""
1499
+
1500
+ session_value = getattr(args, "session_id", None)
1501
+ run_value = getattr(args, "run_id", None)
1502
+ if (session_value is None) == (run_value is None):
1503
+ raise ThinLaneError(
1504
+ "THIN_REQUEST_INVALID", "supply exactly one of SESSION_ID or --run RUN_ID"
1505
+ )
1506
+ if run_value is not None:
1507
+ return "", _identifier(run_value, what="a research session Run identifier", flag="--run")
1508
+ return _identifier(session_value, what="a research session identifier", flag="SESSION_ID"), None
1509
+
1510
+
1511
+ def declare_arguments(name: str, parser: argparse.ArgumentParser) -> None:
1512
+ """Add one research command's arguments to ``parser``.
1513
+
1514
+ ⚠ CALLED BY BOTH PARSERS. The thin router's parser and ``cli``'s parser each add the four
1515
+ names, and each hands its subparser here, so the argument surface is declared once and the
1516
+ two cannot disagree -- the property ``note`` has by a test and these have by construction.
1517
+ """
1518
+
1519
+ if name == "research-open":
1520
+ parser.add_argument(
1521
+ "--dataset",
1522
+ help="the Dataset the session probes for; taken from --bundle when one is given",
1523
+ )
1524
+ parser.add_argument(
1525
+ "--question",
1526
+ help="the question being explored, by its identifier; taken from --bundle when given",
1527
+ )
1528
+ parser.add_argument(
1529
+ "--bundle",
1530
+ metavar="BUNDLE_DIR",
1531
+ help="an authoring bundle whose propose journal names the Dataset, question and "
1532
+ "sources to probe, so they are read rather than repeated",
1533
+ )
1534
+ parser.add_argument("--table", help="a Table already under construction, when there is one")
1535
+ parser.add_argument(
1536
+ "--source",
1537
+ action="append",
1538
+ metavar="SOURCE_ID",
1539
+ help="a source the session may probe; repeatable, at most 32; added to --bundle's",
1540
+ )
1541
+ parser.add_argument("--label", help="a display label for the session; never an identity")
1542
+ parser.add_argument(
1543
+ "--no-wait",
1544
+ dest="no_wait",
1545
+ action="store_true",
1546
+ help="return once Studio has recorded the session, without waiting for its worker",
1547
+ )
1548
+ return
1549
+ if name == "research-probe":
1550
+ parser.add_argument("session_id", help=_SESSION_POSITIONAL)
1551
+ parser.add_argument(
1552
+ "--kind",
1553
+ required=True,
1554
+ choices=PROBE_KINDS,
1555
+ help="what to ask the worker; the request body under it is checked before it is sent",
1556
+ )
1557
+ parser.add_argument(
1558
+ "--request", help="the probe request body, as one JSON object on the command line"
1559
+ )
1560
+ parser.add_argument(
1561
+ "--request-file",
1562
+ dest="request_file",
1563
+ metavar="FILE",
1564
+ help="the probe request body, read from a JSON file instead of the command line",
1565
+ )
1566
+ parser.add_argument(
1567
+ "--no-wait",
1568
+ dest="no_wait",
1569
+ action="store_true",
1570
+ help="return once the probe is queued, without waiting for its answer",
1571
+ )
1572
+ return
1573
+ if name == "research-status":
1574
+ parser.add_argument("session_id", nargs="?", metavar="SESSION_ID", help=_SESSION_POSITIONAL)
1575
+ parser.add_argument(
1576
+ "--run",
1577
+ dest="run_id",
1578
+ metavar="RUN_ID",
1579
+ help="the displayed research session Run; Studio resolves its session directly",
1580
+ )
1581
+ parser.add_argument(
1582
+ "--probe",
1583
+ metavar="PROBE_ID",
1584
+ help="also report this probe, with its answer if it has one",
1585
+ )
1586
+ return
1587
+ if name == "research-close":
1588
+ parser.add_argument("session_id", nargs="?", metavar="SESSION_ID", help=_SESSION_POSITIONAL)
1589
+ parser.add_argument(
1590
+ "--run",
1591
+ dest="run_id",
1592
+ metavar="RUN_ID",
1593
+ help="the displayed research session Run; Studio resolves and closes its session",
1594
+ )
1595
+ parser.add_argument("--reason", help="why the session is being closed; recorded by Studio")
1596
+ return
1597
+ raise ValueError(f"{name!r} is not a research command") # pragma: no cover - closed tuple
1598
+
1599
+
1600
+ #: The handler and the exit-code rule behind each name, for both front doors.
1601
+ HANDLERS: dict[str, Callable[[argparse.Namespace], dict[str, Any]]] = {
1602
+ "research-open": research_open,
1603
+ "research-probe": research_probe,
1604
+ "research-status": research_status,
1605
+ "research-close": research_close,
1606
+ }
1607
+
1608
+ EXIT_CODES: dict[str, Callable[[Mapping[str, Any]], int]] = {
1609
+ "research-open": research_open_exit_code,
1610
+ "research-probe": research_probe_exit_code,
1611
+ }
1612
+
1613
+
1614
+ __all__ = [
1615
+ "BUDGET_EXCEEDED_CODE",
1616
+ "CLOSE_SESSION_BY_RUN_PATH",
1617
+ "CLOSE_SESSION_PATH",
1618
+ "COMMANDS",
1619
+ "COMMAND_HELP",
1620
+ "DEFAULT_PROBE_WAIT_SECONDS",
1621
+ "EXIT_CODES",
1622
+ "GET_PROBE_PATH",
1623
+ "GET_SESSION_BY_RUN_PATH",
1624
+ "GET_SESSION_PATH",
1625
+ "HANDLERS",
1626
+ "LIVE_SESSION_STATES",
1627
+ "MAX_CLOSE_REASON_CHARS",
1628
+ "MAX_LABEL_CHARS",
1629
+ "MAX_OPEN_WAIT_SECONDS",
1630
+ "MAX_PROBE_WAIT_SECONDS",
1631
+ "MAX_SOURCE_IDS",
1632
+ "NOT_EVIDENCE",
1633
+ "OPEN_POLL_SECONDS",
1634
+ "OPEN_SESSION_PATH",
1635
+ "PROBE_KINDS",
1636
+ "PROBE_STATES",
1637
+ "RESEARCH_CLOSE_SCHEMA",
1638
+ "RESEARCH_OPEN_SCHEMA",
1639
+ "RESEARCH_PROBE_SCHEMA",
1640
+ "RESEARCH_STATUS_SCHEMA",
1641
+ "SESSION_STATES",
1642
+ "SETTLED_PROBE_STATES",
1643
+ "SUBMIT_PROBE_PATH",
1644
+ "TERMINAL_SESSION_STATES",
1645
+ "RetryAfterRefusal",
1646
+ "StudioResearchClient",
1647
+ "bundle_session_context",
1648
+ "close_session_body",
1649
+ "declare_arguments",
1650
+ "open_session_body",
1651
+ "probe_receipt",
1652
+ "probe_request",
1653
+ "research_close",
1654
+ "research_open",
1655
+ "research_open_exit_code",
1656
+ "research_probe",
1657
+ "research_probe_exit_code",
1658
+ "research_status",
1659
+ "session_context",
1660
+ "session_label",
1661
+ "session_receipt",
1662
+ "submit_probe_body",
1663
+ ]