mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,239 @@
1
+ """One certified fast row splitter for delimited text, and the closed subset it is certified over.
2
+
3
+ Two places in the harness turn delimited bytes into rows: the Reader family's
4
+ ``decode_delimited_stream``, which normalizes a publisher's file into the canonical sealed
5
+ artifact, and the build kernel's ``_stream_csv_rows``, which reads that artifact back. Both
6
+ reach the stdlib ``csv`` module, whose tokenizer is already C. What costs the time at
7
+ year scale is not the tokenizing but the per-cell Python objects on either side of it, and
8
+ what costs the memory is decoding a whole year-scale file into one Python ``str`` before the
9
+ first row is looked at.
10
+
11
+ This module addresses both without moving a single sealed byte, by doing less rather than by
12
+ doing something different.
13
+
14
+ The certified subset
15
+ --------------------
16
+ ``certifies`` answers one question about exact bytes: is this input inside a subset where
17
+ splitting on line terminators and then on the delimiter is *provably* the same tokenizing the
18
+ stdlib parser performs? The subset is closed and each member of it is a refusal the stdlib
19
+ parser would otherwise have had to reason about:
20
+
21
+ * **No quote character anywhere.** A quote is the only byte that makes the stdlib tokenizer's
22
+ state machine leave its plain-field state -- it opens a quoted field, doubles into a literal
23
+ quote, or ends one. With no quote in the input, the machine never leaves ``IN_FIELD``, so a
24
+ field ends at the delimiter or at the line terminator and nowhere else.
25
+ * **No carriage return anywhere.** A lone carriage return terminates a record for the stdlib
26
+ reader, and a carriage return inside a quoted value survives into the value. Excluding the
27
+ byte removes both cases, leaving the newline as the one record terminator.
28
+ * **No NUL byte anywhere.** Both call sites refuse a NUL; excluding it here means the
29
+ per-cell scan each of them performs is provably answering "no", so it does not have to run.
30
+ * **No empty record.** The stdlib reader yields an empty row for a blank line, which is not
31
+ what splitting produces, and which the callers refuse on width. An input with a blank line
32
+ is left to the stdlib reader so that it refuses it in its own words, at its own row.
33
+ * **Strict UTF-8 over the whole input, and a non-empty input.** Validity is settled here,
34
+ over every byte, before a single row is yielded -- so an input whose last byte is invalid is
35
+ never certified, and is refused by the existing path exactly as early as it always was.
36
+
37
+ Anything outside the subset is not approximated and not repaired: ``certifies`` returns
38
+ ``False`` and the caller runs its existing stdlib path, untouched, from the beginning. There
39
+ is no partial fast lane and no mid-input hand-over, so the only inputs this module has an
40
+ opinion about are inputs whose tokenizing has one possible answer.
41
+
42
+ Why this is byte-equivalence rather than a close approximation
43
+ --------------------------------------------------------------
44
+ For an input in the subset the claim is small enough to check by reading it. The stdlib
45
+ tokenizer starts each record in ``START_FIELD``; with no quote and no escape character
46
+ configured, the only transitions available are "delimiter ends this field", "line terminator
47
+ ends this field and this record", and "any other character joins this field". That is the
48
+ definition of splitting the record on the delimiter. The records themselves are the input
49
+ split on the newline, with the terminator of the final record dropped when the input ends
50
+ with one. The differential tests compare the two implementations over random inputs drawn
51
+ from the subset and its boundary rather than trusting this paragraph.
52
+
53
+ The one refusal the splitter has to make for itself
54
+ ---------------------------------------------------
55
+ Splitting removes every refusal the state machine could raise except one: the stdlib parser
56
+ refuses a field longer than ``csv.field_size_limit()`` characters, and splitting has no such
57
+ bound. That refusal is *reproduced* rather than dropped or approximated. ``certified_rows``
58
+ raises the same ``csv.Error`` the stdlib parser raises, in the same place in the iteration --
59
+ before the row carrying the oversized field is yielded -- so the callers' existing
60
+ ``except csv.Error`` handlers turn it into the refusal they always turned it into.
61
+
62
+ A block shorter than the limit cannot contain a field longer than it, so in the ordinary case
63
+ the per-cell comparison is skipped for the whole block rather than run and passed.
64
+
65
+ The limit is read at parse time rather than at import, because it is process-global and
66
+ mutable, and once per block rather than once per row. That granularity is stated rather than
67
+ hidden: a caller that changes the limit *while rows are being pulled* has it take effect at
68
+ the next block instead of at the next row, where the stdlib parser would apply it at the next
69
+ character. Reading it per row instead costs about a sixth of the splitting at year scale, to
70
+ buy exactness in a case that is already racy -- the limit is one process-global integer, and
71
+ nothing in this repository moves it except a test that restores it afterwards.
72
+
73
+ Bounded memory
74
+ --------------
75
+ Rows are produced from blocks cut at record terminators, so the decoded text alive at any
76
+ moment is one block rather than the whole input. A block always ends at a newline (or at the
77
+ end of the input), which is never inside a multi-byte UTF-8 sequence, so decoding block by
78
+ block admits exactly what decoding the whole input admits.
79
+
80
+ The bound is the block, not one row: a block is split into its lines at once, so the live
81
+ cost is one ``str`` per line of one block. For short records that per-line overhead is
82
+ several times the block's own bytes, which is why the block is sized in tens of kilobytes
83
+ rather than megabytes. Measured at the worst record shape this can be given -- two-byte
84
+ records, which is one ``str`` object per two bytes of input -- the transient is about 1.2 MB
85
+ at 64 KiB blocks, against 76 MB at 4 MiB blocks. It does not grow with the input.
86
+ """
87
+
88
+ from __future__ import annotations
89
+
90
+ import codecs
91
+ import csv
92
+ from collections.abc import Iterator
93
+
94
+ __all__ = [
95
+ "CERTIFIED_BLOCK_BYTES",
96
+ "certified_rows",
97
+ "certifies",
98
+ ]
99
+
100
+ # The decoded text alive while rows are pulled. Two things pin it rather than throughput: the
101
+ # per-line ``str`` overhead of splitting a block is a multiple of the block's own size, and a
102
+ # block below ``csv.field_size_limit()``'s default cannot carry an over-limit field, so the
103
+ # ordinary input skips the per-cell comparison entirely.
104
+ CERTIFIED_BLOCK_BYTES = 64 * 1024
105
+
106
+ # The window the up-front validity pass feeds the incremental decoder. Separate from the row
107
+ # block size because this pass keeps nothing it decodes; it only has to answer whether every
108
+ # byte of the input is strict UTF-8.
109
+ _VALIDATION_WINDOW_BYTES = 1 * 1024 * 1024
110
+
111
+ # The bytes whose presence takes an input out of the subset, each with the state-machine
112
+ # transition it is here to rule out.
113
+ _QUOTE_BYTE = b'"'
114
+ _CARRIAGE_RETURN_BYTE = b"\r"
115
+ _NUL_BYTE = b"\x00"
116
+ _EMPTY_RECORD = b"\n\n"
117
+ _RECORD_TERMINATOR = b"\n"
118
+
119
+
120
+ def certifies(content: bytes, *, delimiter: str) -> bool:
121
+ """Answer whether these exact bytes lie inside the certified subset.
122
+
123
+ The scans are the whole of the decision and each one is a single pass the interpreter
124
+ performs in C. Validity of the text is settled last because it is the only pass whose
125
+ cost is proportional to the work rather than to the search, and because an input that
126
+ fails a cheaper test never reaches it.
127
+
128
+ A delimiter this module cannot reason about -- one that is not a single character, or one
129
+ that is itself a byte the subset excludes -- is not certified rather than special-cased.
130
+ The caller's own delimiter contract is stricter, so this is a second lock on the same
131
+ door rather than the only one.
132
+ """
133
+
134
+ if not isinstance(content, (bytes, bytearray)):
135
+ return False
136
+ if not isinstance(delimiter, str) or len(delimiter) != 1 or delimiter in '"\r\n\x00':
137
+ return False
138
+ if not content:
139
+ return False
140
+ if content.startswith(_RECORD_TERMINATOR):
141
+ return False
142
+ if _QUOTE_BYTE in content:
143
+ return False
144
+ if _CARRIAGE_RETURN_BYTE in content:
145
+ return False
146
+ if _NUL_BYTE in content:
147
+ return False
148
+ if _EMPTY_RECORD in content:
149
+ return False
150
+ return _is_strict_utf8(content)
151
+
152
+
153
+ def _is_strict_utf8(content: bytes) -> bool:
154
+ """Decide strict UTF-8 validity over every byte without holding the decoded text.
155
+
156
+ The incremental decoder carries the partial sequence across window boundaries, so a
157
+ multi-byte character split by a window is decoded rather than refused, and the final call
158
+ refuses an input that ends mid-sequence. Nothing it returns is kept: the question is
159
+ whether the whole input decodes, and the answer is a boolean.
160
+ """
161
+
162
+ decoder = codecs.getincrementaldecoder("utf-8")("strict")
163
+ view = memoryview(content)
164
+ try:
165
+ for start in range(0, len(content), _VALIDATION_WINDOW_BYTES):
166
+ decoder.decode(view[start : start + _VALIDATION_WINDOW_BYTES])
167
+ decoder.decode(b"", final=True)
168
+ except UnicodeDecodeError:
169
+ return False
170
+ return True
171
+
172
+
173
+ def certified_rows(
174
+ content: bytes,
175
+ *,
176
+ delimiter: str,
177
+ block_bytes: int = CERTIFIED_BLOCK_BYTES,
178
+ ) -> Iterator[list[str]]:
179
+ """Yield each record of a certified input as its list of cells.
180
+
181
+ The caller has already had ``certifies`` answer ``True`` for these exact bytes; this
182
+ function does not re-decide it, because a second, differently worded copy of the subset
183
+ is exactly how two statements of one rule drift apart. It performs none of the caller's
184
+ admission -- no width, no budget, no NUL -- so that every refusal an input can draw is
185
+ raised by the caller that always raised it, in the caller's own vocabulary, at the
186
+ caller's own row.
187
+
188
+ The one exception is the stdlib parser's own field-size refusal, which belongs to the
189
+ tokenizing rather than to any caller, and is therefore reproduced here as the same
190
+ ``csv.Error``, before the row carrying the oversized field is yielded.
191
+ """
192
+
193
+ if block_bytes < 1:
194
+ # A block that cannot hold a byte cannot advance the cursor, and the loop below would
195
+ # never terminate. Refused rather than clamped, because a caller asking for it has a
196
+ # bug rather than a preference.
197
+ raise ValueError("block_bytes must be at least one byte")
198
+ view = memoryview(content)
199
+ total = len(content)
200
+ position = 0
201
+ while position < total:
202
+ window = min(position + block_bytes, total)
203
+ cut = content.rfind(_RECORD_TERMINATOR, position, window)
204
+ if cut < 0:
205
+ # One record is longer than a block. Take it whole rather than splitting a record
206
+ # across two blocks: a block that ends anywhere but a record terminator would need
207
+ # state carried between blocks, which is the complication this module exists
208
+ # without. The caller's field and row budgets are what bound such a record.
209
+ cut = content.find(_RECORD_TERMINATOR, window)
210
+ stop = total if cut < 0 else cut + 1
211
+ block = str(view[position:stop], "utf-8")
212
+ position = stop
213
+ lines = block.split("\n")
214
+ if lines and lines[-1] == "":
215
+ # The terminator of the block's final record, which is not itself a record. Only
216
+ # the last block of an input that does not end with a terminator keeps its tail.
217
+ lines.pop()
218
+ # Read per block rather than once: the limit is process-global and mutable, and the
219
+ # stdlib parser consults it as it goes rather than remembering what it was.
220
+ limit = csv.field_size_limit()
221
+ # A block no longer than the limit cannot contain a field longer than it, so the
222
+ # comparison below is already answered for every cell in it.
223
+ checked = len(block) > limit
224
+ # The block's own text is redundant once it has been split into separately allocated
225
+ # lines, and the generator's locals outlive each yield -- so it is released here rather
226
+ # than being held alive while the next block is cut.
227
+ block = ""
228
+ if checked:
229
+ for line in lines:
230
+ cells = line.split(delimiter)
231
+ for cell in cells:
232
+ if len(cell) > limit:
233
+ raise csv.Error(f"field larger than field limit ({limit})")
234
+ yield cells
235
+ else:
236
+ for line in lines:
237
+ yield line.split(delimiter)
238
+ # Likewise the lines, which are all spent by now.
239
+ lines = []
@@ -0,0 +1,237 @@
1
+ """Scan a directory of runs once and report which runs need operator action.
2
+
3
+ The command binds no port. ``--html`` writes a self-contained document to stdout. State publication
4
+ uses ``os.replace``, so the scan reads without taking the coordinator lock.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import os
11
+ import stat
12
+ from collections.abc import Mapping, Sequence
13
+ from pathlib import Path
14
+ from typing import Any
15
+
16
+ from mostlyright.data_harness import events
17
+ from mostlyright.data_harness.nbrender.parse import esc
18
+ from mostlyright.data_harness.watch import stylesheet
19
+
20
+ FLEET_SCHEMA_VERSION = "mr-fleet.v1"
21
+
22
+ # A root is walked one level deep and no further, and no more than this many children are read. A
23
+ # directory of runs is a working directory, not a filesystem to crawl (T-30-35, T-30-36).
24
+ MAX_FLEET_RUNS = 500
25
+
26
+ _STATE_PATH = (".mr-data", "state.json")
27
+ _MAX_STATE_BYTES = 4 * 1024 * 1024
28
+
29
+ # Display labels for persisted run phases.
30
+ _PLAIN_STATE: Mapping[str, str] = {
31
+ "initialized": "not started",
32
+ "planned": "ready to build",
33
+ "building": "building",
34
+ "candidate_built": "built",
35
+ "review_recorded": "reviewed",
36
+ "failed": "stopped",
37
+ }
38
+
39
+ _WAITING = "waiting for a person to look at it"
40
+ _RIGHTS_REASON = "stopped: the rights position needs a decision"
41
+ # Keep completed reviews that require fixes distinct from reviews that have not started.
42
+ _FIXES_REASON = "reviewed: the review asked for fixes"
43
+
44
+
45
+ def _read_json(name: str | os.PathLike[str], *, dir_fd: int | None = None) -> Any:
46
+ """Read bounded JSON through one nonblocking, no-follow descriptor.
47
+
48
+ ``dir_fd`` anchors every path component beneath the validated child directory. Disk and decode
49
+ failures return ``None``.
50
+ """
51
+
52
+ try:
53
+ return json.loads(
54
+ events.read_regular_bytes(name, dir_fd=dir_fd, max_bytes=_MAX_STATE_BYTES).decode(
55
+ "utf-8"
56
+ )
57
+ )
58
+ except (OSError, ValueError, RecursionError):
59
+ return None
60
+
61
+
62
+ def _feed_dirs(child: Path) -> list[Path]:
63
+ """The feed directories that could belong to this child, newest-run-shape first.
64
+
65
+ ``events.run_feed_dir`` scopes a feed by the run's own name, so a workspace's feed is at
66
+ ``<child>/.mr-events/result/`` and a bare run directory's is at
67
+ ``<root>/.mr-events/<child name>/``. A child can only be one of the two, and a directory that
68
+ does not exist simply yields no feed.
69
+ """
70
+
71
+ found: list[Path] = []
72
+ for run_dir in (child / "result", child):
73
+ try:
74
+ found.append(events.run_feed_dir(run_dir))
75
+ except ValueError:
76
+ continue
77
+ return found
78
+
79
+
80
+ def _terminal_record(child: Path) -> Mapping[str, Any] | None:
81
+ """Return the terminal record selected by the shared attempt selector, if present."""
82
+
83
+ for feed_dir in _feed_dirs(child):
84
+ _path, _size, records = events.select_narrated_feed(feed_dir)
85
+ if not records:
86
+ continue
87
+ last = records[-1]
88
+ if last.get("event") in events.TERMINAL_EVENT_NAMES:
89
+ return last
90
+ return None
91
+ return None
92
+
93
+
94
+ def _run_directory(child: Path) -> Path | None:
95
+ """Return the directory holding ``candidate/`` for this child, or ``None``.
96
+
97
+ ``lstat``, for the reason ``scan_fleet`` already gives about a symlinked child: a sealed
98
+ candidate directory is one ``build_candidate`` renamed into place, never a link, and a link at
99
+ that name would let anything on this disk answer "this run built" for a run that did not.
100
+ """
101
+
102
+ for candidate in (child / "result", child):
103
+ try:
104
+ if stat.S_ISDIR(os.lstat(candidate / "candidate").st_mode):
105
+ return candidate
106
+ except OSError:
107
+ continue
108
+ return None
109
+
110
+
111
+ def _failure_reason(record: Mapping[str, Any]) -> str:
112
+ facts = record.get("facts")
113
+ facts = facts if isinstance(facts, Mapping) else {}
114
+ if record.get("event") == "run_interrupted":
115
+ return "stopped: the build was interrupted before it finished"
116
+ finding = facts.get("finding_id")
117
+ if isinstance(finding, str) and finding.startswith("RIGHTS_"):
118
+ # Kept distinct on purpose: a rights escalation is a decision a person owes the run, and
119
+ # burying it in a generic failure line is how it gets missed.
120
+ return _RIGHTS_REASON
121
+ message = facts.get("message")
122
+ if isinstance(message, str) and message:
123
+ return f"stopped: {message}"
124
+ return "stopped: the build did not finish"
125
+
126
+
127
+ def scan_fleet(root: str | os.PathLike[str]) -> list[dict[str, Any]]:
128
+ """Scan ``root``'s IMMEDIATE children and return one row per run, in sorted name order.
129
+
130
+ A child that is neither a workspace nor a run directory is skipped silently -- a directory of
131
+ runs routinely holds notes, archives and half-finished experiments, and a scan that complained
132
+ about each one would be a scan nobody runs twice. A symlinked child is skipped without being
133
+ followed.
134
+ """
135
+
136
+ base = Path(root)
137
+ rows: list[dict[str, Any]] = []
138
+ try:
139
+ names = sorted(entry.name for entry in os.scandir(base))
140
+ except OSError:
141
+ return rows
142
+
143
+ for name in names[:MAX_FLEET_RUNS]:
144
+ child = base / name
145
+ try:
146
+ info = os.lstat(child)
147
+ except OSError:
148
+ continue
149
+ if not stat.S_ISDIR(info.st_mode):
150
+ continue # never follow a symlinked child, never read a file as a run
151
+
152
+ # Anchor the state read to the validated child descriptor so no component follows a link.
153
+ child_fd = None
154
+ try:
155
+ child_fd = events.open_directory(child)
156
+ state = _read_json(os.path.join(*_STATE_PATH), dir_fd=child_fd)
157
+ except OSError:
158
+ state = None
159
+ finally:
160
+ if child_fd is not None:
161
+ os.close(child_fd)
162
+ run_dir = _run_directory(child)
163
+ terminal = _terminal_record(child)
164
+ if state is None and run_dir is None:
165
+ continue
166
+
167
+ phase = state.get("phase") if isinstance(state, Mapping) else None
168
+ if isinstance(phase, str):
169
+ plain = _PLAIN_STATE.get(phase, "state not recorded")
170
+ else:
171
+ plain = "built" if run_dir is not None else "state not recorded"
172
+
173
+ # A sealed candidate outranks a terminal feed record. Persisted ``failed`` state remains
174
+ # authoritative when no sealed candidate exists.
175
+ if run_dir is not None:
176
+ terminal = None
177
+
178
+ # `phase == "review_recorded"` covers both review verdicts. Read `verified_status` from the
179
+ # review record.
180
+ review = state.get("review") if isinstance(state, Mapping) else None
181
+ verdict = review.get("verified_status") if isinstance(review, Mapping) else None
182
+
183
+ reasons: list[str] = []
184
+ if phase == "failed" or terminal is not None:
185
+ plain = "stopped"
186
+ reasons.append(
187
+ _failure_reason(terminal) if terminal is not None else "stopped: the build failed"
188
+ )
189
+ elif verdict == "fixes_required":
190
+ # Ahead of the waiting check on purpose: a recorded verdict is durable state, and it
191
+ # outranks "is there a review directory" whether or not one was ever written.
192
+ reasons.append(_FIXES_REASON)
193
+ elif run_dir is not None and not (run_dir / "review").is_dir():
194
+ reasons.append(_WAITING)
195
+
196
+ rows.append(
197
+ {
198
+ # Sanitize only the reported name. Filesystem calls above use the original bytes.
199
+ "name": events.writable_text(name),
200
+ "state": plain,
201
+ "needs_person": bool(reasons),
202
+ "reasons": reasons,
203
+ }
204
+ )
205
+ return rows
206
+
207
+
208
+ def render_fleet_html(rows: Sequence[Mapping[str, Any]]) -> str:
209
+ """Render the scan as one self-contained page. Printed to stdout, never served."""
210
+
211
+ waiting = sum(1 for row in rows if row.get("needs_person"))
212
+ body = "".join(
213
+ "<tr>"
214
+ f"<td>{esc(row.get('name', ''))}</td>"
215
+ f"<td>{esc(row.get('state', ''))}</td>"
216
+ f"<td>{esc('yes' if row.get('needs_person') else 'no')}</td>"
217
+ f"<td>{esc('; '.join(str(reason) for reason in row.get('reasons') or []))}</td>"
218
+ "</tr>"
219
+ for row in rows
220
+ )
221
+ return (
222
+ "<!doctype html>\n"
223
+ '<html lang="en"><head><meta charset="utf-8">'
224
+ '<meta name="viewport" content="width=device-width, initial-scale=1">'
225
+ "<title>Runs</title>"
226
+ f"{stylesheet()}"
227
+ "</head><body>\n"
228
+ '<div class="mr-watch-page">'
229
+ '<h1 class="mr-watch-title">Runs</h1>'
230
+ f'<p class="mr-watch-run">{esc(len(rows))} runs. '
231
+ f"{esc(waiting)} need a person.</p>"
232
+ '<table class="mr-watch-table"><thead><tr>'
233
+ "<th>Run</th><th>State</th><th>Needs a person</th><th>Why</th>"
234
+ f"</tr></thead><tbody>{body}</tbody></table>"
235
+ "</div>\n"
236
+ "</body></html>\n"
237
+ )