mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,394 @@
1
+ """nbformat 4.x JSON to a typed, escape-ready model, plus the shared render helpers.
2
+
3
+ Notebook JSON is attacker-shaped: a cell may be any type, ``source`` may be a string or a list,
4
+ ``outputs`` may be missing, a mimetype value may be a dict or a list of lines. Nothing here
5
+ raises on a malformed shape — an unrecognized cell or output degrades to an ``unknown`` kind so a
6
+ single bad record never blanks a render. Every value that reaches HTML must pass through ``esc``;
7
+ that discipline is enforced downstream, but the model keeps raw strings so callers can escape at
8
+ the point of interpolation rather than trusting pre-escaped data.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import html
14
+ import re
15
+ from dataclasses import dataclass, field
16
+ from typing import Any, NamedTuple
17
+
18
+ # Mimetype precedence for a multi-format output bundle. Highest priority first.
19
+ MIMETYPE_PRECEDENCE: tuple[str, ...] = (
20
+ "application/vnd.jupyter.widget-view+json",
21
+ "application/vnd.mostlyright.column-profile.v1+json",
22
+ "application/vnd.mostlyright.source-observation+json",
23
+ "application/vnd.mostlyright.stage-observation+json",
24
+ "application/vnd.mostlyright.stage-observation.v1+json",
25
+ "text/html",
26
+ "image/svg+xml",
27
+ "image/png",
28
+ "application/json",
29
+ "text/plain",
30
+ )
31
+
32
+ # Cell tags the viewer renders (B10); all others are ignored.
33
+ KNOWN_TAGS: tuple[str, ...] = ("cached", "slow", "skip", "hide-input")
34
+
35
+ _ANSI_RE = re.compile(r"\x1b\[(?P<params>[0-9;]*)m")
36
+ _ENTITY_RE = re.compile(r"\[\[(?P<type>[a-z][a-z0-9_-]*):(?P<id>[^\]]+)\]\]")
37
+ # tqdm renders "<pct>%|bar| n/total [elapsed<remaining, rate]". Match the leading percent and the
38
+ # n/total fraction; either half may be absent on the first frame.
39
+ _TQDM_RE = re.compile(
40
+ r"(?P<pct>\d{1,3})%\s*\|.*?\|\s*(?P<n>\d+)\s*/\s*(?P<total>\d+)",
41
+ re.DOTALL,
42
+ )
43
+
44
+
45
+ def esc(value: Any) -> str:
46
+ """HTML-escape any notebook-derived value, quotes included, for attribute-safe output.
47
+
48
+ Coerces non-strings so a malformed numeric or ``None`` field cannot inject unescaped text.
49
+ """
50
+ if value is None:
51
+ return ""
52
+ return html.escape(str(value), quote=True)
53
+
54
+
55
+ class StrippedAnsi(NamedTuple):
56
+ """Result of ``strip_ansi``: the plain text and whether any SGR-bold code appeared."""
57
+
58
+ text: str
59
+ bold: bool
60
+
61
+
62
+ def strip_ansi(value: str) -> StrippedAnsi:
63
+ """Remove ANSI SGR sequences, keeping only whether bold (SGR 1) was ever set.
64
+
65
+ Color codes are discarded per D17/D23 — the viewer re-tokenizes itself. Bold survives as a
66
+ single flag the caller maps to weight 600; per-span bold is out of scope for the shared helper.
67
+ """
68
+ saw_bold = False
69
+ for match in _ANSI_RE.finditer(value):
70
+ params = match.group("params")
71
+ codes = params.split(";") if params else ["0"]
72
+ if "1" in codes:
73
+ saw_bold = True
74
+ return StrippedAnsi(_ANSI_RE.sub("", value), saw_bold)
75
+
76
+
77
+ class TqdmProgress(NamedTuple):
78
+ """A parsed tqdm frame: ``fraction`` in [0, 1], the label, and the raw n/total stat."""
79
+
80
+ fraction: float
81
+ label: str
82
+ stat: str
83
+
84
+
85
+ def detect_tqdm(text: str) -> TqdmProgress | None:
86
+ """Return a parsed progress frame when ``text`` matches the tqdm pattern, else ``None``.
87
+
88
+ Used to route a text/plain or stdout output to D24 instead of a plain stream well. A carriage
89
+ return typically overwrites frames; the last frame in the string wins.
90
+ """
91
+ last = text.rsplit("\r", 1)[-1].strip()
92
+ match = _TQDM_RE.search(last)
93
+ if match is None:
94
+ return None
95
+ n = int(match.group("n"))
96
+ total = int(match.group("total"))
97
+ fraction = (n / total) if total else 0.0
98
+ fraction = max(0.0, min(1.0, fraction))
99
+ label = last[: match.start()].strip(": ")
100
+ return TqdmProgress(fraction=fraction, label=label, stat=f"{n}/{total}")
101
+
102
+
103
+ class EntityRef(NamedTuple):
104
+ """A ``[[type:id]]`` reference from markdown source (F29)."""
105
+
106
+ type: str
107
+ id: str
108
+
109
+
110
+ def split_entity_refs(source: str) -> list[str | EntityRef]:
111
+ """Split markdown source into plain runs and ``[[type:id]]`` entity refs, in order.
112
+
113
+ Plain text arrives as ``str`` segments; refs as ``EntityRef``. The markdown engine renders
114
+ the plain runs normally and hands each ref to the F29 chip renderer.
115
+ """
116
+ segments: list[str | EntityRef] = []
117
+ cursor = 0
118
+ for match in _ENTITY_RE.finditer(source):
119
+ if match.start() > cursor:
120
+ segments.append(source[cursor : match.start()])
121
+ segments.append(EntityRef(match.group("type"), match.group("id").strip()))
122
+ cursor = match.end()
123
+ if cursor < len(source):
124
+ segments.append(source[cursor:])
125
+ return segments
126
+
127
+
128
+ # Heading anchors (A04 outline links resolve to these). ``heading_text`` reduces a raw ATX
129
+ # heading to plain link text and ``slugify`` turns that into a deterministic ``#fragment``; both
130
+ # the outline rail (chrome.py) and the rendered markdown headings (markdown_body.py) run the SAME
131
+ # two helpers so an outline anchor always resolves to a real ``id`` on the heading it names.
132
+ _HEADING_IMG_RE = re.compile(r"!\[([^\]]*)\]\([^)]*\)")
133
+ _HEADING_LINK_RE = re.compile(r"\[([^\]]*)\]\([^)]*\)")
134
+
135
+
136
+ def heading_text(raw: str) -> str:
137
+ """Reduce a markdown heading to plain link text: unwrap entity refs, links, images, code."""
138
+ parts = split_entity_refs(raw)
139
+ text = "".join(p if isinstance(p, str) else p.id for p in parts)
140
+ text = _HEADING_IMG_RE.sub(r"\1", text)
141
+ text = _HEADING_LINK_RE.sub(r"\1", text)
142
+ text = text.replace("`", "")
143
+ return text.strip()
144
+
145
+
146
+ def slugify(text: str) -> str:
147
+ """Deterministic anchor fragment from heading text (lowercased, non-alnum collapsed)."""
148
+ s = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
149
+ return s or "section"
150
+
151
+
152
+ def _join_source(value: Any) -> str:
153
+ """nbformat stores multiline text as a list of lines or a single string; normalize to str."""
154
+ if isinstance(value, list):
155
+ return "".join(str(part) for part in value)
156
+ if value is None:
157
+ return ""
158
+ return str(value)
159
+
160
+
161
+ def _normalize_bundle(data: Any) -> dict[str, Any]:
162
+ """Normalize a mimetype bundle: join text-line lists; keep JSON objects and image blobs."""
163
+ if not isinstance(data, dict):
164
+ return {}
165
+ out: dict[str, Any] = {}
166
+ for mimetype, value in data.items():
167
+ if not isinstance(mimetype, str):
168
+ continue
169
+ if isinstance(value, list) and all(isinstance(part, str) for part in value):
170
+ out[mimetype] = "".join(value)
171
+ else:
172
+ out[mimetype] = value
173
+ return out
174
+
175
+
176
+ @dataclass
177
+ class Output:
178
+ """One entry from a code cell's ``outputs`` list, normalized across output types.
179
+
180
+ ``kind`` collapses the nbformat ``output_type`` and, for streams, the channel name. Malformed
181
+ records land as ``kind == 'unknown'`` with whatever fields survived.
182
+ """
183
+
184
+ kind: str # stream-stdout | stream-stderr | execute_result | display_data | error | unknown
185
+ text: str = "" # stream text or joined text/plain, raw (unescaped, ANSI intact)
186
+ data: dict[str, Any] = field(default_factory=dict) # normalized mimetype bundle
187
+ metadata: dict[str, Any] = field(default_factory=dict)
188
+ execution_count: int | None = None
189
+ ename: str | None = None
190
+ evalue: str | None = None
191
+ traceback: list[str] = field(default_factory=list)
192
+
193
+ def primary_mimetype(self) -> str | None:
194
+ """Highest-precedence mimetype present in the bundle, or the first unknown key, or None."""
195
+ if not self.data:
196
+ return None
197
+ for mimetype in MIMETYPE_PRECEDENCE:
198
+ if mimetype in self.data:
199
+ return mimetype
200
+ # An unrecognized mimetype still resolves deterministically so section-4 routing can send
201
+ # it to the labeled unsupported well rather than dropping it.
202
+ return sorted(self.data)[0]
203
+
204
+ def text_of(self, mimetype: str) -> str:
205
+ """Bundle value for ``mimetype`` as a string (already line-joined at parse time)."""
206
+ value = self.data.get(mimetype, "")
207
+ return value if isinstance(value, str) else _join_source(value)
208
+
209
+
210
+ @dataclass
211
+ class Cell:
212
+ """A single notebook cell. ``cell_type`` is one of markdown/code/raw/unknown."""
213
+
214
+ cell_type: str
215
+ cell_id: str = ""
216
+ source: str = ""
217
+ metadata: dict[str, Any] = field(default_factory=dict)
218
+ execution_count: int | None = None
219
+ outputs: list[Output] = field(default_factory=list)
220
+
221
+ @property
222
+ def tags(self) -> list[str]:
223
+ """Known B10 tags on this cell, in ``KNOWN_TAGS`` order; unknown tags are dropped."""
224
+ raw = self.metadata.get("tags")
225
+ present = set(raw) if isinstance(raw, list) else set()
226
+ return [tag for tag in KNOWN_TAGS if tag in present]
227
+
228
+ @property
229
+ def research_activity(self) -> str | None:
230
+ """Return an exact persisted research activity state, never an inferred one."""
231
+
232
+ mostlyright = self.metadata.get("mostlyright")
233
+ activity = mostlyright.get("activity") if isinstance(mostlyright, dict) else None
234
+ if (
235
+ not isinstance(activity, dict)
236
+ or set(activity) != {"schema_version", "state"}
237
+ or activity.get("schema_version") != "mostlyright.research-cell-activity.v1"
238
+ or activity.get("state") not in {"writing", "executing"}
239
+ ):
240
+ return None
241
+ state = str(activity["state"])
242
+ if state == "executing" and self.cell_type != "code":
243
+ return None
244
+ if self.cell_type == "code":
245
+ if self.execution_count is not None:
246
+ return None
247
+ if state == "writing" and self.outputs:
248
+ return None
249
+ elif self.cell_type != "markdown" or state != "writing":
250
+ return None
251
+ return state
252
+
253
+
254
+ @dataclass
255
+ class Notebook:
256
+ """A parsed notebook: format version, notebook-level metadata, and cells."""
257
+
258
+ nbformat: int = 4
259
+ nbformat_minor: int = 0
260
+ metadata: dict[str, Any] = field(default_factory=dict)
261
+ cells: list[Cell] = field(default_factory=list)
262
+
263
+ @property
264
+ def _mostlyright(self) -> dict[str, Any]:
265
+ block = self.metadata.get("mostlyright")
266
+ return block if isinstance(block, dict) else {}
267
+
268
+ @property
269
+ def provenance(self) -> dict[str, Any] | None:
270
+ """``metadata.mostlyright.provenance`` for F27, or None when absent."""
271
+ block = self._mostlyright.get("provenance")
272
+ return block if isinstance(block, dict) else None
273
+
274
+ @property
275
+ def leakage(self) -> list[dict[str, Any]]:
276
+ """``metadata.mostlyright.leakage[]`` entries for F28; empty when absent."""
277
+ block = self._mostlyright.get("leakage")
278
+ if not isinstance(block, list):
279
+ return []
280
+ return [entry for entry in block if isinstance(entry, dict)]
281
+
282
+
283
+ def _parse_output(raw: Any) -> Output:
284
+ if not isinstance(raw, dict):
285
+ return Output(kind="unknown")
286
+ output_type = raw.get("output_type")
287
+ if output_type == "stream":
288
+ name = raw.get("name", "stdout")
289
+ kind = "stream-stderr" if name == "stderr" else "stream-stdout"
290
+ return Output(kind=kind, text=_join_source(raw.get("text")))
291
+ if output_type in ("execute_result", "display_data"):
292
+ return Output(
293
+ kind=output_type,
294
+ data=_normalize_bundle(raw.get("data")),
295
+ metadata=raw.get("metadata") if isinstance(raw.get("metadata"), dict) else {},
296
+ execution_count=_coerce_int(raw.get("execution_count")),
297
+ )
298
+ if output_type == "error":
299
+ tb = raw.get("traceback")
300
+ return Output(
301
+ kind="error",
302
+ ename=str(raw["ename"]) if "ename" in raw else None,
303
+ evalue=str(raw["evalue"]) if "evalue" in raw else None,
304
+ traceback=[str(line) for line in tb] if isinstance(tb, list) else [],
305
+ )
306
+ return Output(kind="unknown")
307
+
308
+
309
+ def _coerce_int(value: Any) -> int | None:
310
+ if isinstance(value, bool):
311
+ return None
312
+ if isinstance(value, int):
313
+ return value
314
+ return None
315
+
316
+
317
+ def _parse_cell(raw: Any) -> Cell:
318
+ if not isinstance(raw, dict):
319
+ return Cell(cell_type="unknown")
320
+ cell_type = raw.get("cell_type")
321
+ metadata = raw.get("metadata") if isinstance(raw.get("metadata"), dict) else {}
322
+ cell_id = str(raw.get("id", ""))
323
+ source = _join_source(raw.get("source"))
324
+ if cell_type == "code":
325
+ outputs_raw = raw.get("outputs")
326
+ outputs = [_parse_output(o) for o in outputs_raw] if isinstance(outputs_raw, list) else []
327
+ return Cell(
328
+ cell_type="code",
329
+ cell_id=cell_id,
330
+ source=source,
331
+ metadata=metadata,
332
+ execution_count=_coerce_int(raw.get("execution_count")),
333
+ outputs=outputs,
334
+ )
335
+ if cell_type in ("markdown", "raw"):
336
+ return Cell(cell_type=cell_type, cell_id=cell_id, source=source, metadata=metadata)
337
+ return Cell(cell_type="unknown", cell_id=cell_id, source=source, metadata=metadata)
338
+
339
+
340
+ def coalesce_stream_outputs(outputs: list[Output]) -> list[Output]:
341
+ """Merge consecutive same-channel stream outputs into one (D16, and D17 by symmetry).
342
+
343
+ A kernel emits one ``stream`` record per ``print`` / ``write`` flush, so a single logical block
344
+ of stdout (or stderr) arrives as a run of adjacent :class:`Output` records. Consecutive stdout
345
+ renders as one well, so this concatenates the text of each same-channel run into a
346
+ single ``Output`` while leaving every non-stream record (results, errors, rich data) untouched
347
+ and in order. A fresh ``Output`` is built for each merged run so the parsed model's records are
348
+ never mutated in place.
349
+ """
350
+ merged: list[Output] = []
351
+ for out in outputs:
352
+ if (
353
+ out.kind in ("stream-stdout", "stream-stderr")
354
+ and merged
355
+ and merged[-1].kind == out.kind
356
+ ):
357
+ prev = merged[-1]
358
+ merged[-1] = Output(kind=prev.kind, text=prev.text + out.text)
359
+ else:
360
+ merged.append(out)
361
+ return merged
362
+
363
+
364
+ def parse_notebook(nb: dict[str, Any]) -> Notebook:
365
+ """Build a :class:`Notebook` from raw nbformat JSON, tolerating any malformed shape.
366
+
367
+ A non-dict document, a non-list ``cells``, or a broken cell/output never raises: the offending
368
+ record becomes an ``unknown`` cell or output so the render always completes.
369
+ """
370
+ if not isinstance(nb, dict):
371
+ return Notebook()
372
+ cells_raw = nb.get("cells")
373
+ cells = [_parse_cell(c) for c in cells_raw] if isinstance(cells_raw, list) else []
374
+ metadata = nb.get("metadata") if isinstance(nb.get("metadata"), dict) else {}
375
+ return Notebook(
376
+ nbformat=_coerce_int(nb.get("nbformat")) or 4,
377
+ nbformat_minor=_coerce_int(nb.get("nbformat_minor")) or 0,
378
+ metadata=metadata,
379
+ cells=cells,
380
+ )
381
+
382
+
383
+ @dataclass
384
+ class RenderContext:
385
+ """Per-cell render state threaded to every component.
386
+
387
+ ``prev_execution_count`` is the execution count of the previous code cell, used by B11 to flag
388
+ out-of-order runs; it is None before the first executed cell.
389
+ """
390
+
391
+ mode: str = "static"
392
+ cell_index: int = 0
393
+ cell_count: int = 0
394
+ prev_execution_count: int | None = None
@@ -0,0 +1,40 @@
1
+ """Canonical presentation for exact status values in technical tables.
2
+
3
+ The notebook remains authoritative. This module only normalizes the visible label and tone when a
4
+ recognized status word appears in a semantic status/decision column.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ STATUS_COLUMNS = frozenset({"status", "decision"})
10
+
11
+ _STATUS_LABELS = {
12
+ "pass": ("pass", "Pass"),
13
+ "passed": ("pass", "Pass"),
14
+ "fail": ("fail", "Fail"),
15
+ "failed": ("fail", "Fail"),
16
+ "pending": ("pending", "Pending"),
17
+ "running": ("running", "Running"),
18
+ "warn": ("warning", "Warning"),
19
+ "warning": ("warning", "Warning"),
20
+ "skip": ("neutral", "Skipped"),
21
+ "skipped": ("neutral", "Skipped"),
22
+ "unavailable": ("neutral", "Unavailable"),
23
+ "verified": ("pass", "Verified"),
24
+ "complete": ("pass", "Complete"),
25
+ "completed": ("pass", "Complete"),
26
+ }
27
+
28
+
29
+ def is_status_column(label: str) -> bool:
30
+ """Return whether a table header declares a semantic status column."""
31
+ return label.strip().casefold() in STATUS_COLUMNS
32
+
33
+
34
+ def render_status(value: str) -> str | None:
35
+ """Return fixed safe markup for a recognized exact status value."""
36
+ status = _STATUS_LABELS.get(value.strip().casefold())
37
+ if status is None:
38
+ return None
39
+ tone, label = status
40
+ return f'<span class="nb-status nb-status--{tone}">{label}</span>'