mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,485 @@
1
+ """Declaratively project bounded JSON or NDJSON documents into canonical tabular CSV.
2
+
3
+ The Reader deliberately implements selection rather than inference. A recipe identifies the
4
+ record collection, any arrays to expand, and every output column with RFC 6901 JSON Pointers.
5
+ The same implementation therefore handles a top-level array, an API response envelope, and
6
+ nested observations without source-specific code or executable expressions.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ import math
13
+ import re
14
+ from collections.abc import Iterator, Mapping, Sequence
15
+ from dataclasses import dataclass, field
16
+ from typing import Any
17
+
18
+ from mostlyright.data_harness.formats import (
19
+ FORMAT_MEDIA_TYPES,
20
+ FORMAT_SUFFIXES,
21
+ READER_CONTRACT_VERSION,
22
+ READER_OUTPUT_FORMATS,
23
+ )
24
+ from mostlyright.data_harness.readers.contracts import (
25
+ ReaderBudgets,
26
+ ReaderError,
27
+ ReaderPin,
28
+ ReaderResult,
29
+ bulk_default_budgets,
30
+ )
31
+ from mostlyright.data_harness.readers.tabular import (
32
+ FALLBACK_STEM,
33
+ encode_canonical_csv,
34
+ sealed_filename,
35
+ )
36
+
37
+ __all__ = ["JsonTabularReader"]
38
+
39
+ (_OUTPUT_FORMAT,) = READER_OUTPUT_FORMATS
40
+ _OUTPUT_MEDIA_TYPE = sorted(FORMAT_MEDIA_TYPES[_OUTPUT_FORMAT])[0]
41
+ _OUTPUT_SUFFIX = sorted(FORMAT_SUFFIXES[_OUTPUT_FORMAT])[0]
42
+
43
+ _DOCUMENT_FORMATS = ("json", "ndjson")
44
+ _OPTION_KEYS = frozenset({"document_format", "records_pointer", "expand", "columns"})
45
+ _COLUMN_KEYS = frozenset({"name", "pointer", "required"})
46
+ _SAFE_COLUMN = re.compile(r"^[^\x00-\x1f\x7f]{1,256}$")
47
+ _MAX_POINTER_BYTES = 1_024
48
+ _MAX_EXPANSIONS = 16
49
+ _MAX_JSON_DEPTH = 64
50
+ _MISSING = object()
51
+
52
+ # JSON expansion retains one small immutable view per output row and expansion level. Keeping the
53
+ # family ceiling below the shared million-row default makes even the worst 16-level chain bounded
54
+ # well inside the 512 MiB clean-room boundary, while ordinary non-expanded tables remain large.
55
+ _JSON_DEFAULT_BUDGETS = bulk_default_budgets()
56
+ _MAX_OVERLAY_RECORDS = 200_000
57
+
58
+
59
+ def _refuse_unknown(mapping: Mapping[str, Any], admitted: frozenset[str], subject: str) -> None:
60
+ unknown = sorted(str(key) for key in mapping if key not in admitted)
61
+ if unknown:
62
+ raise ReaderError(
63
+ "READER_OPTIONS",
64
+ subject,
65
+ f"names no such setting: {', '.join(unknown)}; admitted settings are "
66
+ f"{', '.join(sorted(admitted))}",
67
+ )
68
+
69
+
70
+ def _pointer(value: Any, subject: str) -> tuple[str, ...]:
71
+ if not isinstance(value, str):
72
+ raise ReaderError(
73
+ "READER_OPTIONS", subject, "must be an RFC 6901 JSON Pointer of at most 1024 bytes"
74
+ )
75
+ try:
76
+ encoded = value.encode("utf-8")
77
+ except UnicodeEncodeError:
78
+ raise ReaderError(
79
+ "READER_OPTIONS", subject, "must contain valid Unicode scalar values"
80
+ ) from None
81
+ if len(encoded) > _MAX_POINTER_BYTES:
82
+ raise ReaderError(
83
+ "READER_OPTIONS", subject, "must be an RFC 6901 JSON Pointer of at most 1024 bytes"
84
+ )
85
+ if value == "":
86
+ return ()
87
+ if not value.startswith("/"):
88
+ raise ReaderError("READER_OPTIONS", subject, "must be empty or start with '/'")
89
+ tokens: list[str] = []
90
+ for raw in value[1:].split("/"):
91
+ index = 0
92
+ decoded: list[str] = []
93
+ while index < len(raw):
94
+ if raw[index] != "~":
95
+ decoded.append(raw[index])
96
+ index += 1
97
+ continue
98
+ if index + 1 >= len(raw) or raw[index + 1] not in "01":
99
+ raise ReaderError(
100
+ "READER_OPTIONS", subject, "contains an invalid RFC 6901 '~' escape"
101
+ )
102
+ decoded.append("~" if raw[index + 1] == "0" else "/")
103
+ index += 2
104
+ tokens.append("".join(decoded))
105
+ return tuple(tokens)
106
+
107
+
108
+ def _validated_options(options: Any, budgets: ReaderBudgets) -> dict[str, Any]:
109
+ subject = "reader.json.tabular.decode_options"
110
+ if not isinstance(options, Mapping):
111
+ raise ReaderError("READER_OPTIONS", subject, "must be an object")
112
+ _refuse_unknown(options, _OPTION_KEYS, subject)
113
+
114
+ document_format = options.get("document_format", "json")
115
+ if document_format not in _DOCUMENT_FORMATS:
116
+ raise ReaderError(
117
+ "READER_OPTIONS",
118
+ f"{subject}.document_format",
119
+ f"must be one of {', '.join(_DOCUMENT_FORMATS)}",
120
+ )
121
+ records_pointer = options.get("records_pointer", "")
122
+ _pointer(records_pointer, f"{subject}.records_pointer")
123
+
124
+ expand = options.get("expand", [])
125
+ if not isinstance(expand, list) or len(expand) > _MAX_EXPANSIONS:
126
+ raise ReaderError(
127
+ "READER_OPTIONS",
128
+ f"{subject}.expand",
129
+ f"must be an array of at most {_MAX_EXPANSIONS} JSON Pointers",
130
+ )
131
+ for index, item in enumerate(expand):
132
+ _pointer(item, f"{subject}.expand[{index}]")
133
+
134
+ columns = options.get("columns")
135
+ if not isinstance(columns, list) or not columns or len(columns) > budgets.max_columns:
136
+ raise ReaderError(
137
+ "READER_OPTIONS",
138
+ f"{subject}.columns",
139
+ "must be a non-empty array within the column budget",
140
+ )
141
+ admitted_columns: list[dict[str, Any]] = []
142
+ names: list[str] = []
143
+ for index, column in enumerate(columns):
144
+ item_subject = f"{subject}.columns[{index}]"
145
+ if not isinstance(column, Mapping):
146
+ raise ReaderError("READER_OPTIONS", item_subject, "must be an object")
147
+ _refuse_unknown(column, _COLUMN_KEYS, item_subject)
148
+ name = column.get("name")
149
+ if not isinstance(name, str) or _SAFE_COLUMN.fullmatch(name) is None:
150
+ raise ReaderError(
151
+ "READER_OPTIONS",
152
+ f"{item_subject}.name",
153
+ "must be bounded text without control characters",
154
+ )
155
+ pointer = column.get("pointer")
156
+ _pointer(pointer, f"{item_subject}.pointer")
157
+ required = column.get("required", True)
158
+ if not isinstance(required, bool):
159
+ raise ReaderError("READER_OPTIONS", f"{item_subject}.required", "must be boolean")
160
+ names.append(name)
161
+ admitted_columns.append({"name": name, "pointer": pointer, "required": required})
162
+ if len(set(names)) != len(names):
163
+ raise ReaderError("READER_OPTIONS", f"{subject}.columns", "column names must be unique")
164
+
165
+ return {
166
+ "document_format": document_format,
167
+ "records_pointer": records_pointer,
168
+ "expand": list(expand),
169
+ "columns": admitted_columns,
170
+ }
171
+
172
+
173
+ def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
174
+ value: dict[str, Any] = {}
175
+ for key, item in pairs:
176
+ if key in value:
177
+ raise ReaderError(
178
+ "READER_ADMISSION",
179
+ "reader.json.tabular.content",
180
+ f"JSON object repeats key {key!r}",
181
+ )
182
+ value[key] = item
183
+ return value
184
+
185
+
186
+ def _decode_documents(content: bytes, document_format: str, budgets: ReaderBudgets) -> list[Any]:
187
+ if not isinstance(content, (bytes, bytearray)):
188
+ raise ReaderError("READER_ADMISSION", "reader.json.tabular.content", "must be exact bytes")
189
+ if not content or len(content) > budgets.max_input_bytes:
190
+ raise ReaderError(
191
+ "READER_BUDGET",
192
+ "reader.json.tabular.content",
193
+ "is empty or exceeds the input byte budget",
194
+ )
195
+ try:
196
+ text = bytes(content).removeprefix(b"\xef\xbb\xbf").decode("utf-8", errors="strict")
197
+ except UnicodeDecodeError:
198
+ raise ReaderError(
199
+ "READER_ADMISSION", "reader.json.tabular.content", "must be strict UTF-8"
200
+ ) from None
201
+
202
+ decoder = json.JSONDecoder(
203
+ object_pairs_hook=_unique_object,
204
+ parse_constant=lambda value: (_ for _ in ()).throw(
205
+ ReaderError(
206
+ "READER_ADMISSION",
207
+ "reader.json.tabular.content",
208
+ f"non-finite JSON number {value!r} is not admitted",
209
+ )
210
+ ),
211
+ )
212
+ try:
213
+ if document_format == "json":
214
+ documents = [decoder.decode(text)]
215
+ else:
216
+ documents = [decoder.decode(line) for line in text.splitlines() if line.strip()]
217
+ except ReaderError:
218
+ raise
219
+ except (json.JSONDecodeError, ValueError):
220
+ raise ReaderError(
221
+ "READER_DECODE", "reader.json.tabular.content", "JSON syntax is malformed"
222
+ ) from None
223
+ except RecursionError:
224
+ raise ReaderError(
225
+ "READER_BUDGET", "reader.json.tabular.content", "JSON nesting exceeds the depth limit"
226
+ ) from None
227
+ if not documents:
228
+ raise ReaderError(
229
+ "READER_ADMISSION", "reader.json.tabular.content", "contains no JSON documents"
230
+ )
231
+ return documents
232
+
233
+
234
+ def _validate_tree(value: Any) -> None:
235
+ pending: list[tuple[Any, int]] = [(value, 0)]
236
+ while pending:
237
+ item, depth = pending.pop()
238
+ if depth > _MAX_JSON_DEPTH:
239
+ raise ReaderError(
240
+ "READER_BUDGET",
241
+ "reader.json.tabular.content",
242
+ f"JSON nesting exceeds the fixed {_MAX_JSON_DEPTH}-level limit",
243
+ )
244
+ if isinstance(item, dict):
245
+ for key in item:
246
+ try:
247
+ key.encode("utf-8")
248
+ except UnicodeEncodeError:
249
+ raise ReaderError(
250
+ "READER_ADMISSION",
251
+ "reader.json.tabular.content",
252
+ "contains an object key that is not valid Unicode scalar text",
253
+ ) from None
254
+ pending.extend((child, depth + 1) for child in item.values())
255
+ elif isinstance(item, list):
256
+ pending.extend((child, depth + 1) for child in item)
257
+ elif isinstance(item, float) and not math.isfinite(item):
258
+ raise ReaderError(
259
+ "READER_ADMISSION",
260
+ "reader.json.tabular.content",
261
+ "contains a number outside finite binary64 range",
262
+ )
263
+ elif isinstance(item, str):
264
+ try:
265
+ item.encode("utf-8")
266
+ except UnicodeEncodeError:
267
+ raise ReaderError(
268
+ "READER_ADMISSION",
269
+ "reader.json.tabular.content",
270
+ "contains a string that is not valid Unicode scalar text",
271
+ ) from None
272
+ elif item is not None and not isinstance(item, str | bool | int | float):
273
+ raise ReaderError(
274
+ "READER_ADMISSION",
275
+ "reader.json.tabular.content",
276
+ f"contains unsupported {type(item).__name__} value",
277
+ )
278
+
279
+
280
+ def _array_index(token: str, subject: str) -> int:
281
+ if (
282
+ token == "-"
283
+ or not token.isascii()
284
+ or not token.isdigit()
285
+ or (len(token) > 1 and token.startswith("0"))
286
+ ):
287
+ raise ReaderError("READER_ADMISSION", subject, "does not name an RFC 6901 array index")
288
+ return int(token)
289
+
290
+
291
+ @dataclass(frozen=True, slots=True)
292
+ class _ExpandedRecord:
293
+ """One constant-size overlay; parents and wide source containers are never copied."""
294
+
295
+ parent: Any
296
+ pointer: str
297
+ replacement: Any
298
+
299
+
300
+ def _overlay(value: Any, pointer: str) -> tuple[Any, tuple[str, ...]]:
301
+ tokens = _pointer(pointer, "reader.json.tabular.pointer")
302
+ current = value
303
+ while isinstance(current, _ExpandedRecord):
304
+ replacement_tokens = _pointer(current.pointer, "reader.json.tabular.expand")
305
+ if tokens[: len(replacement_tokens)] == replacement_tokens:
306
+ return current.replacement, tokens[len(replacement_tokens) :]
307
+ current = current.parent
308
+ return current, tokens
309
+
310
+
311
+ def _resolve(value: Any, pointer: str, subject: str, *, missing: Any = _MISSING) -> Any:
312
+ current, tokens = _overlay(value, pointer)
313
+ for token in tokens:
314
+ if isinstance(current, dict):
315
+ if token not in current:
316
+ if missing is not _MISSING:
317
+ return missing
318
+ raise ReaderError("READER_ADMISSION", subject, f"does not exist at {pointer!r}")
319
+ current = current[token]
320
+ elif isinstance(current, list):
321
+ index = _array_index(token, subject)
322
+ if index >= len(current):
323
+ if missing is not _MISSING:
324
+ return missing
325
+ raise ReaderError("READER_ADMISSION", subject, f"does not exist at {pointer!r}")
326
+ current = current[index]
327
+ else:
328
+ raise ReaderError("READER_ADMISSION", subject, f"crosses a scalar at {token!r}")
329
+ return current
330
+
331
+
332
+ def _selected_records(documents: Sequence[Any], pointer: str, budgets: ReaderBudgets) -> list[Any]:
333
+ selected: list[Any] = []
334
+ for index, document in enumerate(documents):
335
+ _validate_tree(document)
336
+ value = _resolve(
337
+ document, pointer, f"reader.json.tabular.documents[{index}].records_pointer"
338
+ )
339
+ if isinstance(value, list):
340
+ if len(selected) + len(value) > budgets.max_rows:
341
+ raise ReaderError(
342
+ "READER_BUDGET",
343
+ f"reader.json.tabular.documents[{index}].records_pointer",
344
+ "selection exceeds the row budget",
345
+ )
346
+ selected.extend(value)
347
+ elif isinstance(value, dict):
348
+ if len(selected) >= budgets.max_rows:
349
+ raise ReaderError(
350
+ "READER_BUDGET",
351
+ f"reader.json.tabular.documents[{index}].records_pointer",
352
+ "selection exceeds the row budget",
353
+ )
354
+ selected.append(value)
355
+ else:
356
+ raise ReaderError(
357
+ "READER_ADMISSION",
358
+ f"reader.json.tabular.documents[{index}].records_pointer",
359
+ "must select an object or an array of records",
360
+ )
361
+ return selected
362
+
363
+
364
+ def _expanded_records(
365
+ records: list[Any], pointers: Sequence[str], budgets: ReaderBudgets
366
+ ) -> list[Any]:
367
+ current = records
368
+ overlay_records = 0
369
+ for expansion_index, pointer in enumerate(pointers):
370
+ expanded: list[Any] = []
371
+ for record_index, record in enumerate(current):
372
+ subject = f"reader.json.tabular.expand[{expansion_index}].records[{record_index}]"
373
+ values = _resolve(record, pointer, subject)
374
+ if not isinstance(values, list):
375
+ raise ReaderError("READER_ADMISSION", subject, "must select an array")
376
+ if len(expanded) + len(values) > budgets.max_rows:
377
+ raise ReaderError("READER_BUDGET", subject, "expansion exceeds the row budget")
378
+ if overlay_records + len(values) > _MAX_OVERLAY_RECORDS:
379
+ raise ReaderError(
380
+ "READER_BUDGET", subject, "expansion exceeds the cumulative overlay budget"
381
+ )
382
+ expanded.extend(_ExpandedRecord(record, pointer, item) for item in values)
383
+ overlay_records += len(values)
384
+ current = expanded
385
+ if len(current) > budgets.max_rows:
386
+ raise ReaderError("READER_BUDGET", "reader.json.tabular.records", "exceeds the row budget")
387
+ return current
388
+
389
+
390
+ def _project(records: Sequence[Any], columns: Sequence[Mapping[str, Any]]) -> Iterator[list[Any]]:
391
+ for row_index, record in enumerate(records):
392
+ row: list[Any] = []
393
+ for column in columns:
394
+ subject = f"reader.json.tabular.rows[{row_index}].{column['name']}"
395
+ value = _resolve(
396
+ record,
397
+ str(column["pointer"]),
398
+ subject,
399
+ missing=_MISSING if bool(column["required"]) else None,
400
+ )
401
+ if isinstance(value, dict | list):
402
+ raise ReaderError(
403
+ "READER_ADMISSION",
404
+ subject,
405
+ "selects an object or array; expand it or select a scalar child",
406
+ )
407
+ row.append(value)
408
+ yield row
409
+
410
+
411
+ @dataclass(frozen=True)
412
+ class JsonTabularReader:
413
+ """A generic, versioned JSON/NDJSON-to-table Reader."""
414
+
415
+ family_id: str = "json.tabular"
416
+ family_version: str = "1.0.0"
417
+ contract_version: str = READER_CONTRACT_VERSION
418
+ output_format: str = _OUTPUT_FORMAT
419
+ accepted_media_types: tuple[str, ...] = ("application/json", "application/x-ndjson")
420
+ default_budgets: ReaderBudgets = field(default_factory=lambda: _JSON_DEFAULT_BUDGETS)
421
+
422
+ def validate_options(self, options: Mapping[str, Any]) -> Mapping[str, Any]:
423
+ return _validated_options(options, self.default_budgets)
424
+
425
+ def decode(self, content: bytes, pin: ReaderPin, budgets: ReaderBudgets) -> ReaderResult:
426
+ options = _validated_options(pin.decode_options, budgets)
427
+ documents = _decode_documents(content, str(options["document_format"]), budgets)
428
+ records = _selected_records(documents, str(options["records_pointer"]), budgets)
429
+ records = _expanded_records(records, list(options["expand"]), budgets)
430
+ columns = list(options["columns"])
431
+ if len(records) * len(columns) > budgets.max_declared_cells:
432
+ raise ReaderError(
433
+ "READER_BUDGET", "reader.json.tabular.records", "exceeds the cell budget"
434
+ )
435
+ names = tuple(str(column["name"]) for column in columns)
436
+ return ReaderResult(
437
+ content=encode_canonical_csv(names, _project(records, columns), budgets=budgets),
438
+ data_format=_OUTPUT_FORMAT,
439
+ media_type=_OUTPUT_MEDIA_TYPE,
440
+ filename=sealed_filename(FALLBACK_STEM, suffix=_OUTPUT_SUFFIX),
441
+ row_count=len(records),
442
+ column_names=names,
443
+ declared_cell_count=len(records) * len(names),
444
+ )
445
+
446
+
447
+ @dataclass(frozen=True)
448
+ class JsonTabularReaderV1_1(JsonTabularReader):
449
+ """``json.tabular@1.1.0`` admits the weak labels, and decodes nothing new.
450
+
451
+ A publisher that serves a JSON document under ``text/plain`` is not describing a different
452
+ document; it is declining to describe the document at all.
453
+ ``acquisition/http.py`` already writes that rule down for the byte-stream spelling -- a weak
454
+ label "says only that a server declined to say anything", and the control is the structural
455
+ check behind the sandbox boundary, never the header. Version ``1.0.0`` admitted the two
456
+ labels that name the format outright, which refused whole archives that were never unfit:
457
+ a large share of public agency JSON and NDJSON endpoints answer under ``text/plain``.
458
+
459
+ This is an admission change and not a decode change. ``decode`` is inherited untouched, so
460
+ the bytes a recipe seals under this coordinate are the bytes ``1.0.0`` would have sealed;
461
+ what moves is only which responses reach the decoder. It is a new coordinate for the reason
462
+ ``delimited_text@1.1.0`` and ``archive.zip@1.2.0`` are: ``1.0.0`` stays closed to the weak
463
+ labels, so an already-approved recipe cannot silently begin admitting responses its review
464
+ never saw.
465
+
466
+ The label is the only thing widened. ``text/html`` remains outside every allowlist, which is
467
+ what keeps the bot-wall refusal intact, and the decoder's own refusals -- a document that is
468
+ not valid JSON or NDJSON, a pointer that selects nothing, a column that resolves to an object
469
+ or array, and every budget bound -- are unchanged and remain the whole admission. Archive and
470
+ executable bytes arriving under these labels are refused as invalid JSON rather than by
471
+ prefix: the magic table is enforced in ``acquisition/parsing`` and in the container families,
472
+ and this family imports neither.
473
+
474
+ ``application/octet-stream`` is deliberately not added here. It is the weak label an object
475
+ store uses for a stored document, and admitting it is a separate widening with its own
476
+ evidence, taken the way ``archive.zip`` took its two: one label at a time, each its own
477
+ coordinate.
478
+ """
479
+
480
+ family_version: str = "1.1.0"
481
+ accepted_media_types: tuple[str, ...] = (
482
+ "application/json",
483
+ "application/x-ndjson",
484
+ "text/plain",
485
+ )