mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,366 @@
1
+ """The polars backend: ``PolarsBackend`` over an eager ``pl.DataFrame`` container.
2
+
3
+ The direct twin of ``PandasBackend`` with the engine swapped. ``PolarsBackend`` re-owns the
4
+ narrow ``clean -> join -> select`` triple behind the ``Backend`` protocol; every other
5
+ candidate byte is still produced by the shared layer (the one pyarrow sink, the canonical
6
+ schema, ``_coerce_native_rows``, the evidence serializer). Because the shared layer
7
+ downstream is unchanged, the backend's job is to reproduce the reference compute
8
+ byte-identically while genuinely exercising a polars object.
9
+
10
+ Design (the load-bearing execution posture):
11
+
12
+ - The row-shape transforms (``clean``/``join``/``select``) are computed **value-wise** in
13
+ plain Python, mirroring ``pipeline._apply_cleaning`` / ``_join`` / ``_select`` exactly so
14
+ unicode-whitespace, empty-vs-null, ordering, and the typed ``BuildError`` refusals match
15
+ the reference bit-for-bit. Every value-level cast routes through the shared
16
+ ``pipeline._cast`` — never an engine-native parser, which is what keeps cast results and
17
+ cast refusals identical across engines.
18
+ - The op's output is then materialized as an **eager** ``pl.DataFrame`` — one explicit
19
+ polars dtype per column drawn from the closed ``_POLARS_BY_LOGICAL`` vocabulary aligned to
20
+ the sink's ``pipeline._ARROW_BY_LOGICAL`` map, constructed from the already-typed native
21
+ values (never inferred, never an engine-native cast token, never a ``pl.DataFrame`` handed
22
+ to the sink). Explicit ``pl.Int64``/``pl.Float64``/``pl.String``/``pl.Boolean``/``pl.Date``/
23
+ UTC ``pl.Datetime`` columns carry
24
+ true polars nulls (never ``NaN``) and keep an ``int64`` column ``int64`` under nulls (no
25
+ int->float promotion).
26
+ - The frame is used **only** as a typed shaping/materialization container: it is never a
27
+ lazy frame, is never collected, never runs a polars ``join``/``group_by``, never enables the
28
+ global string cache, and never uses a polars categorical type. Order is imposed
29
+ positionally in Python (strictly stronger than any polars ordering guarantee), so polars'
30
+ documented ordering/streaming/thread nondeterminism can never touch the digest path, and the
31
+ streaming engine is structurally unreachable.
32
+ - The frame is converted back to an ordered native ``list[dict]`` at the boundary
33
+ (``_rows_from_frame``): column order = the frame's schema order, and each cell is read
34
+ through polars' native ``to_dicts()`` so a polars null becomes Python ``None``, a
35
+ ``pl.String`` cell (Arrow ``large_utf8`` internally) becomes a native ``str`` with no width
36
+ tag, a ``pl.Date`` cell becomes ``datetime.date``, and an ``pl.Int64``/``pl.Float64`` cell
37
+ becomes a native ``int``/``float`` — no engine scalar. Because the backend returns native
38
+ rows (not a ``pl.DataFrame`` or its Arrow table), the ``large_utf8`` distinction is gone at
39
+ the boundary and the shared sink's ``pa.string()`` imposition normalizes it with no sink
40
+ change. ``pipeline._coerce_native_rows`` remains the shared backstop.
41
+
42
+ ``import polars`` is **function-local** (mirroring ``reference.py`` / ``pandas_backend.py``'s
43
+ lazy-engine-import ethos), so ``import backends`` and constructing ``PolarsBackend()`` at
44
+ registry load never import polars. The module is named ``polars_backend`` (not ``polars``) to
45
+ avoid the ``import polars`` self-shadow footgun.
46
+ """
47
+
48
+ from __future__ import annotations
49
+
50
+ from typing import Any
51
+
52
+ from mostlyright.data_harness.local_contracts import CleaningStep, TablePlan
53
+
54
+
55
+ class PolarsBackend:
56
+ """polars engine for the ``clean -> join -> select`` triple over an eager frame."""
57
+
58
+ # A plain literal (not a lazy ``pipeline`` import), so constructing the backend at
59
+ # ``registry`` module load never imports ``pipeline`` or polars. The seam resolves the
60
+ # backend by this exact name; any drift fails closed at ``get_backend`` (BACKEND_UNKNOWN).
61
+ name: str = "polars"
62
+
63
+ def versions(self) -> dict[str, str]:
64
+ # Provenance only (engine-runtime.v2 ``backend_versions``). The polars import is
65
+ # function-local (only reached during a build, when polars is already dispatched), so
66
+ # ``import backends`` / constructing the backend stay polars-free. The version is read
67
+ # from the module attribute ``pl.__version__`` rather than the stdlib metadata-version
68
+ # helper: that helper's module name is one of the dynamic-import tokens banned under
69
+ # ``backends/`` by ``test_backend_registry.py``. For a pinned release the two
70
+ # spellings are identical.
71
+ import polars as pl
72
+
73
+ return {"polars": str(pl.__version__)}
74
+
75
+ def materialize_graph_rows(
76
+ self,
77
+ columns: tuple[str, ...],
78
+ rows: tuple[tuple[Any, ...], ...],
79
+ ) -> tuple[tuple[Any, ...], ...]:
80
+ """Exercise a Polars object frame while preserving contract-owned Python scalars."""
81
+
82
+ import polars as pl
83
+
84
+ if any(len(row) != len(columns) for row in rows):
85
+ raise ValueError("graph row width differs from its columns")
86
+ frame = pl.DataFrame(
87
+ [
88
+ pl.Series(column, [row[index] for row in rows], dtype=pl.Object)
89
+ for index, column in enumerate(columns)
90
+ ]
91
+ )
92
+ return tuple(tuple(row) for row in frame.iter_rows())
93
+
94
+ def clean(
95
+ self,
96
+ rows: list[dict[str, Any]],
97
+ columns: list[str],
98
+ lineage: dict[str, dict[str, Any]],
99
+ step: CleaningStep,
100
+ ) -> tuple[list[dict[str, Any]], list[str], dict[str, dict[str, Any]]]:
101
+ # Value-wise reproduction of ``pipeline._apply_cleaning`` (the DataFrame is a shaping
102
+ # container, not a transform engine): shared ``_require_columns`` refusals, Python
103
+ # shared ASCII-only trim, ``== ""`` -> None, positional
104
+ # rename with ``RENAME_COLLISION``, and every cast through the shared ``_cast``.
105
+ from mostlyright.data_harness import pipeline
106
+
107
+ output_rows = [dict(row) for row in rows]
108
+ output_columns = list(columns)
109
+ output_lineage = {
110
+ column: {**entry, "operations": list(entry["operations"])}
111
+ for column, entry in lineage.items()
112
+ }
113
+ if step.operation in {"trim", "empty_to_null"}:
114
+ target_columns = tuple(step.columns)
115
+ pipeline._require_columns(output_columns, target_columns, step.operation)
116
+ for row in output_rows:
117
+ for column in target_columns:
118
+ value = row[column]
119
+ if step.operation == "trim":
120
+ if value is not None and not isinstance(value, str):
121
+ raise pipeline.BuildError(
122
+ "CLEAN_TYPE", f"trim requires strings in {column!r}"
123
+ )
124
+ row[column] = None if value is None else pipeline._trim_ascii(value)
125
+ elif value == "":
126
+ row[column] = None
127
+ for column in target_columns:
128
+ output_lineage[column]["operations"].append(step.operation)
129
+ elif step.operation == "rename":
130
+ mapping = dict(step.columns)
131
+ pipeline._require_columns(output_columns, tuple(mapping), "rename")
132
+ final_columns = [mapping.get(column, column) for column in output_columns]
133
+ if len(final_columns) != len(set(final_columns)):
134
+ raise pipeline.BuildError("RENAME_COLLISION", "rename creates a column collision")
135
+ output_rows = [
136
+ {mapping.get(column, column): row[column] for column in output_columns}
137
+ for row in output_rows
138
+ ]
139
+ new_lineage: dict[str, dict[str, Any]] = {}
140
+ for column in output_columns:
141
+ target = mapping.get(column, column)
142
+ entry = output_lineage[column]
143
+ if target != column:
144
+ entry["operations"].append(f"rename:{column}->{target}")
145
+ new_lineage[target] = entry
146
+ output_columns = final_columns
147
+ output_lineage = new_lineage
148
+ elif step.operation == "cast":
149
+ mapping = dict(step.columns)
150
+ pipeline._require_columns(output_columns, tuple(mapping), "cast")
151
+ for row_index, row in enumerate(output_rows, start=1):
152
+ for column, target in mapping.items():
153
+ row[column] = pipeline._cast(row[column], target, column, row_index)
154
+ for column, target in mapping.items():
155
+ output_lineage[column]["operations"].append(f"cast:{target}")
156
+ else: # pragma: no cover - contract parser makes this unreachable
157
+ raise pipeline.ContractError(
158
+ "plan.cleaning.operation",
159
+ "CLEAN_OPERATION_UNSUPPORTED",
160
+ f"unsupported operation: {step.operation}",
161
+ )
162
+
163
+ shaped = _rows_from_frame(
164
+ _frame_from_rows(
165
+ output_rows, output_columns, _logical_by_column(output_lineage, output_columns)
166
+ )
167
+ )
168
+ return shaped, output_columns, output_lineage
169
+
170
+ def join(
171
+ self,
172
+ plan: TablePlan,
173
+ rows_by_source: dict[str, list[dict[str, Any]]],
174
+ columns_by_source: dict[str, list[str]],
175
+ lineage_by_source: dict[str, dict[str, dict[str, Any]]],
176
+ ) -> tuple[
177
+ list[dict[str, Any]],
178
+ list[str],
179
+ dict[str, dict[str, Any]],
180
+ dict[str, Any],
181
+ ]:
182
+ # Value-wise reproduction of ``pipeline._join``. Left-source order is imposed
183
+ # positionally (iterate the left rows; look up the right by an equality index) — never
184
+ # a polars ``join``/``group_by``, so polars' ordering nondeterminism never touches the
185
+ # digest. Same guards and typed refusals; output columns ``[*left_columns,
186
+ # *right_nonkeys]``; merged lineage; and a ``report`` byte-identical to the reference
187
+ # one. The report counts stay native ``int`` because they are pure-Python
188
+ # ``len(...)``; an engine scalar there would change the evidence bytes.
189
+ from mostlyright.data_harness import pipeline
190
+
191
+ spec = plan.join
192
+ left_rows = rows_by_source[spec.left]
193
+ right_rows = rows_by_source[spec.right]
194
+ left_columns = columns_by_source[spec.left]
195
+ right_columns = columns_by_source[spec.right]
196
+ pipeline._require_columns(left_columns, spec.on, "join.left keys")
197
+ pipeline._require_columns(right_columns, spec.on, "join.right keys")
198
+ for key in spec.on:
199
+ left_types = {type(row[key]) for row in left_rows if row[key] is not None}
200
+ right_types = {type(row[key]) for row in right_rows if row[key] is not None}
201
+ if any(row[key] is None for row in left_rows + right_rows):
202
+ raise pipeline.BuildError("JOIN_NULL_KEY", f"join key {key!r} contains null")
203
+ if len(left_types) != 1 or left_types != right_types:
204
+ raise pipeline.BuildError(
205
+ "JOIN_KEY_TYPE", f"join key {key!r} types do not match exactly"
206
+ )
207
+ right_nonkeys = [column for column in right_columns if column not in spec.on]
208
+ collisions = set(left_columns) & set(right_nonkeys)
209
+ if collisions:
210
+ raise pipeline.BuildError(
211
+ "JOIN_COLUMN_COLLISION", f"join column collision: {sorted(collisions)}"
212
+ )
213
+ right_index: dict[tuple[Any, ...], dict[str, Any]] = {}
214
+ for row in right_rows:
215
+ key = tuple(row[column] for column in spec.on)
216
+ if key in right_index:
217
+ raise pipeline.BuildError("JOIN_CARDINALITY", "right join keys are not unique")
218
+ right_index[key] = row
219
+ if spec.cardinality == "one_to_one":
220
+ left_keys = [tuple(row[column] for column in spec.on) for row in left_rows]
221
+ if len(left_keys) != len(set(left_keys)):
222
+ raise pipeline.BuildError("JOIN_CARDINALITY", "left join keys are not unique")
223
+ output: list[dict[str, Any]] = []
224
+ unmatched = 0
225
+ for left_row in left_rows:
226
+ key = tuple(left_row[column] for column in spec.on)
227
+ right_row = right_index.get(key)
228
+ if right_row is None:
229
+ unmatched += 1
230
+ merged = {**left_row, **{column: None for column in right_nonkeys}}
231
+ else:
232
+ merged = {**left_row, **{column: right_row[column] for column in right_nonkeys}}
233
+ output.append(merged)
234
+ multiplier = len(output) / len(left_rows)
235
+ if unmatched:
236
+ raise pipeline.BuildError("JOIN_UNMATCHED", f"left join has {unmatched} unmatched rows")
237
+ if multiplier != 1.0:
238
+ raise pipeline.BuildError("JOIN_MULTIPLIER", f"join row multiplier is {multiplier}")
239
+ columns = [*left_columns, *right_nonkeys]
240
+ lineage = {
241
+ **lineage_by_source[spec.left],
242
+ **{column: lineage_by_source[spec.right][column] for column in right_nonkeys},
243
+ }
244
+ report = {
245
+ "schema_version": "join-evidence.v1",
246
+ "left_source": spec.left,
247
+ "right_source": spec.right,
248
+ "keys": list(spec.on),
249
+ "cardinality": spec.cardinality,
250
+ "left_rows": len(left_rows),
251
+ "right_rows": len(right_rows),
252
+ "output_rows": len(output),
253
+ "matched_left_rows": len(output) - unmatched,
254
+ "unmatched_left_rows": unmatched,
255
+ "row_multiplier": multiplier,
256
+ "stable_order": "left_source_order",
257
+ }
258
+ shaped = _rows_from_frame(
259
+ _frame_from_rows(output, columns, _logical_by_column(lineage, columns))
260
+ )
261
+ return shaped, columns, lineage, report
262
+
263
+ def select(
264
+ self,
265
+ plan: TablePlan,
266
+ rows: list[dict[str, Any]],
267
+ columns: list[str],
268
+ lineage: dict[str, dict[str, Any]],
269
+ ) -> tuple[list[dict[str, Any]], dict[str, dict[str, Any]]]:
270
+ # Value-wise reproduction of ``pipeline._select``: project/order columns to
271
+ # ``plan.select`` (shared ``_require_columns`` for the missing-column refusal);
272
+ # ``selected_lineage`` in ``plan.select`` order.
273
+ from mostlyright.data_harness import pipeline
274
+
275
+ pipeline._require_columns(columns, plan.select, "select")
276
+ selected_rows = [{column: row[column] for column in plan.select} for row in rows]
277
+ selected_lineage = {column: lineage[column] for column in plan.select}
278
+ shaped = _rows_from_frame(
279
+ _frame_from_rows(
280
+ selected_rows,
281
+ list(plan.select),
282
+ _logical_by_column(selected_lineage, plan.select),
283
+ )
284
+ )
285
+ return shaped, selected_lineage
286
+
287
+
288
+ def _logical_by_column(lineage: dict[str, dict[str, Any]], columns: Any) -> dict[str, str]:
289
+ """Map each column to its closed logical type via the shared ``_logical_type_for``.
290
+
291
+ The logical type is drawn only from the column's lineage ``operations`` (the last
292
+ ``cast:<target>``, else ``"string"``) — never from an engine scalar's class name — so
293
+ the polars dtype the frame is built with matches the sink's imposed schema.
294
+ """
295
+
296
+ from mostlyright.data_harness import pipeline
297
+
298
+ return {column: pipeline._logical_type_for(lineage[column]) for column in columns}
299
+
300
+
301
+ def _polars_dtype_for(logical: str) -> Any:
302
+ """Return the explicit polars dtype for one authoritative local logical type.
303
+
304
+ ``_POLARS_BY_LOGICAL`` is aligned one-to-one with the sink's closed
305
+ ``pipeline._ARROW_BY_LOGICAL`` vocabulary, so the frame's per-column dtype is canonical and
306
+ can never be inferred/lenient. Built with a function-local ``pl`` reference so ``import
307
+ backends`` never imports polars.
308
+ """
309
+
310
+ import polars as pl
311
+
312
+ _POLARS_BY_LOGICAL = {
313
+ "string": pl.String,
314
+ "int64": pl.Int64,
315
+ "float64": pl.Float64,
316
+ "boolean": pl.Boolean,
317
+ "date": pl.Date,
318
+ "timestamp_utc": pl.Datetime(time_unit="us", time_zone="UTC"),
319
+ }
320
+ return _POLARS_BY_LOGICAL[logical]
321
+
322
+
323
+ def _frame_from_rows(
324
+ rows: list[dict[str, Any]], columns: Any, logical_by_column: dict[str, str]
325
+ ) -> Any:
326
+ """Build an EAGER ``pl.DataFrame`` with an explicit per-column dtype.
327
+
328
+ Each column is constructed as a ``pl.Series(name, values, dtype=<explicit>)`` from the
329
+ already-typed native values — never inferred, never an engine-native cast token, never a
330
+ ``pl.DataFrame`` handed to the sink. Explicit closed dtypes carry true polars nulls (never
331
+ ``NaN``) and keep ``int64`` as ``int64`` under nulls (no float promotion). The zero-column
332
+ shape is unreachable in the contract
333
+ (every stage yields >=1 output column) but is handled defensively — an empty frame — so a
334
+ degenerate op cannot raise here.
335
+ """
336
+
337
+ import polars as pl
338
+
339
+ column_order = list(columns)
340
+ if not column_order:
341
+ return pl.DataFrame()
342
+ series = [
343
+ pl.Series(
344
+ column,
345
+ [row[column] for row in rows],
346
+ dtype=_polars_dtype_for(logical_by_column[column]),
347
+ )
348
+ for column in column_order
349
+ ]
350
+ return pl.DataFrame(series)
351
+
352
+
353
+ def _rows_from_frame(frame: Any) -> list[dict[str, Any]]:
354
+ """Convert an eager ``pl.DataFrame`` back to an ordered native ``list[dict]``.
355
+
356
+ Column (key) order follows the frame's schema order; each cell is read through polars'
357
+ native ``to_dicts()`` so a polars null becomes Python ``None`` and every scalar is a native
358
+ Python builtin — ``pl.String`` (Arrow ``large_utf8``) -> ``str`` with no width tag,
359
+ ``pl.Int64`` -> ``int``, ``pl.Float64`` -> ``float``, ``pl.Date`` -> ``datetime.date`` — no
360
+ engine scalar, no polars categorical type. Returning native rows (not a ``pl.DataFrame`` /
361
+ its Arrow table) is what strips the ``large_utf8`` width tag, so the shared sink's imposed
362
+ ``pa.string()`` normalizes it with no sink change. ``pipeline._coerce_native_rows`` remains
363
+ the shared backstop downstream.
364
+ """
365
+
366
+ return frame.to_dicts()
@@ -0,0 +1,124 @@
1
+ """The named backend protocol at the ``clean -> join -> select`` row-transform seam.
2
+
3
+ The protocol boundary is the inner ``clean -> join -> select`` triple over ``list[dict]``
4
+ of native Python scalars — not the outer ``pipeline._replay_candidate_derivations``
5
+ signature. Making the whole replay pluggable would force every backend to re-own the
6
+ Parquet sink, the casts, validation and evidence, which is the entire byte-parity risk
7
+ surface. Keeping the seam narrow means the shared layer still produces every candidate
8
+ byte no matter which engine is selected.
9
+
10
+ The three method signatures are the signatures of ``pipeline._apply_cleaning`` /
11
+ ``pipeline._join`` / ``pipeline._select`` (only ``self`` is added), so a backend is a
12
+ drop-in for those functions and ``reference.py`` can delegate to them unchanged.
13
+
14
+ A backend must be byte-identical to the reference implementation, not merely equivalent:
15
+ same row order, same column order, same Python scalar types, same typed refusals in the
16
+ same order. Any divergence changes the sealed candidate digest.
17
+
18
+ This module imports only ``local_contracts`` for typing and never ``pipeline``, which
19
+ keeps ``import backends`` free of a module-load dependency on ``pipeline``. ``pipeline``
20
+ imports the registry, so a top-level ``pipeline`` import here would close a cycle.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ from typing import Any, Protocol, runtime_checkable
26
+
27
+ from mostlyright.data_harness.local_contracts import CleaningStep, TablePlan
28
+
29
+
30
+ @runtime_checkable
31
+ class Backend(Protocol):
32
+ """A named engine for the ``clean -> join -> select`` row-transform triple."""
33
+
34
+ name: str
35
+
36
+ def versions(self) -> dict[str, str]:
37
+ """Engine-library versions for provenance (``{}`` for a pure-Python backend)."""
38
+ ...
39
+
40
+ def materialize_graph_rows(
41
+ self,
42
+ columns: tuple[str, ...],
43
+ rows: tuple[tuple[Any, ...], ...],
44
+ ) -> tuple[tuple[Any, ...], ...]:
45
+ """Round-trip one graph node through this engine's deterministic container."""
46
+ ...
47
+
48
+ def clean(
49
+ self,
50
+ rows: list[dict[str, Any]],
51
+ columns: list[str],
52
+ lineage: dict[str, dict[str, Any]],
53
+ step: CleaningStep,
54
+ ) -> tuple[list[dict[str, Any]], list[str], dict[str, dict[str, Any]]]:
55
+ """Apply one cleaning step and return ``(rows, columns, lineage)``.
56
+
57
+ ``step.operation`` is one of ``trim`` / ``empty_to_null`` / ``rename`` / ``cast``.
58
+ The inputs must not be mutated; return fresh structures. Row order is the input
59
+ order. ``columns`` is the output column order (only ``rename`` changes it, and it
60
+ renames in place, positionally). ``lineage`` keeps one entry per output column,
61
+ with the applied operation appended to that entry's ``operations`` list
62
+ (``rename:<from>-><to>`` and ``cast:<target>`` carry their argument).
63
+
64
+ Required typed refusals, all ``pipeline.BuildError``: ``COLUMN_MISSING`` when the
65
+ step names a column that is absent, ``CLEAN_TYPE`` when ``trim`` meets a non-string
66
+ non-null value, ``RENAME_COLLISION`` when a rename would produce duplicate column
67
+ names, and ``CAST_INVALID`` / ``CAST_TYPE`` from the cast. Casts must route through
68
+ ``pipeline._cast`` rather than an engine-native parser, so cast results and cast
69
+ refusals stay identical across engines. Reference implementation:
70
+ ``pipeline._apply_cleaning``.
71
+ """
72
+ ...
73
+
74
+ def join(
75
+ self,
76
+ plan: TablePlan,
77
+ rows_by_source: dict[str, list[dict[str, Any]]],
78
+ columns_by_source: dict[str, list[str]],
79
+ lineage_by_source: dict[str, dict[str, dict[str, Any]]],
80
+ ) -> tuple[
81
+ list[dict[str, Any]],
82
+ list[str],
83
+ dict[str, dict[str, Any]],
84
+ dict[str, Any],
85
+ ]:
86
+ """Run ``plan.join`` and return ``(rows, columns, lineage, report)``.
87
+
88
+ One left equality join on ``plan.join.on``. Output row order is left-source order,
89
+ positionally: iterate the left rows and look up the right row by key. Do not rely
90
+ on an engine join operator's ordering. Output columns are
91
+ ``[*left_columns, *right_non_key_columns]``; lineage merges the two sources under
92
+ those columns.
93
+
94
+ Required typed refusals, all ``pipeline.BuildError``: ``COLUMN_MISSING`` for an
95
+ absent key, ``JOIN_NULL_KEY`` for a null key value on either side,
96
+ ``JOIN_KEY_TYPE`` when the left key does not have exactly one observed non-null
97
+ value type, or when the left and right key value types are not exactly equal
98
+ (so two empty sources refuse, even though their type sets are both empty),
99
+ ``JOIN_COLUMN_COLLISION`` when a right non-key column name already exists on the
100
+ left, ``JOIN_CARDINALITY`` for non-unique right keys (or non-unique left keys under
101
+ ``one_to_one``), ``JOIN_UNMATCHED`` for any unmatched left row, and
102
+ ``JOIN_MULTIPLIER`` when the output/left row ratio is not exactly 1.
103
+
104
+ ``report`` is the ``join-evidence.v1`` payload that is serialized into candidate
105
+ evidence, so its counts must be plain Python ``int`` — never an engine scalar type.
106
+ Reference implementation: ``pipeline._join``.
107
+ """
108
+ ...
109
+
110
+ def select(
111
+ self,
112
+ plan: TablePlan,
113
+ rows: list[dict[str, Any]],
114
+ columns: list[str],
115
+ lineage: dict[str, dict[str, Any]],
116
+ ) -> tuple[list[dict[str, Any]], dict[str, dict[str, Any]]]:
117
+ """Project to ``plan.select`` and return ``(rows, lineage)``.
118
+
119
+ Every returned row carries exactly the columns of ``plan.select``, in that order;
120
+ ``lineage`` is restricted and reordered the same way. Row order is the input order.
121
+ Raise ``pipeline.BuildError("COLUMN_MISSING", ...)`` when ``plan.select`` names a
122
+ column that is not present. Reference implementation: ``pipeline._select``.
123
+ """
124
+ ...
@@ -0,0 +1,83 @@
1
+ """The reference backend: the pure-Python compute behind the protocol.
2
+
3
+ ``ReferenceBackend`` is a thin dispatch adapter that delegates ``clean`` / ``join`` /
4
+ ``select`` to ``pipeline._apply_cleaning`` / ``pipeline._join`` / ``pipeline._select``.
5
+ It must never re-implement that compute: delegation is what makes its bytes provably the
6
+ shared layer's own, which is why the sealed goldens reproduce byte-for-byte when the build
7
+ seam dispatches through it.
8
+
9
+ The compute functions live in ``pipeline``, and ``pipeline`` imports the registry, so
10
+ importing ``pipeline`` at module top would form a
11
+ ``pipeline -> registry -> reference -> pipeline`` cycle. Each method therefore imports
12
+ ``pipeline`` function-locally, so ``import backends`` never triggers ``import pipeline``
13
+ at module-load time.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from typing import Any
19
+
20
+ from mostlyright.data_harness.local_contracts import CleaningStep, TablePlan
21
+
22
+
23
+ class ReferenceBackend:
24
+ """Pure-Python reference engine delegating to the shared ``pipeline`` compute."""
25
+
26
+ # Must stay equal to ``pipeline.REFERENCE_BACKEND_NAME``, which is the single source of
27
+ # truth. Deliberately a plain literal here — not a lazy ``pipeline`` import — so
28
+ # constructing the backend (done at ``registry`` module load) never imports
29
+ # ``pipeline`` and cannot form the import cycle. Drift fails closed at build time: the
30
+ # seam resolves ``get_backend(pipeline.REFERENCE_BACKEND_NAME)``, which raises
31
+ # ``BACKEND_UNKNOWN`` if this name no longer matches.
32
+ name: str = "reference"
33
+
34
+ def versions(self) -> dict[str, str]:
35
+ # Pure Python — no external engine library to version.
36
+ return {}
37
+
38
+ def materialize_graph_rows(
39
+ self,
40
+ columns: tuple[str, ...],
41
+ rows: tuple[tuple[Any, ...], ...],
42
+ ) -> tuple[tuple[Any, ...], ...]:
43
+ if any(len(row) != len(columns) for row in rows):
44
+ raise ValueError("graph row width differs from its columns")
45
+ return tuple(tuple(row) for row in rows)
46
+
47
+ def clean(
48
+ self,
49
+ rows: list[dict[str, Any]],
50
+ columns: list[str],
51
+ lineage: dict[str, dict[str, Any]],
52
+ step: CleaningStep,
53
+ ) -> tuple[list[dict[str, Any]], list[str], dict[str, dict[str, Any]]]:
54
+ from mostlyright.data_harness import pipeline
55
+
56
+ return pipeline._apply_cleaning(rows, columns, lineage, step)
57
+
58
+ def join(
59
+ self,
60
+ plan: TablePlan,
61
+ rows_by_source: dict[str, list[dict[str, Any]]],
62
+ columns_by_source: dict[str, list[str]],
63
+ lineage_by_source: dict[str, dict[str, dict[str, Any]]],
64
+ ) -> tuple[
65
+ list[dict[str, Any]],
66
+ list[str],
67
+ dict[str, dict[str, Any]],
68
+ dict[str, Any],
69
+ ]:
70
+ from mostlyright.data_harness import pipeline
71
+
72
+ return pipeline._join(plan, rows_by_source, columns_by_source, lineage_by_source)
73
+
74
+ def select(
75
+ self,
76
+ plan: TablePlan,
77
+ rows: list[dict[str, Any]],
78
+ columns: list[str],
79
+ lineage: dict[str, dict[str, Any]],
80
+ ) -> tuple[list[dict[str, Any]], dict[str, dict[str, Any]]]:
81
+ from mostlyright.data_harness import pipeline
82
+
83
+ return pipeline._select(plan, rows, columns, lineage)
@@ -0,0 +1,55 @@
1
+ """Explicit named backend registration and fail-closed dispatch.
2
+
3
+ The registry is a literal dict populated at import time by explicit named registration.
4
+ No dynamic module import, no packaging entry-point discovery, no resource-based plugin
5
+ lookup, and no runtime import hook of any kind is permitted anywhere under ``backends/``:
6
+ an engine that can be swapped in at runtime is an engine that can change sealed candidate
7
+ bytes without a code review. ``tests/test_backend_registry.py`` statically asserts none of
8
+ those dynamic-import tokens appear in this package.
9
+
10
+ ``get_backend`` fails closed on an unknown name with a typed
11
+ ``BuildError("BACKEND_UNKNOWN", ...)`` — never a silent default engine, because silently
12
+ falling back would produce bytes under an engine name the caller did not select.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from mostlyright.data_harness.backends.pandas_backend import PandasBackend
18
+ from mostlyright.data_harness.backends.polars_backend import PolarsBackend
19
+ from mostlyright.data_harness.backends.protocol import Backend
20
+ from mostlyright.data_harness.backends.reference import ReferenceBackend
21
+
22
+ # Explicit, literal registration — the only way a backend enters the table. Constructing
23
+ # ``PandasBackend()``/``PolarsBackend()`` here does NOT import pandas/polars: each engine
24
+ # import is function-local, so ``import backends`` stays import-light (the lazy-import
25
+ # contract, mirrored from the reference). ``get_backend`` still fails closed on an unknown
26
+ # name (BACKEND_UNKNOWN).
27
+ _REGISTRY: dict[str, Backend] = {
28
+ ReferenceBackend.name: ReferenceBackend(),
29
+ PandasBackend.name: PandasBackend(),
30
+ PolarsBackend.name: PolarsBackend(),
31
+ }
32
+
33
+
34
+ def get_backend(name: str) -> Backend:
35
+ """Return the registered backend for ``name`` or fail closed.
36
+
37
+ An unknown/unregistered name is a typed refusal
38
+ (``BuildError("BACKEND_UNKNOWN", ...)``), never a silent default backend.
39
+ """
40
+
41
+ backend = _REGISTRY.get(name)
42
+ if backend is None:
43
+ # Lazy import keeps ``import backends`` free of a module-load dependency on
44
+ # ``pipeline``; ``pipeline`` imports this module, so a top-level import would
45
+ # cycle. Only this failure path needs ``BuildError``.
46
+ from mostlyright.data_harness.pipeline import BuildError
47
+
48
+ raise BuildError("BACKEND_UNKNOWN", f"no registered backend {name!r}")
49
+ return backend
50
+
51
+
52
+ def registered_backends() -> frozenset[str]:
53
+ """Return the frozenset of registered backend names."""
54
+
55
+ return frozenset(_REGISTRY)