mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,1007 @@
1
+ """Closed capability authority for the versioned table-plan graph.
2
+
3
+ The registry contains declarations only. Plan values cannot choose an import, plugin, callable,
4
+ or engine expression. Later phases attach the named executor hooks in reviewed code while this
5
+ module remains the single source for operation coordinates, arity, and parameter shape.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from collections.abc import Mapping
11
+ from dataclasses import dataclass
12
+ from types import MappingProxyType
13
+ from typing import Any
14
+
15
+ from mostlyright.data_harness.units import UnitError, resolve_declared_unit
16
+
17
+ GRAPH_OPERATION_VERSION = "1.0.0"
18
+ GRAPH_OPERATION_CONTRACT_VERSION = "local-operations.v2"
19
+ MAX_PARAMETER_MEMBERS = 4_096
20
+ MAX_PARAMETER_DEPTH = 16
21
+ # A fixed availability offset is a cleaning declaration, not an unbounded duration language. Ten
22
+ # calendar years covers publication-delay corrections without turning this one-row operation into
23
+ # a general temporal calculation surface. The bound is symmetric because a source can record when
24
+ # a value became available before a declared rate date as well as after it.
25
+ MAX_DATE_ADD_DAYS = 3_650
26
+
27
+ # The two availabilities an entry in this table can have. ``EXECUTABLE`` is an operation a plan
28
+ # may name. ``RETIRED`` is one that no longer meets the admission contract and that no newly
29
+ # authored plan may name, kept in the table because a Build sealed before it was retired, and a
30
+ # Recipe approved before it was retired, both re-parse their stored plan through this table --
31
+ # dropping the entry would make their own sealed bytes unreadable, which is a different and worse
32
+ # thing than retiring an operation. The refusal for new bytes lives at the authoring surfaces in
33
+ # ``local_contracts.refuse_retired_operations``, not in ``resolve_operation``, precisely because
34
+ # ``resolve_operation`` is on both paths and cannot tell them apart.
35
+ EXECUTABLE = "executable"
36
+ RETIRED = "retired"
37
+ _AVAILABILITIES = frozenset({EXECUTABLE, RETIRED})
38
+
39
+ # The join kinds a graph node may name, and the only place join-kind growth happens. The older
40
+ # plan version's ``JoinSpec`` is frozen at ``left`` alone, so this set and
41
+ # ``local_contracts.JOIN_KINDS`` are deliberately different sizes, and ``docs/TRANSFORMS.md``
42
+ # records the ruling that keeps them that way. Both sets are pinned by equality in one test,
43
+ # ``tests/test_local_contracts.py``, so adding a kind here reds the assertion that sits beside the
44
+ # frozen one -- which is the point: the decision about the older version is taken in the same
45
+ # place, rather than the two vocabularies drifting apart quietly.
46
+ GRAPH_JOIN_KINDS = frozenset({"left", "inner", "anti"})
47
+
48
+
49
+ class OperationRegistryError(ValueError):
50
+ """A stable feasibility refusal produced by the graph capability authority."""
51
+
52
+ def __init__(self, code: str, path: str, detail: str) -> None:
53
+ self.code = code
54
+ self.path = path
55
+ self.detail = detail
56
+ super().__init__(f"{path}: {detail} [{code}]")
57
+
58
+
59
+ @dataclass(frozen=True)
60
+ class AdmissionRow:
61
+ """One operation's row in the admission table in ``docs/TRANSFORMS.md``.
62
+
63
+ The table judges every operation against the scope test on that page, so each row is a
64
+ written finding rather than anything derivable: what the verdict was, one example that shows
65
+ the shape, and the reason in a few words. It is carried on the entry so that the page and the
66
+ entry cannot disagree -- ``scripts/generate_registry_tables.py`` emits the page's rows from
67
+ here -- and it is inert: nothing in resolution or parameter validation reads it.
68
+ """
69
+
70
+ verdict: str
71
+ example: str
72
+ rationale: str
73
+
74
+ def __post_init__(self) -> None:
75
+ # The value has to *be* its one line, not merely split into one. ``splitlines`` honours
76
+ # every separator that would break a table row, including the vertical tab and the
77
+ # paragraph separator, but it drops a trailing one -- so a cell ending in a newline
78
+ # splits into a single line and would push the row after it out of the table anyway.
79
+ for name in ("verdict", "example", "rationale"):
80
+ value = getattr(self, name)
81
+ lines = value.splitlines() if isinstance(value, str) else []
82
+ if (
83
+ not isinstance(value, str)
84
+ or not value.strip()
85
+ or "|" in value
86
+ or len(lines) != 1
87
+ or lines[0] != value
88
+ ):
89
+ raise OperationRegistryError(
90
+ "OPERATION_ADMISSION",
91
+ f"operation_registry.admission.{name}",
92
+ "must be one nonempty line holding no table separator",
93
+ )
94
+
95
+
96
+ @dataclass(frozen=True)
97
+ class GraphOperation:
98
+ name: str
99
+ version: str
100
+ input_arity: tuple[int, int]
101
+ required_parameters: frozenset[str]
102
+ optional_parameters: frozenset[str]
103
+ contract_parser_hook: str
104
+ executor_hook: str
105
+ admission: AdmissionRow
106
+ availability: str = EXECUTABLE
107
+
108
+ def __post_init__(self) -> None:
109
+ # RETIRED_OPERATIONS derives by equality against RETIRED, so a misspelling here would not
110
+ # fail anywhere -- it would quietly leave a retired operation admitted for new plans.
111
+ if self.availability not in _AVAILABILITIES:
112
+ raise OperationRegistryError(
113
+ "OPERATION_AVAILABILITY",
114
+ f"operation_registry.{self.name}.availability",
115
+ f"availability must be one of {sorted(_AVAILABILITIES)}",
116
+ )
117
+ # The written verdict and the field the software acts on are two statements of one
118
+ # decision, and a page saying "Admitted" beside a table refusing the operation is worse
119
+ # than either alone. Binding them here is what keeps the emitted row honest.
120
+ expected = "Retired" if self.availability == RETIRED else "Admitted"
121
+ if not self.admission.verdict.startswith(expected):
122
+ raise OperationRegistryError(
123
+ "OPERATION_ADMISSION",
124
+ f"operation_registry.{self.name}.admission.verdict",
125
+ f"an operation whose availability is {self.availability!r} reads {expected!r}",
126
+ )
127
+
128
+ def validate_parameters(self, value: Any, *, path: str) -> tuple[tuple[str, Any], ...]:
129
+ if not isinstance(value, Mapping) or any(not isinstance(key, str) for key in value):
130
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "must be an object")
131
+ keys = frozenset(value)
132
+ missing = sorted(self.required_parameters - keys)
133
+ unknown = sorted(keys - self.required_parameters - self.optional_parameters)
134
+ if missing or unknown:
135
+ raise OperationRegistryError(
136
+ "OPERATION_PARAMETERS",
137
+ path,
138
+ f"fields invalid; missing={missing}, unknown={unknown}",
139
+ )
140
+ _validate_semantics(self.name, value, path)
141
+ budget = [0]
142
+ return tuple(
143
+ (key, _freeze_json(value[key], f"{path}.{key}", depth=0, budget=budget))
144
+ for key in sorted(value)
145
+ )
146
+
147
+
148
+ def _operation(
149
+ name: str,
150
+ arity: tuple[int, int],
151
+ required: set[str],
152
+ optional: set[str] | None = None,
153
+ *,
154
+ availability: str = EXECUTABLE,
155
+ verdict: str,
156
+ example: str,
157
+ rationale: str,
158
+ ) -> GraphOperation:
159
+ return GraphOperation(
160
+ name=name,
161
+ version=GRAPH_OPERATION_VERSION,
162
+ input_arity=arity,
163
+ required_parameters=frozenset(required),
164
+ optional_parameters=frozenset(optional or set()),
165
+ contract_parser_hook=f"parse_{name}_parameters",
166
+ executor_hook=f"execute_{name}",
167
+ admission=AdmissionRow(verdict=verdict, example=example, rationale=rationale),
168
+ availability=availability,
169
+ )
170
+
171
+
172
+ def _choice(
173
+ value: Any, allowed: set[str], path: str, *, code: str = "OPERATION_PARAMETERS"
174
+ ) -> None:
175
+ if not isinstance(value, str) or value not in allowed:
176
+ raise OperationRegistryError(code, path, f"must be one of {sorted(allowed)}")
177
+
178
+
179
+ def _text(value: Any, path: str) -> str:
180
+ if (
181
+ not isinstance(value, str)
182
+ or not value
183
+ or len(value) > 128
184
+ or any(ord(char) < 32 for char in value)
185
+ ):
186
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "must be bounded text")
187
+ return value
188
+
189
+
190
+ def _text_list(value: Any, path: str, *, empty: bool = False) -> list[str]:
191
+ if (
192
+ not isinstance(value, list)
193
+ or (not empty and not value)
194
+ or len(value) > 256
195
+ or any(not isinstance(item, str) for item in value)
196
+ ):
197
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "must be a bounded text list")
198
+ for index, item in enumerate(value):
199
+ _text(item, f"{path}[{index}]")
200
+ if len(value) != len(set(value)):
201
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "values must be unique")
202
+ return value
203
+
204
+
205
+ def _positive_int(value: Any, path: str, *, maximum: int) -> int:
206
+ if isinstance(value, bool) or not isinstance(value, int) or not 1 <= value <= maximum:
207
+ raise OperationRegistryError(
208
+ "OPERATION_PARAMETERS", path, f"must be an integer from 1 to {maximum}"
209
+ )
210
+ return value
211
+
212
+
213
+ def _date_add_days_offset(value: Any, path: str) -> int:
214
+ """Accept the one signed, fixed calendar offset this operation may declare."""
215
+
216
+ if isinstance(value, bool) or not isinstance(value, int):
217
+ raise OperationRegistryError(
218
+ "DATE_ADD_DAYS_TYPE", path, "days must be a signed int64 calendar-day offset"
219
+ )
220
+ if not -MAX_DATE_ADD_DAYS <= value <= MAX_DATE_ADD_DAYS:
221
+ raise OperationRegistryError(
222
+ "DATE_ADD_DAYS_RANGE",
223
+ path,
224
+ f"days must be from {-MAX_DATE_ADD_DAYS} to {MAX_DATE_ADD_DAYS}",
225
+ )
226
+ return value
227
+
228
+
229
+ def _exact(value: Mapping[str, Any], expected: set[str], path: str) -> None:
230
+ if set(value) != expected:
231
+ raise OperationRegistryError(
232
+ "OPERATION_PARAMETERS", path, f"fields must be exactly {sorted(expected)}"
233
+ )
234
+
235
+
236
+ def _operand(value: Any, path: str, *, membership: bool = False) -> None:
237
+ if not isinstance(value, Mapping):
238
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "operand must be an object")
239
+ if set(value) == {"column"}:
240
+ _text(value["column"], f"{path}.column")
241
+ return
242
+ if set(value) != {"literal"}:
243
+ raise OperationRegistryError(
244
+ "OPERATION_PARAMETERS", path, "operand must name one column or literal"
245
+ )
246
+ literal = value["literal"]
247
+ if membership:
248
+ if not isinstance(literal, list) or not 1 <= len(literal) <= 256:
249
+ raise OperationRegistryError(
250
+ "OPERATION_PARAMETERS", f"{path}.literal", "membership literal must be bounded list"
251
+ )
252
+ if any(type(item) not in {bool, int, str} and item is not None for item in literal):
253
+ raise OperationRegistryError(
254
+ "OPERATION_PARAMETERS", f"{path}.literal", "membership values are not scalar"
255
+ )
256
+ elif type(literal) not in {bool, int, str} and literal is not None:
257
+ raise OperationRegistryError(
258
+ "OPERATION_PARAMETERS", f"{path}.literal", "literal must be a scalar"
259
+ )
260
+
261
+
262
+ def _predicate(value: Any, path: str, *, depth: int = 0) -> None:
263
+ if depth > 16 or not isinstance(value, Mapping):
264
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "predicate is invalid")
265
+ op = value.get("op")
266
+ if op in {"and", "or"}:
267
+ _exact(value, {"op", "args"}, path)
268
+ args = value["args"]
269
+ if not isinstance(args, list) or not 2 <= len(args) <= 64:
270
+ raise OperationRegistryError(
271
+ "OPERATION_PARAMETERS", f"{path}.args", "boolean op needs 2..64 predicates"
272
+ )
273
+ for index, item in enumerate(args):
274
+ _predicate(item, f"{path}.args[{index}]", depth=depth + 1)
275
+ return
276
+ if op == "not":
277
+ _exact(value, {"op", "arg"}, path)
278
+ _predicate(value["arg"], f"{path}.arg", depth=depth + 1)
279
+ return
280
+ if op in {"is_null", "not_null"}:
281
+ _exact(value, {"op", "left"}, path)
282
+ _operand(value["left"], f"{path}.left")
283
+ return
284
+ if op in {"eq", "ne", "lt", "le", "gt", "ge", "in"}:
285
+ _exact(value, {"op", "left", "right"}, path)
286
+ _operand(value["left"], f"{path}.left")
287
+ _operand(value["right"], f"{path}.right", membership=op == "in")
288
+ return
289
+ raise OperationRegistryError(
290
+ "OPERATION_PARAMETERS", f"{path}.op", "predicate operation is unsupported"
291
+ )
292
+
293
+
294
+ def _derive_expression(value: Any, path: str, *, depth: int = 0) -> None:
295
+ if depth > 16 or not isinstance(value, Mapping):
296
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "derive expression is invalid")
297
+ if set(value) in ({"column"}, {"literal"}):
298
+ _operand(value, path)
299
+ return
300
+ op = value.get("op")
301
+ if op == "if":
302
+ _exact(value, {"op", "condition", "then", "else"}, path)
303
+ _predicate(value["condition"], f"{path}.condition", depth=depth + 1)
304
+ _derive_expression(value["then"], f"{path}.then", depth=depth + 1)
305
+ _derive_expression(value["else"], f"{path}.else", depth=depth + 1)
306
+ return
307
+ _exact(value, {"op", "args"}, path)
308
+ args = value["args"]
309
+ arity = DERIVE_EXPRESSION_ARITY
310
+ if op == "concat":
311
+ valid = isinstance(args, list) and 1 <= len(args) <= 64
312
+ else:
313
+ valid = op in arity and isinstance(args, list) and len(args) == arity[op]
314
+ if not valid:
315
+ raise OperationRegistryError(
316
+ "OPERATION_PARAMETERS", f"{path}.args", "derive operation or arity is unsupported"
317
+ )
318
+ for index, item in enumerate(args):
319
+ _derive_expression(item, f"{path}.args[{index}]", depth=depth + 1)
320
+
321
+
322
+ # The two denominator policies a dividing ``derive`` node declares, in the style of the node's
323
+ # ``overflow_policy`` and ``null_policy``: one admitted spelling each, named in the plan so the
324
+ # behaviour is read from the recipe rather than inferred from the kernel.
325
+ #: Every operator a derive expression can name, and how many arguments each takes. It is stated
326
+ #: once, here, because two things read it: the shape check below, and the unit analysis in
327
+ #: :mod:`unit_flow`, which has to say what each operator does to a declared unit. An operator added
328
+ #: to one and not the other is what let a conditional go unexamined by both unit gates.
329
+ DERIVE_EXPRESSION_ARITY: Mapping[str, int] = MappingProxyType(
330
+ {
331
+ "add": 2,
332
+ "subtract": 2,
333
+ "multiply": 2,
334
+ # The numerator comes first and the denominator second; the order is contract-significant
335
+ # because division is the one arithmetic operator here that is not commutative.
336
+ "divide": 2,
337
+ "year": 1,
338
+ "month": 1,
339
+ "day": 1,
340
+ "hour": 1,
341
+ # The two lexical normalizers. Both are bounded rewrites of one string operand: neither
342
+ # parses, matches a pattern, or accepts one from a plan.
343
+ "trim": 1,
344
+ "case_fold": 1,
345
+ }
346
+ )
347
+ #: The two the arity table cannot hold: a conditional's parts are named rather than counted, and a
348
+ #: concatenation takes one to sixty-four.
349
+ DERIVE_EXPRESSION_VARIADIC: frozenset[str] = frozenset({"if", "concat"})
350
+
351
+ DIVIDE_POLICIES: tuple[str, ...] = ("zero_denominator_policy", "null_denominator_policy")
352
+ _DIVIDE_POLICY_CHOICES = {
353
+ "zero_denominator_policy": {"reject"},
354
+ "null_denominator_policy": {"propagate"},
355
+ }
356
+
357
+
358
+ def _divides(value: Any) -> bool:
359
+ """Report whether one derive expression tree contains a division anywhere inside it."""
360
+
361
+ if isinstance(value, Mapping):
362
+ return value.get("op") == "divide" or any(_divides(item) for item in value.values())
363
+ if isinstance(value, list):
364
+ return any(_divides(item) for item in value)
365
+ return False
366
+
367
+
368
+ def _divide_policies(value: Mapping[str, Any], expressions: Mapping[str, Any], path: str) -> None:
369
+ """Bind the denominator policies to the presence of a division, in both directions.
370
+
371
+ A node that divides must declare both policies, and a node that does not divide must declare
372
+ neither. Making the pair exactly conditional keeps the parameter set closed: a plan cannot
373
+ carry a policy that governs nothing, and a division cannot run under a policy nobody wrote.
374
+ """
375
+
376
+ dividing = any(
377
+ _divides(specification.get("expression"))
378
+ for specification in expressions.values()
379
+ if isinstance(specification, Mapping)
380
+ )
381
+ declared = [name for name in DIVIDE_POLICIES if name in value]
382
+ if not dividing:
383
+ if declared:
384
+ raise OperationRegistryError(
385
+ "OPERATION_PARAMETERS",
386
+ f"{path}.{declared[0]}",
387
+ "only a derive node containing a divide expression declares denominator policies",
388
+ )
389
+ return
390
+ missing = [name for name in DIVIDE_POLICIES if name not in value]
391
+ if missing:
392
+ raise OperationRegistryError(
393
+ "OPERATION_PARAMETERS",
394
+ f"{path}.{missing[0]}",
395
+ "a divide expression requires a declared denominator policy",
396
+ )
397
+ for name in DIVIDE_POLICIES:
398
+ _choice(value[name], _DIVIDE_POLICY_CHOICES[name], f"{path}.{name}")
399
+
400
+
401
+ def _unit_code(value: Any, path: str) -> None:
402
+ """Admit one unit code the grammar resolves, or refuse it by the construct that failed."""
403
+
404
+ if not isinstance(value, str):
405
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "a unit code must be text")
406
+ try:
407
+ resolve_declared_unit(value)
408
+ except UnitError as exc:
409
+ raise OperationRegistryError(
410
+ "UNIT_CODE", path, f"{value!r} is not a unit code: {exc.detail}"
411
+ ) from None
412
+
413
+
414
+ def _column_units(value: Any, path: str) -> None:
415
+ """Admit a bounded column-to-unit declaration, or nothing at all.
416
+
417
+ Declaring nothing is the ordinary case and stays legal forever: a column with no declared unit
418
+ is untouched by the gates that read this, so adding the parameter cannot refuse a plan that
419
+ was already correct.
420
+ """
421
+
422
+ if value is None:
423
+ return
424
+ if not isinstance(value, Mapping) or not value:
425
+ raise OperationRegistryError(
426
+ "OPERATION_PARAMETERS", path, "must be a non-empty object or absent"
427
+ )
428
+ if len(value) > 256:
429
+ raise OperationRegistryError(
430
+ "OPERATION_PARAMETERS", path, "must declare at most 256 columns"
431
+ )
432
+ for column, code in value.items():
433
+ if not isinstance(column, str):
434
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "column names must be text")
435
+ _text(column, path)
436
+ _unit_code(code, f"{path}.{column}")
437
+
438
+
439
+ def _sort_keys(value: Any, path: str) -> None:
440
+ if not isinstance(value, list) or not 1 <= len(value) <= 256:
441
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "sort keys must be bounded list")
442
+ for index, item in enumerate(value):
443
+ item_path = f"{path}[{index}]"
444
+ if isinstance(item, str):
445
+ _text(item, item_path)
446
+ continue
447
+ if not isinstance(item, Mapping):
448
+ raise OperationRegistryError("OPERATION_PARAMETERS", item_path, "sort key is invalid")
449
+ _exact(item, {"column", "direction", "nulls"}, item_path)
450
+ _text(item["column"], f"{item_path}.column")
451
+ _choice(item["direction"], {"asc", "desc"}, f"{item_path}.direction")
452
+ _choice(item["nulls"], {"first", "last"}, f"{item_path}.nulls")
453
+
454
+
455
+ def _measures(value: Any, path: str) -> None:
456
+ if not isinstance(value, list) or not 1 <= len(value) <= 256:
457
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "measures must be bounded list")
458
+ names: list[str] = []
459
+ for index, item in enumerate(value):
460
+ item_path = f"{path}[{index}]"
461
+ if not isinstance(item, Mapping):
462
+ raise OperationRegistryError("OPERATION_PARAMETERS", item_path, "measure is invalid")
463
+ _exact(item, {"column", "op", "as"}, item_path)
464
+ _text(item["column"], f"{item_path}.column")
465
+ _choice(
466
+ item["op"], {"count", "sum", "min", "max", "mean", "first", "last"}, f"{item_path}.op"
467
+ )
468
+ names.append(_text(item["as"], f"{item_path}.as"))
469
+ if len(names) != len(set(names)):
470
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "measure outputs must be unique")
471
+
472
+
473
+ def _validate_semantics(name: str, value: Mapping[str, Any], path: str) -> None:
474
+ """Admit only policy variants the deterministic v2 kernel actually implements."""
475
+
476
+ if name == "union":
477
+ _choice(
478
+ value["column_matching"], {"exact"}, f"{path}.column_matching", code="UNION_UNSUPPORTED"
479
+ )
480
+ safe_casts = value["safe_casts"]
481
+ if not isinstance(safe_casts, Mapping) or any(
482
+ not isinstance(column, str) for column in safe_casts
483
+ ):
484
+ raise OperationRegistryError(
485
+ "OPERATION_PARAMETERS", f"{path}.safe_casts", "must be an object"
486
+ )
487
+ for column, target in safe_casts.items():
488
+ _text(column, f"{path}.safe_casts")
489
+ _choice(target, {"decimal"}, f"{path}.safe_casts.{column}")
490
+ _choice(
491
+ value["grain_policy"], {"preserve"}, f"{path}.grain_policy", code="UNION_UNSUPPORTED"
492
+ )
493
+ elif name == "source":
494
+ _text(value["source"], f"{path}.source")
495
+ _column_units(value.get("column_units"), f"{path}.column_units")
496
+ elif name == "filter":
497
+ _choice(value["null_policy"], {"drop"}, f"{path}.null_policy")
498
+ _predicate(value["predicate"], f"{path}.predicate")
499
+ elif name == "project":
500
+ _text_list(value["columns"], f"{path}.columns")
501
+ elif name == "rename":
502
+ mappings = value["mappings"]
503
+ if not isinstance(mappings, Mapping) or not mappings:
504
+ raise OperationRegistryError(
505
+ "OPERATION_PARAMETERS", f"{path}.mappings", "must be a non-empty object"
506
+ )
507
+ targets = []
508
+ for source, target in mappings.items():
509
+ _text(source, f"{path}.mappings")
510
+ targets.append(_text(target, f"{path}.mappings.{source}"))
511
+ if len(targets) != len(set(targets)):
512
+ raise OperationRegistryError(
513
+ "OPERATION_PARAMETERS", f"{path}.mappings", "targets must be unique"
514
+ )
515
+ elif name == "cast":
516
+ _choice(value["overflow_policy"], {"reject"}, f"{path}.overflow_policy")
517
+ columns = value["columns"]
518
+ if not isinstance(columns, Mapping) or not columns:
519
+ raise OperationRegistryError(
520
+ "OPERATION_PARAMETERS", f"{path}.columns", "must be a non-empty object"
521
+ )
522
+ for column, kind in columns.items():
523
+ if not isinstance(column, str):
524
+ raise OperationRegistryError(
525
+ "OPERATION_PARAMETERS", f"{path}.columns", "column names must be text"
526
+ )
527
+ _choice(
528
+ kind,
529
+ {
530
+ "boolean",
531
+ "int64",
532
+ "float64",
533
+ "string",
534
+ "date",
535
+ "timestamp_utc",
536
+ "decimal",
537
+ },
538
+ f"{path}.columns.{column}",
539
+ )
540
+ elif name == "deduplicate":
541
+ _text_list(value["keys"], f"{path}.keys")
542
+ _sort_keys(value["authority"], f"{path}.authority")
543
+ _choice(value["null_key_policy"], {"reject"}, f"{path}.null_key_policy")
544
+ _choice(value["tie_policy"], {"stable_first"}, f"{path}.tie_policy")
545
+ elif name == "sort":
546
+ _sort_keys(value["keys"], f"{path}.keys")
547
+ elif name == "derive":
548
+ _choice(value["overflow_policy"], {"reject"}, f"{path}.overflow_policy")
549
+ _choice(value["null_policy"], {"propagate"}, f"{path}.null_policy")
550
+ expressions = value["expressions"]
551
+ if not isinstance(expressions, Mapping) or not expressions:
552
+ raise OperationRegistryError(
553
+ "OPERATION_PARAMETERS", f"{path}.expressions", "must be a non-empty object"
554
+ )
555
+ for output, specification in expressions.items():
556
+ _text(output, f"{path}.expressions")
557
+ if not isinstance(specification, Mapping):
558
+ raise OperationRegistryError(
559
+ "OPERATION_PARAMETERS", f"{path}.expressions.{output}", "must be an object"
560
+ )
561
+ fields = set(specification)
562
+ if (
563
+ not {"output_type", "expression"}
564
+ <= fields
565
+ <= {
566
+ "output_type",
567
+ "expression",
568
+ "output_unit",
569
+ }
570
+ ):
571
+ raise OperationRegistryError(
572
+ "OPERATION_PARAMETERS",
573
+ f"{path}.expressions.{output}",
574
+ "fields must be output_type and expression, and may add output_unit",
575
+ )
576
+ if "output_unit" in specification:
577
+ # A unit is a claim about a quantity, so the column has to hold one. Text and dates
578
+ # do not, and a unit on one of them would leave the merge gate downstream
579
+ # arbitrating units on strings.
580
+ if specification["output_type"] not in {"int64", "float64", "decimal"}:
581
+ raise OperationRegistryError(
582
+ "OPERATION_PARAMETERS",
583
+ f"{path}.expressions.{output}.output_unit",
584
+ "only an int64, float64 or decimal column carries a unit",
585
+ )
586
+ _unit_code(specification["output_unit"], f"{path}.expressions.{output}.output_unit")
587
+ _choice(
588
+ specification["output_type"],
589
+ {
590
+ "boolean",
591
+ "int64",
592
+ "float64",
593
+ "string",
594
+ "date",
595
+ "timestamp_utc",
596
+ "decimal",
597
+ },
598
+ f"{path}.expressions.{output}.output_type",
599
+ )
600
+ _derive_expression(
601
+ specification["expression"], f"{path}.expressions.{output}.expression"
602
+ )
603
+ _divide_policies(value, expressions, path)
604
+ elif name == "unpivot":
605
+ # Wide to long. The columns that carry values are named in the plan and are never read
606
+ # out of the data, so the output schema is a function of the recipe alone: the columns
607
+ # the plan does not name stay as they are, and exactly two new columns are appended.
608
+ # ``value_type`` is declared for the same reason ``derive`` declares ``output_type``.
609
+ # Inferring it from the wide columns would have made the long column's type depend on the
610
+ # data: a refresh in which every declared column happened to be empty would type it
611
+ # differently from the last one, off the same recipe.
612
+ value_columns = _text_list(value["value_columns"], f"{path}.value_columns")
613
+ name_column = _text(value["name_column"], f"{path}.name_column")
614
+ value_column = _text(value["value_column"], f"{path}.value_column")
615
+ _choice(
616
+ value["value_type"],
617
+ {
618
+ "boolean",
619
+ "int64",
620
+ "float64",
621
+ "string",
622
+ "date",
623
+ "timestamp_utc",
624
+ "decimal",
625
+ },
626
+ f"{path}.value_type",
627
+ )
628
+ _choice(value["null_policy"], {"retain"}, f"{path}.null_policy")
629
+ if name_column == value_column:
630
+ raise OperationRegistryError(
631
+ "OPERATION_PARAMETERS",
632
+ f"{path}.value_column",
633
+ "must differ from name_column",
634
+ )
635
+ reused = sorted({name_column, value_column} & set(value_columns))
636
+ if reused:
637
+ raise OperationRegistryError(
638
+ "OPERATION_PARAMETERS",
639
+ f"{path}.value_columns",
640
+ f"output columns must be new names: {reused}",
641
+ )
642
+ elif name == "aggregate":
643
+ _text_list(value["group_by"], f"{path}.group_by", empty=True)
644
+ _measures(value["measures"], f"{path}.measures")
645
+ _choice(value["empty_group_policy"], {"omit"}, f"{path}.empty_group_policy")
646
+ elif name == "resample":
647
+ fixed = {
648
+ "origin": {"unix_epoch"},
649
+ "timezone": {"UTC"},
650
+ "ambiguity_policy": {"reject"},
651
+ "nonexistent_policy": {"reject"},
652
+ "closure": {"left"},
653
+ "label": {"left"},
654
+ "missing_period_policy": {"omit"},
655
+ }
656
+ for field, allowed in fixed.items():
657
+ _choice(value[field], allowed, f"{path}.{field}", code="RESAMPLE_UNSUPPORTED")
658
+ _text_list(value["group_by"], f"{path}.group_by", empty=True)
659
+ _text(value["event_time"], f"{path}.event_time")
660
+ _measures(value["aggregates"], f"{path}.aggregates")
661
+ if (
662
+ isinstance(value["duration_seconds"], bool)
663
+ or not isinstance(value["duration_seconds"], int)
664
+ or value["duration_seconds"] <= 0
665
+ ):
666
+ raise OperationRegistryError(
667
+ "OPERATION_PARAMETERS", f"{path}.duration_seconds", "must be a positive integer"
668
+ )
669
+ elif name == "prediction_label":
670
+ _text_list(value["partition_by"], f"{path}.partition_by", empty=True)
671
+ for field in ("time_column", "source_column", "output_column"):
672
+ _text(value[field], f"{path}.{field}")
673
+ if value["time_column"] == value["source_column"] or {
674
+ value["time_column"],
675
+ value["source_column"],
676
+ } & set(value["partition_by"]):
677
+ raise OperationRegistryError(
678
+ "OPERATION_PARAMETERS",
679
+ f"{path}.partition_by",
680
+ "partition, time, and source-value columns must be distinct",
681
+ )
682
+ if value["output_column"] in {
683
+ value["time_column"],
684
+ value["source_column"],
685
+ *value["partition_by"],
686
+ }:
687
+ raise OperationRegistryError(
688
+ "OPERATION_PARAMETERS",
689
+ f"{path}.output_column",
690
+ "must be a new column distinct from the keys and source value",
691
+ )
692
+ _positive_int(value["horizon_days"], f"{path}.horizon_days", maximum=3_650)
693
+ _choice(value["null_key_policy"], {"reject"}, f"{path}.null_key_policy")
694
+ _choice(value["duplicate_key_policy"], {"reject"}, f"{path}.duplicate_key_policy")
695
+ _choice(value["missing_target_policy"], {"drop"}, f"{path}.missing_target_policy")
696
+ _choice(value["null_target_policy"], {"drop"}, f"{path}.null_target_policy")
697
+ _choice(value["boundary_policy"], {"drop"}, f"{path}.boundary_policy")
698
+ elif name == "date_add_days":
699
+ date_column = _text(value["date_column"], f"{path}.date_column")
700
+ output_column = _text(value["output_column"], f"{path}.output_column")
701
+ if date_column == output_column:
702
+ raise OperationRegistryError(
703
+ "DATE_ADD_DAYS_COLUMN",
704
+ f"{path}.output_column",
705
+ "must be a new column distinct from date_column",
706
+ )
707
+ _date_add_days_offset(value["days"], f"{path}.days")
708
+ _choice(value["null_policy"], {"propagate"}, f"{path}.null_policy")
709
+ _choice(value["overflow_policy"], {"reject"}, f"{path}.overflow_policy")
710
+ elif name == "join":
711
+ _choice(value["kind"], set(GRAPH_JOIN_KINDS), f"{path}.kind")
712
+ _choice(value["cardinality"], {"one_to_one", "many_to_one"}, f"{path}.cardinality")
713
+ _choice(value["null_key_policy"], {"reject"}, f"{path}.null_key_policy")
714
+ _choice(
715
+ value["unmatched_policy"], {"preserve", "reject", "drop"}, f"{path}.unmatched_policy"
716
+ )
717
+ # An anti join is the unmatched left rows and nothing else, so ``preserve`` is the only
718
+ # unmatched policy that describes it: dropping the unmatched rows would leave no output,
719
+ # and rejecting them would refuse every anti join that found anything.
720
+ if (value["kind"], value["unmatched_policy"]) not in {
721
+ ("left", "preserve"),
722
+ ("left", "reject"),
723
+ ("inner", "drop"),
724
+ ("inner", "reject"),
725
+ ("anti", "preserve"),
726
+ }:
727
+ raise OperationRegistryError(
728
+ "OPERATION_PARAMETERS",
729
+ f"{path}.unmatched_policy",
730
+ "unmatched policy is incompatible with join kind",
731
+ )
732
+ _text_list(value["on"], f"{path}.on")
733
+
734
+
735
+ _OPERATIONS = {
736
+ # ``column_units`` is optional so a plan sealed before units were declarable keeps validating
737
+ # unchanged, in the same style as the conditional denominator policies below. A source that
738
+ # declares nothing is untouched by both unit gates; a source that declares a column's unit is
739
+ # what gives the gates something to check.
740
+ "source": _operation(
741
+ "source",
742
+ (0, 0),
743
+ {"source"},
744
+ {"column_units"},
745
+ verdict="Admitted",
746
+ example="load one snapshotted delimited file",
747
+ rationale="leaf, no choice",
748
+ ),
749
+ "union": _operation(
750
+ "union",
751
+ (2, 64),
752
+ {"column_matching", "safe_casts", "grain_policy"},
753
+ verdict="Admitted",
754
+ example="stack twelve monthly files",
755
+ rationale="same facts, one table",
756
+ ),
757
+ "filter": _operation(
758
+ "filter",
759
+ (1, 1),
760
+ {"predicate", "null_policy"},
761
+ verdict="Admitted",
762
+ example="drop rows before 2010",
763
+ rationale="selects, invents nothing",
764
+ ),
765
+ "project": _operation(
766
+ "project",
767
+ (1, 1),
768
+ {"columns"},
769
+ verdict="Admitted",
770
+ example="drop forty unused columns",
771
+ rationale="selection",
772
+ ),
773
+ "rename": _operation(
774
+ "rename",
775
+ (1, 1),
776
+ {"mappings"},
777
+ verdict="Admitted",
778
+ example="one supplier's column name becomes the declared one",
779
+ rationale="label only",
780
+ ),
781
+ "cast": _operation(
782
+ "cast",
783
+ (1, 1),
784
+ {"columns", "overflow_policy"},
785
+ verdict="Admitted",
786
+ example="text digits become a whole number",
787
+ rationale="type fix",
788
+ ),
789
+ # The two denominator policies are optional in the parameter table and mandatory in the
790
+ # semantics: ``_divide_policies`` requires both exactly when an expression divides and refuses
791
+ # both when none does. Keeping them out of the required set is what lets a ``derive`` node
792
+ # sealed before division existed keep validating unchanged.
793
+ "derive": _operation(
794
+ "derive",
795
+ (1, 1),
796
+ {"expressions", "overflow_policy", "null_policy"},
797
+ set(DIVIDE_POLICIES),
798
+ verdict="Admitted, with a limit",
799
+ example="the year of a timestamp",
800
+ rationale="see the limit below",
801
+ ),
802
+ "deduplicate": _operation(
803
+ "deduplicate",
804
+ (1, 1),
805
+ {"keys", "authority", "null_key_policy", "tie_policy"},
806
+ verdict="Admitted",
807
+ example="two feeds report one reading, keep the declared authority",
808
+ rationale="resolves a conflict that was declared, not noticed",
809
+ ),
810
+ "sort": _operation(
811
+ "sort",
812
+ (1, 1),
813
+ {"keys"},
814
+ verdict="Admitted",
815
+ example="order by facility, then date",
816
+ rationale="presentation",
817
+ ),
818
+ "aggregate": _operation(
819
+ "aggregate",
820
+ (1, 1),
821
+ {"group_by", "measures", "empty_group_policy"},
822
+ verdict="Admitted",
823
+ example="daily total per facility",
824
+ rationale="declares grain",
825
+ ),
826
+ "resample": _operation(
827
+ "resample",
828
+ (1, 1),
829
+ {
830
+ "group_by",
831
+ "event_time",
832
+ "duration_seconds",
833
+ "origin",
834
+ "timezone",
835
+ "ambiguity_policy",
836
+ "nonexistent_policy",
837
+ "closure",
838
+ "label",
839
+ "aggregates",
840
+ "missing_period_policy",
841
+ },
842
+ verdict="Admitted",
843
+ example="hourly readings become daily means",
844
+ rationale="grain, in time",
845
+ ),
846
+ # Retired. It fails the admission contract in ``docs/TRANSFORMS.md``: ``horizon_days`` is a
847
+ # claim about what somebody intends to predict, which is the modeling side of the line. It
848
+ # stays in the registry with ``availability="retired"`` so a Build already sealed with it, and
849
+ # a Recipe already approved with it, keep resolving -- both re-parse their stored plan through
850
+ # this table to verify. Newly authored bytes are refused; see
851
+ # ``local_contracts.refuse_retired_operations``.
852
+ "prediction_label": _operation(
853
+ "prediction_label",
854
+ (1, 1),
855
+ {
856
+ "partition_by",
857
+ "time_column",
858
+ "source_column",
859
+ "output_column",
860
+ "horizon_days",
861
+ "null_key_policy",
862
+ "duplicate_key_policy",
863
+ "missing_target_policy",
864
+ "null_target_policy",
865
+ "boundary_policy",
866
+ },
867
+ availability=RETIRED,
868
+ verdict="Retired",
869
+ example="an outcome thirty days ahead",
870
+ rationale="the horizon is the analyst's modeling choice",
871
+ ),
872
+ "join": _operation(
873
+ "join",
874
+ (2, 2),
875
+ {"kind", "cardinality", "on", "null_key_policy", "unmatched_policy"},
876
+ verdict="Admitted",
877
+ example="readings joined to facility metadata",
878
+ rationale="combines, invents nothing",
879
+ ),
880
+ # New vocabulary is appended rather than inserted: the order below is contract-significant,
881
+ # so an addition extends the tuple and never rewrites the part of it that already shipped.
882
+ "unpivot": _operation(
883
+ "unpivot",
884
+ (1, 1),
885
+ {"value_columns", "name_column", "value_column", "value_type", "null_policy"},
886
+ verdict="Admitted",
887
+ example="one column per year becomes a year column and a value column",
888
+ rationale="reshape only",
889
+ ),
890
+ # This is a row-local date correction, not a lag, lead, target, or window: each output date
891
+ # comes from exactly the date on its own input row plus one fixed offset declared in the plan.
892
+ # It is appended because graph-vocabulary order is contract-significant.
893
+ "date_add_days": _operation(
894
+ "date_add_days",
895
+ (1, 1),
896
+ {"date_column", "output_column", "days", "null_policy", "overflow_policy"},
897
+ verdict="Admitted",
898
+ example="a rate date becomes its next calendar availability date",
899
+ rationale="fixed row-local calendar correction",
900
+ ),
901
+ }
902
+
903
+ # Tuple order is contract-significant and deliberately follows the graph vocabulary.
904
+ OPERATION_REGISTRY: Mapping[str, GraphOperation] = MappingProxyType(_OPERATIONS)
905
+
906
+ # The retired names, read off the table rather than restated beside it.
907
+ RETIRED_OPERATIONS: frozenset[str] = frozenset(
908
+ name for name, operation in _OPERATIONS.items() if operation.availability == RETIRED
909
+ )
910
+
911
+
912
+ def resolve_operation(name: Any, version: Any) -> GraphOperation:
913
+ if not isinstance(name, str) or not isinstance(version, str):
914
+ raise OperationRegistryError(
915
+ "OPERATION_COORDINATE", "plan.nodes.operation", "name and version must be strings"
916
+ )
917
+ if name in {"window_target", "lag", "lead", "rolling"}:
918
+ raise OperationRegistryError(
919
+ "WINDOW_TARGET_UNSUPPORTED",
920
+ "plan.nodes.operation",
921
+ "generic window, lag, lead, and rolling operations are outside local-operations.v2",
922
+ )
923
+ operation = OPERATION_REGISTRY.get(name)
924
+ if operation is None:
925
+ raise OperationRegistryError(
926
+ "OPERATION_UNSUPPORTED", "plan.nodes.operation", f"operation {name!r} is not registered"
927
+ )
928
+ if version != operation.version:
929
+ raise OperationRegistryError(
930
+ "OPERATION_VERSION_UNSUPPORTED",
931
+ "plan.nodes.operation_version",
932
+ f"operation {name!r} requires version {operation.version!r}",
933
+ )
934
+ return operation
935
+
936
+
937
+ def _freeze_json(value: Any, path: str, *, depth: int, budget: list[int]) -> Any:
938
+ if depth > MAX_PARAMETER_DEPTH:
939
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "nesting is too deep")
940
+ budget[0] += 1
941
+ if budget[0] > MAX_PARAMETER_MEMBERS:
942
+ raise OperationRegistryError("OPERATION_PARAMETERS", path, "contains too many members")
943
+ if value is None or type(value) in {bool, int, str}:
944
+ return value
945
+ if type(value) is float:
946
+ raise OperationRegistryError(
947
+ "OPERATION_PARAMETERS", path, "floating JSON numbers are not contract values"
948
+ )
949
+ if isinstance(value, list):
950
+ return (
951
+ "$array",
952
+ tuple(
953
+ _freeze_json(item, f"{path}[{index}]", depth=depth + 1, budget=budget)
954
+ for index, item in enumerate(value)
955
+ ),
956
+ )
957
+ if isinstance(value, Mapping) and all(isinstance(key, str) for key in value):
958
+ return (
959
+ "$object",
960
+ tuple(
961
+ (
962
+ key,
963
+ _freeze_json(value[key], f"{path}.{key}", depth=depth + 1, budget=budget),
964
+ )
965
+ for key in sorted(value)
966
+ ),
967
+ )
968
+ raise OperationRegistryError(
969
+ "OPERATION_PARAMETERS", path, "must contain only closed JSON-compatible values"
970
+ )
971
+
972
+
973
+ def thaw_parameter(value: Any) -> Any:
974
+ """Return the canonical JSON representation of one frozen parameter value."""
975
+
976
+ if isinstance(value, tuple):
977
+ if len(value) == 2 and value[0] == "$array" and isinstance(value[1], tuple):
978
+ return [thaw_parameter(item) for item in value[1]]
979
+ if len(value) == 2 and value[0] == "$object" and isinstance(value[1], tuple):
980
+ return {key: thaw_parameter(item) for key, item in value[1]}
981
+ if all(
982
+ isinstance(item, tuple) and len(item) == 2 and isinstance(item[0], str)
983
+ for item in value
984
+ ):
985
+ return {key: thaw_parameter(item) for key, item in value}
986
+ return [thaw_parameter(item) for item in value]
987
+ return value
988
+
989
+
990
+ __all__ = [
991
+ "DERIVE_EXPRESSION_ARITY",
992
+ "DERIVE_EXPRESSION_VARIADIC",
993
+ "DIVIDE_POLICIES",
994
+ "EXECUTABLE",
995
+ "GRAPH_JOIN_KINDS",
996
+ "GRAPH_OPERATION_CONTRACT_VERSION",
997
+ "GRAPH_OPERATION_VERSION",
998
+ "MAX_DATE_ADD_DAYS",
999
+ "OPERATION_REGISTRY",
1000
+ "RETIRED",
1001
+ "RETIRED_OPERATIONS",
1002
+ "AdmissionRow",
1003
+ "GraphOperation",
1004
+ "OperationRegistryError",
1005
+ "resolve_operation",
1006
+ "thaw_parameter",
1007
+ ]