mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,2880 @@
1
+ """Versioned, immutable Harness-private execution contracts.
2
+
3
+ These types describe local agent proposals and deterministic execution inputs. They deliberately
4
+ exclude Studio tenancy, authorization, persistence, and release fields; cross-repository values
5
+ are mapped to Studio's generated client only at the hosted boundary.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import re
11
+ from collections.abc import Iterable, Mapping, Sequence
12
+ from dataclasses import dataclass
13
+ from datetime import datetime
14
+ from pathlib import PurePosixPath
15
+ from typing import Any, ClassVar
16
+ from urllib.parse import parse_qsl, unquote_plus, urlsplit, urlunsplit
17
+
18
+ from mostlyright.data_harness.canonical import canonical_json_bytes, canonical_sha256
19
+ from mostlyright.data_harness.formats import PLAN_DATA_FORMATS
20
+ from mostlyright.data_harness.operation_registry import (
21
+ GRAPH_OPERATION_CONTRACT_VERSION,
22
+ RETIRED_OPERATIONS,
23
+ OperationRegistryError,
24
+ resolve_operation,
25
+ thaw_parameter,
26
+ )
27
+ from mostlyright.data_harness.units import (
28
+ NO_UNIT,
29
+ Unit,
30
+ UnitError,
31
+ is_count_unit,
32
+ legacy_aliases,
33
+ resolve_declared_unit,
34
+ )
35
+
36
+ QUESTION_SCHEMA_VERSION = "local-question.v1"
37
+ OUTPUT_GRAIN_SCHEMA_VERSION = "local-output-grain.v1"
38
+ REQUIREMENTS_SCHEMA_VERSION = "local-requirements.v1"
39
+ SOURCE_PROPOSAL_SCHEMA_VERSION = "local-source-proposal.v2"
40
+ PLAN_SCHEMA_VERSION = "local-table-plan.v1"
41
+ _LEGACY_PLAN_SCHEMA_VERSION = "local-plan.v1"
42
+ OPERATION_CONTRACT_VERSION = "local-operations.v1"
43
+ GRAPH_PLAN_SCHEMA_VERSION = "local-graph-table-plan.v1"
44
+ VALIDATION_POLICY_VERSION = "local-validation.v1"
45
+ # The first semantics generation carries the fifteen hand-grown tokens and nothing else; the
46
+ # second carries a unit code, which is any expression the grammar in :mod:`units` resolves against
47
+ # its pinned table. The two are read side by side forever: a document sealed under the first
48
+ # generation says what vocabulary its author was writing against, and re-reading it must not widen
49
+ # that. New documents are authored at the second.
50
+ TABLE_SEMANTICS_V1 = "table-semantics.v1"
51
+ TABLE_SEMANTICS_V2 = "table-semantics.v2"
52
+ TABLE_SEMANTICS_VERSIONS = (TABLE_SEMANTICS_V1, TABLE_SEMANTICS_V2)
53
+ TABLE_SEMANTICS_VERSION = TABLE_SEMANTICS_V2
54
+
55
+ MAX_IDENTIFIER_LENGTH = 64
56
+ MAX_COLUMN_COUNT = 256
57
+ MAX_SOURCE_COUNT = 64
58
+ MAX_OPERATION_COUNT = 256
59
+ MAX_GRAPH_NODE_COUNT = 512
60
+ MAX_TEXT_LENGTH = 16_384
61
+ MAX_SHORT_TEXT_LENGTH = 2_000
62
+ MAX_PATH_LENGTH = 1_024
63
+ MAX_EVIDENCE_COUNT = 64
64
+ MAX_OUTPUT_ROWS = 100_000
65
+ GRAPH_MAX_OUTPUT_ROWS = 1_000_000
66
+
67
+ _IDENTIFIER = re.compile(r"^[a-z][a-z0-9]*(?:[-_][a-z0-9]+)*$")
68
+ _COLUMN = re.compile(r"^[a-z][a-z0-9_]{0,62}$")
69
+ _SHA256 = re.compile(r"^[0-9a-f]{64}$")
70
+ _UTC_TIMESTAMP = re.compile(
71
+ r"^(?P<date>[0-9]{4}-[0-9]{2}-[0-9]{2})"
72
+ r"T(?P<time>[0-9]{2}:[0-9]{2}:[0-9]{2})"
73
+ r"(?P<fraction>\.[0-9]{1,6})?Z$"
74
+ )
75
+ _SENSITIVE_LOCATOR_QUERY_KEY = re.compile(
76
+ r"(?:^|[_-])(?:api[_-]?key|authorization|bearer|credential|password|private[_-]?key|"
77
+ r"secret|sig|signature|signed|token)(?:$|[_-])",
78
+ re.IGNORECASE,
79
+ )
80
+ _SECRET_LOCATOR_VALUE = re.compile(
81
+ r"^(?:Bearer\s+|Basic\s+|sk[-_]|gh[opusr]_|AIza|AKIA|ASIA|eyJ[A-Za-z0-9_-]*\.)",
82
+ re.IGNORECASE,
83
+ )
84
+
85
+ # The complete set of boundaries Python's text renderer treats as a new line. Tabs and ordinary
86
+ # Unicode remain valid within a line; U+009B and bidi controls are deliberately not decided here.
87
+ # This is structural line safety, not a terminal-control policy.
88
+ _DISPLAY_LINE_BOUNDARIES = frozenset(
89
+ {"\n", "\r", "\v", "\f", "\x1c", "\x1d", "\x1e", "\x85", "\u2028", "\u2029"}
90
+ )
91
+
92
+ _QUESTION_FIELDS = frozenset({"schema_version", "question_id", "text", "created_at"})
93
+ _TARGET_POLICIES = frozenset({"none", "required_if_supportable"})
94
+ _FEASIBILITY_DECISIONS = frozenset({"supportable", "supportable_with_limits", "unsupported"})
95
+ _FEASIBILITY_REASONS = frozenset(
96
+ {
97
+ "sufficient_evidence",
98
+ "limited_coverage",
99
+ "unacceptable_delay",
100
+ "rights_unclear",
101
+ "no_reliable_source",
102
+ "target_not_observable",
103
+ }
104
+ )
105
+ # Narrower than the model's coarse family; see the vocabulary map in sources/contracts.py.
106
+ _SOURCE_CLASSES = frozenset(
107
+ {
108
+ "external_adapter",
109
+ "user_file",
110
+ "user_url",
111
+ "user_api",
112
+ "database_extract",
113
+ "webhook",
114
+ "stream",
115
+ }
116
+ )
117
+ _LOCATOR_KINDS = frozenset({"relative_path", "https_url", "artifact_reference"})
118
+ # Mixes encodings with access shapes; see the vocabulary map in sources/contracts.py.
119
+ # This is the plan-contract set, not the wire-format set: dropping to DATA_FORMATS here
120
+ # would silently remove api and stream from plan validation.
121
+ _DATA_FORMATS = PLAN_DATA_FORMATS
122
+ # No unknown state here, unlike model liveness; see the vocabulary map in sources/contracts.py.
123
+ _LIVE_ENDPOINTS = frozenset({"live", "degraded", "delayed", "dead", "not_applicable"})
124
+ # Same spelling in all three layers; see the vocabulary map in sources/contracts.py.
125
+ _PROPOSAL_RIGHTS = frozenset({"approved", "conditional", "unclear", "prohibited"})
126
+ # Stage-specific triplet; see the vocabulary map in sources/contracts.py.
127
+ _FITNESS_DECISIONS = frozenset({"selected", "eligible", "rejected"})
128
+
129
+ # Rights basis a build reads a file under, not a rights adjudication: this unknown is not
130
+ # _PROPOSAL_RIGHTS.unclear. See the vocabulary map in sources/contracts.py.
131
+ _RIGHTS_STATUSES = frozenset(
132
+ {
133
+ "project_owned",
134
+ "public_domain",
135
+ "permissive_license",
136
+ "authorized_internal",
137
+ "user_authorized",
138
+ "unknown",
139
+ "prohibited",
140
+ }
141
+ )
142
+ _PERMISSIONS = frozenset({"local_use", "redistribute", "host"})
143
+ _ACQUISITION_METHODS = frozenset({"checked_in_fixture", "user_upload", "authorized_export"})
144
+ _OUTPUT_INTENTS = frozenset({"local_use", "redistribute", "host"})
145
+ # The one authoritative logical-type vocabulary shared by Recipe declarations and the
146
+ # deterministic local transform path. ``_CAST_TYPES`` remains as a private compatibility alias
147
+ # for callers that inspect the plan contract, but every accepted logical type is executable.
148
+ LOGICAL_TYPES = frozenset({"string", "int64", "float64", "boolean", "date", "timestamp_utc"})
149
+ _CAST_TYPES = LOGICAL_TYPES
150
+ _CLEANING_OPERATIONS = frozenset({"trim", "empty_to_null", "rename", "cast"})
151
+ # The whole of what a ``local-plan.v1`` join may declare, frozen at one kind by the ruling in
152
+ # ``docs/TRANSFORMS.md``. It is written out as a name rather than left inline because the set is a
153
+ # decision about sealed meaning and not a list that grows: the v1 join rule is named in the
154
+ # ``mandatory_gates`` list ``validation_policy_digest`` hashes, and every approved v1 Recipe is
155
+ # bound to that digest. ``JoinSpec`` below says why each of the graph's other two kinds is refused
156
+ # here; join-kind growth happens in ``operation_registry.GRAPH_JOIN_KINDS`` instead.
157
+ JOIN_KINDS = frozenset({"left"})
158
+ _JOIN_CARDINALITIES = frozenset({"one_to_one", "many_to_one"})
159
+ SEMANTIC_TYPES = frozenset(
160
+ {"identifier", "entity", "category", "measure", "dimension", "target", "other"}
161
+ )
162
+ # The fifteen tokens the hand-grown list held. They are no longer the vocabulary: a declared unit
163
+ # is a code the grammar resolves, and these fourteen names plus ``none`` are read as the codes they
164
+ # always meant. What they still are is the whole of what ``table-semantics.v1`` may declare, and
165
+ # a frozen generation cannot be a function of a data file that a later change might extend -- so
166
+ # the fifteen are written out here, and the equality below binds them to the pinned alias table.
167
+ # Adding an alias to that table without deciding what it means for the frozen generation is an
168
+ # explicit import-time failure rather than a silent widening of sealed meaning.
169
+ UNITS = frozenset(
170
+ {
171
+ "none",
172
+ "count",
173
+ "ratio",
174
+ "percent",
175
+ "celsius",
176
+ "fahrenheit",
177
+ "kelvin",
178
+ "meter",
179
+ "kilometer",
180
+ "mile",
181
+ "microgram_per_cubic_meter",
182
+ "second",
183
+ "minute",
184
+ "hour",
185
+ "hectopascal",
186
+ }
187
+ )
188
+ assert UNITS == frozenset({NO_UNIT}) | frozenset(legacy_aliases())
189
+ UNIT_STATES = frozenset({"declared", "normalized", "not_applicable", "unknown"})
190
+ NORMALIZATIONS = frozenset(
191
+ {"z_score", "min_max", "unit_interval", "percent_of_total", "index_100", "log", "other"}
192
+ )
193
+ CATEGORY_STATUSES = frozenset({"complete", "partial", "not_applicable", "unknown"})
194
+ SEMANTIC_EVIDENCE_KINDS = frozenset({"source_id"})
195
+
196
+
197
+ class ContractError(ValueError):
198
+ """A typed private-contract validation failure with an exact field path."""
199
+
200
+ def __init__(self, path: str, code: str, detail: str) -> None:
201
+ self.path = path
202
+ self.code = code
203
+ self.detail = detail
204
+ super().__init__(f"{path}: {detail} [{code}]")
205
+
206
+
207
+ class _CanonicalContract:
208
+ """Common deterministic serialization behavior for private contracts."""
209
+
210
+ schema_version: str
211
+ EXPECTED_SCHEMA_VERSION: ClassVar[str]
212
+
213
+ def to_dict(self) -> dict[str, Any]: # pragma: no cover - abstract-by-convention
214
+ raise NotImplementedError
215
+
216
+ @property
217
+ def canonical_bytes(self) -> bytes:
218
+ return canonical_json_bytes(self.to_dict())
219
+
220
+ @property
221
+ def digest(self) -> str:
222
+ return canonical_sha256(self.to_dict())
223
+
224
+
225
+ @dataclass(frozen=True)
226
+ class LocalQuestion(_CanonicalContract):
227
+ """A local question proposal without hosted identity or authorization fields."""
228
+
229
+ question_id: str
230
+ text: str
231
+ created_at: str
232
+ schema_version: str = QUESTION_SCHEMA_VERSION
233
+
234
+ EXPECTED_SCHEMA_VERSION: ClassVar[str] = QUESTION_SCHEMA_VERSION
235
+
236
+ def __post_init__(self) -> None:
237
+ _version(self.schema_version, QUESTION_SCHEMA_VERSION, "question.schema_version")
238
+ _identifier(self.question_id, "question.question_id")
239
+ _text(self.text, "question.text", minimum=1, maximum=12_000)
240
+ _utc_timestamp(self.created_at, "question.created_at")
241
+
242
+ def to_dict(self) -> dict[str, Any]:
243
+ return {
244
+ "schema_version": self.schema_version,
245
+ "question_id": self.question_id,
246
+ "text": self.text,
247
+ "created_at": self.created_at,
248
+ }
249
+
250
+ @classmethod
251
+ def from_value(cls, value: Any) -> LocalQuestion:
252
+ data = _object(value, "question")
253
+ _exact_fields(data, _QUESTION_FIELDS, "question")
254
+ return cls(
255
+ schema_version=_required_text(
256
+ data["schema_version"],
257
+ "question.schema_version",
258
+ maximum=64,
259
+ ),
260
+ question_id=_required_text(
261
+ data["question_id"],
262
+ "question.question_id",
263
+ maximum=MAX_IDENTIFIER_LENGTH,
264
+ ),
265
+ text=_required_text(data["text"], "question.text", maximum=12_000),
266
+ created_at=_required_text(
267
+ data["created_at"],
268
+ "question.created_at",
269
+ maximum=32,
270
+ ),
271
+ )
272
+
273
+
274
+ @dataclass(frozen=True)
275
+ class OutputGrain(_CanonicalContract):
276
+ """Ordered local output key definition."""
277
+
278
+ columns: tuple[str, ...]
279
+ description: str
280
+ schema_version: str = OUTPUT_GRAIN_SCHEMA_VERSION
281
+
282
+ EXPECTED_SCHEMA_VERSION: ClassVar[str] = OUTPUT_GRAIN_SCHEMA_VERSION
283
+
284
+ def __post_init__(self) -> None:
285
+ _version(
286
+ self.schema_version,
287
+ OUTPUT_GRAIN_SCHEMA_VERSION,
288
+ "output_grain.schema_version",
289
+ )
290
+ _column_tuple(
291
+ self.columns,
292
+ "output_grain.columns",
293
+ nonempty=True,
294
+ maximum=16,
295
+ )
296
+ _text(
297
+ self.description,
298
+ "output_grain.description",
299
+ minimum=1,
300
+ maximum=1_000,
301
+ )
302
+
303
+ def to_dict(self) -> dict[str, Any]:
304
+ return {
305
+ "schema_version": self.schema_version,
306
+ "columns": list(self.columns),
307
+ "description": self.description,
308
+ }
309
+
310
+ @classmethod
311
+ def from_value(cls, value: Any, path: str = "output_grain") -> OutputGrain:
312
+ data = _object(value, path)
313
+ _exact_fields(data, {"schema_version", "columns", "description"}, path)
314
+ return cls(
315
+ schema_version=_required_text(
316
+ data["schema_version"],
317
+ f"{path}.schema_version",
318
+ maximum=64,
319
+ ),
320
+ columns=_parse_column_array(
321
+ data["columns"],
322
+ f"{path}.columns",
323
+ nonempty=True,
324
+ maximum=16,
325
+ ),
326
+ description=_required_text(
327
+ data["description"],
328
+ f"{path}.description",
329
+ maximum=1_000,
330
+ ),
331
+ )
332
+
333
+
334
+ @dataclass(frozen=True)
335
+ class TimeRange:
336
+ """A half-open UTC time range used by private requirements and proposals."""
337
+
338
+ start_inclusive: str
339
+ end_exclusive: str
340
+
341
+ def __post_init__(self) -> None:
342
+ start = _utc_timestamp(self.start_inclusive, "time_range.start_inclusive")
343
+ end = _utc_timestamp(self.end_exclusive, "time_range.end_exclusive")
344
+ if start >= end:
345
+ raise ContractError(
346
+ "time_range.end_exclusive",
347
+ "RANGE_ORDER",
348
+ "must be later than time_range.start_inclusive",
349
+ )
350
+
351
+ def to_dict(self) -> dict[str, str]:
352
+ return {
353
+ "start_inclusive": self.start_inclusive,
354
+ "end_exclusive": self.end_exclusive,
355
+ }
356
+
357
+ @classmethod
358
+ def from_value(cls, value: Any, path: str) -> TimeRange:
359
+ data = _object(value, path)
360
+ _exact_fields(data, {"start_inclusive", "end_exclusive"}, path)
361
+ start = _required_text(
362
+ data["start_inclusive"],
363
+ f"{path}.start_inclusive",
364
+ maximum=32,
365
+ )
366
+ end = _required_text(
367
+ data["end_exclusive"],
368
+ f"{path}.end_exclusive",
369
+ maximum=32,
370
+ )
371
+ try:
372
+ return cls(start_inclusive=start, end_exclusive=end)
373
+ except ContractError as exc:
374
+ if exc.path.startswith("time_range."):
375
+ suffix = exc.path.removeprefix("time_range.")
376
+ raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
377
+ raise
378
+
379
+
380
+ @dataclass(frozen=True)
381
+ class Feasibility:
382
+ """Deterministically bounded feasibility proposal."""
383
+
384
+ decision: str
385
+ reason_codes: tuple[str, ...]
386
+ narrative: str
387
+
388
+ def __post_init__(self) -> None:
389
+ _choice(self.decision, _FEASIBILITY_DECISIONS, "feasibility.decision")
390
+ _choice_tuple(
391
+ self.reason_codes,
392
+ _FEASIBILITY_REASONS,
393
+ "feasibility.reason_codes",
394
+ maximum=16,
395
+ )
396
+ _text(
397
+ self.narrative,
398
+ "feasibility.narrative",
399
+ minimum=1,
400
+ maximum=4_000,
401
+ )
402
+ if self.decision == "supportable" and "sufficient_evidence" not in self.reason_codes:
403
+ raise ContractError(
404
+ "feasibility.reason_codes",
405
+ "MISSING_REASON",
406
+ "supportable feasibility requires sufficient_evidence",
407
+ )
408
+ if self.decision == "unsupported" and not self.reason_codes:
409
+ raise ContractError(
410
+ "feasibility.reason_codes",
411
+ "EMPTY_COLLECTION",
412
+ "unsupported feasibility requires at least one reason code",
413
+ )
414
+
415
+ def to_dict(self) -> dict[str, Any]:
416
+ return {
417
+ "decision": self.decision,
418
+ "reason_codes": list(self.reason_codes),
419
+ "narrative": self.narrative,
420
+ }
421
+
422
+ @classmethod
423
+ def from_value(cls, value: Any, path: str) -> Feasibility:
424
+ data = _object(value, path)
425
+ _exact_fields(data, {"decision", "reason_codes", "narrative"}, path)
426
+ try:
427
+ return cls(
428
+ decision=_required_text(
429
+ data["decision"],
430
+ f"{path}.decision",
431
+ maximum=64,
432
+ ),
433
+ reason_codes=_parse_choice_array(
434
+ data["reason_codes"],
435
+ f"{path}.reason_codes",
436
+ _FEASIBILITY_REASONS,
437
+ nonempty=False,
438
+ maximum=16,
439
+ ),
440
+ narrative=_required_text(
441
+ data["narrative"],
442
+ f"{path}.narrative",
443
+ maximum=4_000,
444
+ ),
445
+ )
446
+ except ContractError as exc:
447
+ if exc.path.startswith("feasibility."):
448
+ suffix = exc.path.removeprefix("feasibility.")
449
+ raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
450
+ raise
451
+
452
+
453
+ @dataclass(frozen=True)
454
+ class LocalRequirements(_CanonicalContract):
455
+ """Question-derived local dataset requirements."""
456
+
457
+ requirements_id: str
458
+ question_id: str
459
+ population: str
460
+ time_range: TimeRange
461
+ output_grain: OutputGrain
462
+ required_fields: tuple[str, ...]
463
+ target_policy: str
464
+ success_criteria: tuple[str, ...]
465
+ feasibility: Feasibility
466
+ created_at: str
467
+ schema_version: str = REQUIREMENTS_SCHEMA_VERSION
468
+
469
+ EXPECTED_SCHEMA_VERSION: ClassVar[str] = REQUIREMENTS_SCHEMA_VERSION
470
+
471
+ def __post_init__(self) -> None:
472
+ _version(
473
+ self.schema_version,
474
+ REQUIREMENTS_SCHEMA_VERSION,
475
+ "requirements.schema_version",
476
+ )
477
+ _identifier(self.requirements_id, "requirements.requirements_id")
478
+ _identifier(self.question_id, "requirements.question_id")
479
+ _text(
480
+ self.population,
481
+ "requirements.population",
482
+ minimum=1,
483
+ maximum=MAX_SHORT_TEXT_LENGTH,
484
+ )
485
+ if not isinstance(self.time_range, TimeRange):
486
+ raise ContractError(
487
+ "requirements.time_range",
488
+ "TYPE",
489
+ "must be a TimeRange",
490
+ )
491
+ if not isinstance(self.output_grain, OutputGrain):
492
+ raise ContractError(
493
+ "requirements.output_grain",
494
+ "TYPE",
495
+ "must be an OutputGrain",
496
+ )
497
+ _column_tuple(
498
+ self.required_fields,
499
+ "requirements.required_fields",
500
+ nonempty=True,
501
+ maximum=MAX_COLUMN_COUNT,
502
+ )
503
+ missing_grain = set(self.output_grain.columns) - set(self.required_fields)
504
+ if missing_grain:
505
+ raise ContractError(
506
+ "requirements.required_fields",
507
+ "GRAIN_FIELD_MISSING",
508
+ f"must contain output-grain columns {sorted(missing_grain)}",
509
+ )
510
+ _choice(self.target_policy, _TARGET_POLICIES, "requirements.target_policy")
511
+ _text_tuple(
512
+ self.success_criteria,
513
+ "requirements.success_criteria",
514
+ nonempty=True,
515
+ maximum=64,
516
+ item_maximum=1_000,
517
+ )
518
+ if not isinstance(self.feasibility, Feasibility):
519
+ raise ContractError(
520
+ "requirements.feasibility",
521
+ "TYPE",
522
+ "must be Feasibility",
523
+ )
524
+ _utc_timestamp(self.created_at, "requirements.created_at")
525
+
526
+ def to_dict(self) -> dict[str, Any]:
527
+ return {
528
+ "schema_version": self.schema_version,
529
+ "requirements_id": self.requirements_id,
530
+ "question_id": self.question_id,
531
+ "population": self.population,
532
+ "time_range": self.time_range.to_dict(),
533
+ "output_grain": self.output_grain.to_dict(),
534
+ "required_fields": list(self.required_fields),
535
+ "target_policy": self.target_policy,
536
+ "success_criteria": list(self.success_criteria),
537
+ "feasibility": self.feasibility.to_dict(),
538
+ "created_at": self.created_at,
539
+ }
540
+
541
+ @classmethod
542
+ def from_value(cls, value: Any) -> LocalRequirements:
543
+ path = "requirements"
544
+ data = _object(value, path)
545
+ _exact_fields(
546
+ data,
547
+ {
548
+ "schema_version",
549
+ "requirements_id",
550
+ "question_id",
551
+ "population",
552
+ "time_range",
553
+ "output_grain",
554
+ "required_fields",
555
+ "target_policy",
556
+ "success_criteria",
557
+ "feasibility",
558
+ "created_at",
559
+ },
560
+ path,
561
+ )
562
+ return cls(
563
+ schema_version=_required_text(
564
+ data["schema_version"],
565
+ f"{path}.schema_version",
566
+ maximum=64,
567
+ ),
568
+ requirements_id=_required_text(
569
+ data["requirements_id"],
570
+ f"{path}.requirements_id",
571
+ maximum=MAX_IDENTIFIER_LENGTH,
572
+ ),
573
+ question_id=_required_text(
574
+ data["question_id"],
575
+ f"{path}.question_id",
576
+ maximum=MAX_IDENTIFIER_LENGTH,
577
+ ),
578
+ population=_required_text(
579
+ data["population"],
580
+ f"{path}.population",
581
+ maximum=MAX_SHORT_TEXT_LENGTH,
582
+ ),
583
+ time_range=TimeRange.from_value(data["time_range"], f"{path}.time_range"),
584
+ output_grain=OutputGrain.from_value(
585
+ data["output_grain"],
586
+ f"{path}.output_grain",
587
+ ),
588
+ required_fields=_parse_column_array(
589
+ data["required_fields"],
590
+ f"{path}.required_fields",
591
+ nonempty=True,
592
+ maximum=MAX_COLUMN_COUNT,
593
+ ),
594
+ target_policy=_required_text(
595
+ data["target_policy"],
596
+ f"{path}.target_policy",
597
+ maximum=64,
598
+ ),
599
+ success_criteria=_parse_text_array(
600
+ data["success_criteria"],
601
+ f"{path}.success_criteria",
602
+ nonempty=True,
603
+ maximum=64,
604
+ item_maximum=1_000,
605
+ ),
606
+ feasibility=Feasibility.from_value(data["feasibility"], f"{path}.feasibility"),
607
+ created_at=_required_text(
608
+ data["created_at"],
609
+ f"{path}.created_at",
610
+ maximum=32,
611
+ ),
612
+ )
613
+
614
+
615
+ @dataclass(frozen=True)
616
+ class EvidenceReference:
617
+ """One immutable, non-secret source-proposal evidence reference."""
618
+
619
+ uri: str
620
+ observed_at: str
621
+ content_sha256: str
622
+
623
+ def __post_init__(self) -> None:
624
+ _evidence_uri(self.uri, "evidence.uri")
625
+ _utc_timestamp(self.observed_at, "evidence.observed_at")
626
+ _sha256(self.content_sha256, "evidence.content_sha256")
627
+
628
+ def to_dict(self) -> dict[str, str]:
629
+ return {
630
+ "uri": self.uri,
631
+ "observed_at": self.observed_at,
632
+ "content_sha256": self.content_sha256,
633
+ }
634
+
635
+ @classmethod
636
+ def from_value(cls, value: Any, path: str) -> EvidenceReference:
637
+ data = _object(value, path)
638
+ _exact_fields(data, {"uri", "observed_at", "content_sha256"}, path)
639
+ try:
640
+ return cls(
641
+ uri=_required_text(data["uri"], f"{path}.uri", maximum=2_048),
642
+ observed_at=_required_text(
643
+ data["observed_at"],
644
+ f"{path}.observed_at",
645
+ maximum=32,
646
+ ),
647
+ content_sha256=_required_text(
648
+ data["content_sha256"],
649
+ f"{path}.content_sha256",
650
+ maximum=64,
651
+ ),
652
+ )
653
+ except ContractError as exc:
654
+ if exc.path.startswith("evidence."):
655
+ suffix = exc.path.removeprefix("evidence.")
656
+ raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
657
+ raise
658
+
659
+
660
+ @dataclass(frozen=True)
661
+ class SourceProposal(_CanonicalContract):
662
+ """A bounded agent proposal; it is never an acquisition authorization."""
663
+
664
+ proposal_id: str
665
+ question_id: str
666
+ source_id: str
667
+ source_class: str
668
+ display_name: str
669
+ locator_kind: str
670
+ locator: str
671
+ data_format: str
672
+ observed_at: str
673
+ historical_coverage: TimeRange
674
+ live_endpoint: str
675
+ publication_delay_seconds: int
676
+ rights_status: str
677
+ fitness_decision: str
678
+ evidence: tuple[EvidenceReference, ...]
679
+ schema_version: str = SOURCE_PROPOSAL_SCHEMA_VERSION
680
+
681
+ EXPECTED_SCHEMA_VERSION: ClassVar[str] = SOURCE_PROPOSAL_SCHEMA_VERSION
682
+
683
+ def __post_init__(self) -> None:
684
+ _version(
685
+ self.schema_version,
686
+ SOURCE_PROPOSAL_SCHEMA_VERSION,
687
+ "source_proposal.schema_version",
688
+ )
689
+ _identifier(self.proposal_id, "source_proposal.proposal_id")
690
+ _identifier(self.question_id, "source_proposal.question_id")
691
+ _identifier(self.source_id, "source_proposal.source_id")
692
+ _choice(self.source_class, _SOURCE_CLASSES, "source_proposal.source_class")
693
+ _text(
694
+ self.display_name,
695
+ "source_proposal.display_name",
696
+ minimum=1,
697
+ maximum=200,
698
+ )
699
+ _choice(self.locator_kind, _LOCATOR_KINDS, "source_proposal.locator_kind")
700
+ validate_source_locator(self.locator, self.locator_kind, "source_proposal.locator")
701
+ _choice(self.data_format, _DATA_FORMATS, "source_proposal.data_format")
702
+ _utc_timestamp(self.observed_at, "source_proposal.observed_at")
703
+ if not isinstance(self.historical_coverage, TimeRange):
704
+ raise ContractError(
705
+ "source_proposal.historical_coverage",
706
+ "TYPE",
707
+ "must be a TimeRange",
708
+ )
709
+ _choice(self.live_endpoint, _LIVE_ENDPOINTS, "source_proposal.live_endpoint")
710
+ _bounded_int(
711
+ self.publication_delay_seconds,
712
+ "source_proposal.publication_delay_seconds",
713
+ minimum=0,
714
+ maximum=10 * 365 * 24 * 60 * 60,
715
+ )
716
+ _choice(self.rights_status, _PROPOSAL_RIGHTS, "source_proposal.rights_status")
717
+ _choice(
718
+ self.fitness_decision,
719
+ _FITNESS_DECISIONS,
720
+ "source_proposal.fitness_decision",
721
+ )
722
+ _typed_tuple(
723
+ self.evidence,
724
+ EvidenceReference,
725
+ "source_proposal.evidence",
726
+ nonempty=True,
727
+ maximum=MAX_EVIDENCE_COUNT,
728
+ )
729
+ evidence_keys = tuple(
730
+ (item.uri, item.observed_at, item.content_sha256) for item in self.evidence
731
+ )
732
+ _require_unique(evidence_keys, "source_proposal.evidence")
733
+ if self.fitness_decision == "selected" and self.rights_status not in {
734
+ "approved",
735
+ "conditional",
736
+ }:
737
+ raise ContractError(
738
+ "source_proposal.rights_status",
739
+ "RIGHTS_NOT_SELECTABLE",
740
+ "selected proposals require approved or conditional rights",
741
+ )
742
+ if self.fitness_decision == "selected" and self.live_endpoint in {"dead", "delayed"}:
743
+ raise ContractError(
744
+ "source_proposal.live_endpoint",
745
+ "ENDPOINT_NOT_SELECTABLE",
746
+ "selected proposals cannot have a dead or delayed endpoint",
747
+ )
748
+
749
+ def to_dict(self) -> dict[str, Any]:
750
+ return {
751
+ "schema_version": self.schema_version,
752
+ "proposal_id": self.proposal_id,
753
+ "question_id": self.question_id,
754
+ "source_id": self.source_id,
755
+ "source_class": self.source_class,
756
+ "display_name": self.display_name,
757
+ "locator_kind": self.locator_kind,
758
+ "locator": self.locator,
759
+ "data_format": self.data_format,
760
+ "observed_at": self.observed_at,
761
+ "historical_coverage": self.historical_coverage.to_dict(),
762
+ "live_endpoint": self.live_endpoint,
763
+ "publication_delay_seconds": self.publication_delay_seconds,
764
+ "rights_status": self.rights_status,
765
+ "fitness_decision": self.fitness_decision,
766
+ "evidence": [item.to_dict() for item in self.evidence],
767
+ }
768
+
769
+ @classmethod
770
+ def from_value(cls, value: Any) -> SourceProposal:
771
+ path = "source_proposal"
772
+ data = _object(value, path)
773
+ _exact_fields(
774
+ data,
775
+ {
776
+ "schema_version",
777
+ "proposal_id",
778
+ "question_id",
779
+ "source_id",
780
+ "source_class",
781
+ "display_name",
782
+ "locator_kind",
783
+ "locator",
784
+ "data_format",
785
+ "observed_at",
786
+ "historical_coverage",
787
+ "live_endpoint",
788
+ "publication_delay_seconds",
789
+ "rights_status",
790
+ "fitness_decision",
791
+ "evidence",
792
+ },
793
+ path,
794
+ )
795
+ evidence = tuple(
796
+ EvidenceReference.from_value(item, f"{path}.evidence[{index}]")
797
+ for index, item in enumerate(
798
+ _array(
799
+ data["evidence"],
800
+ f"{path}.evidence",
801
+ nonempty=True,
802
+ maximum=MAX_EVIDENCE_COUNT,
803
+ )
804
+ )
805
+ )
806
+ return cls(
807
+ schema_version=_required_text(
808
+ data["schema_version"],
809
+ f"{path}.schema_version",
810
+ maximum=64,
811
+ ),
812
+ proposal_id=_required_text(
813
+ data["proposal_id"],
814
+ f"{path}.proposal_id",
815
+ maximum=MAX_IDENTIFIER_LENGTH,
816
+ ),
817
+ question_id=_required_text(
818
+ data["question_id"],
819
+ f"{path}.question_id",
820
+ maximum=MAX_IDENTIFIER_LENGTH,
821
+ ),
822
+ source_id=_required_text(
823
+ data["source_id"],
824
+ f"{path}.source_id",
825
+ maximum=MAX_IDENTIFIER_LENGTH,
826
+ ),
827
+ source_class=_required_text(
828
+ data["source_class"],
829
+ f"{path}.source_class",
830
+ maximum=64,
831
+ ),
832
+ display_name=_required_text(
833
+ data["display_name"],
834
+ f"{path}.display_name",
835
+ maximum=200,
836
+ ),
837
+ locator_kind=_required_text(
838
+ data["locator_kind"],
839
+ f"{path}.locator_kind",
840
+ maximum=64,
841
+ ),
842
+ locator=_required_text(
843
+ data["locator"],
844
+ f"{path}.locator",
845
+ maximum=MAX_PATH_LENGTH,
846
+ ),
847
+ data_format=_required_text(
848
+ data["data_format"],
849
+ f"{path}.data_format",
850
+ maximum=64,
851
+ ),
852
+ observed_at=_required_text(
853
+ data["observed_at"],
854
+ f"{path}.observed_at",
855
+ maximum=32,
856
+ ),
857
+ historical_coverage=TimeRange.from_value(
858
+ data["historical_coverage"],
859
+ f"{path}.historical_coverage",
860
+ ),
861
+ live_endpoint=_required_text(
862
+ data["live_endpoint"],
863
+ f"{path}.live_endpoint",
864
+ maximum=64,
865
+ ),
866
+ publication_delay_seconds=_integer(
867
+ data["publication_delay_seconds"],
868
+ f"{path}.publication_delay_seconds",
869
+ ),
870
+ rights_status=_required_text(
871
+ data["rights_status"],
872
+ f"{path}.rights_status",
873
+ maximum=64,
874
+ ),
875
+ fitness_decision=_required_text(
876
+ data["fitness_decision"],
877
+ f"{path}.fitness_decision",
878
+ maximum=64,
879
+ ),
880
+ evidence=evidence,
881
+ )
882
+
883
+
884
+ @dataclass(frozen=True)
885
+ class RightsSpec:
886
+ """The rights basis claimed for one source and the uses it permits."""
887
+
888
+ status: str
889
+ evidence: str
890
+ permissions: tuple[str, ...]
891
+
892
+ def __post_init__(self) -> None:
893
+ _choice(self.status, _RIGHTS_STATUSES, "rights.status")
894
+ _text(self.evidence, "rights.evidence", minimum=1, maximum=MAX_TEXT_LENGTH)
895
+ _choice_tuple(
896
+ self.permissions,
897
+ _PERMISSIONS,
898
+ "rights.permissions",
899
+ maximum=len(_PERMISSIONS),
900
+ nonempty=True,
901
+ )
902
+
903
+ def to_dict(self) -> dict[str, Any]:
904
+ return {
905
+ "status": self.status,
906
+ "evidence": self.evidence,
907
+ "permissions": list(self.permissions),
908
+ }
909
+
910
+
911
+ @dataclass(frozen=True)
912
+ class SourceSpec:
913
+ """One input file the build reads.
914
+
915
+ Carries the source identifier, the path relative to the build input root, the format,
916
+ the stated origin, how the file was acquired, and the rights it is read under.
917
+ """
918
+
919
+ source_id: str
920
+ path: str
921
+ format: str
922
+ origin: str
923
+ acquisition_method: str
924
+ rights: RightsSpec
925
+
926
+ def __post_init__(self) -> None:
927
+ _identifier(self.source_id, "source.id")
928
+ _relative_posix_path(self.path, "source.path")
929
+ # Build input encoding, narrower than _DATA_FORMATS; see sources/contracts.py.
930
+ _choice(self.format, frozenset({"csv"}), "source.format")
931
+ _text(self.origin, "source.origin", minimum=1, maximum=4_000)
932
+ _choice(
933
+ self.acquisition_method,
934
+ _ACQUISITION_METHODS,
935
+ "source.acquisition_method",
936
+ )
937
+ if not isinstance(self.rights, RightsSpec):
938
+ raise ContractError("source.rights", "TYPE", "must be a RightsSpec")
939
+
940
+ def to_dict(self) -> dict[str, Any]:
941
+ return {
942
+ "id": self.source_id,
943
+ "path": self.path,
944
+ "format": self.format,
945
+ "origin": self.origin,
946
+ "acquisition_method": self.acquisition_method,
947
+ "rights": self.rights.to_dict(),
948
+ }
949
+
950
+
951
+ @dataclass(frozen=True)
952
+ class CleaningStep:
953
+ """One cleaning operation applied to named columns of one source.
954
+
955
+ ``trim`` and ``empty_to_null`` take plain column names. ``rename`` and ``cast`` take
956
+ ``(column, target)`` pairs, where the target is the new column name or the cast type.
957
+ """
958
+
959
+ source_id: str
960
+ operation: str
961
+ columns: tuple[str, ...] | tuple[tuple[str, str], ...]
962
+
963
+ def __post_init__(self) -> None:
964
+ _identifier(self.source_id, "cleaning.source")
965
+ _choice(self.operation, _CLEANING_OPERATIONS, "cleaning.operation")
966
+ if not isinstance(self.columns, tuple) or not self.columns:
967
+ raise ContractError(
968
+ "cleaning.columns",
969
+ "TYPE",
970
+ "must be a non-empty immutable tuple",
971
+ )
972
+ if self.operation in {"trim", "empty_to_null"}:
973
+ _column_tuple(
974
+ self.columns,
975
+ "cleaning.columns",
976
+ nonempty=True,
977
+ maximum=MAX_COLUMN_COUNT,
978
+ )
979
+ return
980
+ pairs = self.columns
981
+ if len(pairs) > MAX_COLUMN_COUNT:
982
+ raise ContractError(
983
+ "cleaning.columns",
984
+ "COLLECTION_LIMIT",
985
+ f"exceeds {MAX_COLUMN_COUNT} entries",
986
+ )
987
+ sources: list[str] = []
988
+ targets: list[str] = []
989
+ for index, pair in enumerate(pairs):
990
+ if (
991
+ not isinstance(pair, tuple)
992
+ or len(pair) != 2
993
+ or not all(isinstance(item, str) for item in pair)
994
+ ):
995
+ raise ContractError(
996
+ f"cleaning.columns[{index}]",
997
+ "TYPE",
998
+ "must be a (column, value) string pair",
999
+ )
1000
+ source, target = pair
1001
+ _column_name(source, f"cleaning.columns[{index}].source")
1002
+ if self.operation == "rename":
1003
+ _column_name(target, f"cleaning.columns[{index}].target")
1004
+ else:
1005
+ _choice(target, _CAST_TYPES, f"cleaning.columns[{index}].type")
1006
+ sources.append(source)
1007
+ targets.append(target)
1008
+ _require_unique(sources, "cleaning.columns.sources")
1009
+ if self.operation == "rename":
1010
+ _require_unique(targets, "cleaning.columns.targets")
1011
+
1012
+ def to_dict(self) -> dict[str, Any]:
1013
+ columns: Any
1014
+ if self.operation in {"rename", "cast"}:
1015
+ columns = dict(self.columns)
1016
+ else:
1017
+ columns = list(self.columns)
1018
+ return {
1019
+ "source": self.source_id,
1020
+ "operation": self.operation,
1021
+ "columns": columns,
1022
+ }
1023
+
1024
+
1025
+ @dataclass(frozen=True)
1026
+ class JoinSpec:
1027
+ """The single join a v1 plan performs.
1028
+
1029
+ Names two distinct sources, the key columns, the join kind, and the cardinality. The
1030
+ build enforces the declared cardinality rather than inferring it.
1031
+
1032
+ ``kind`` is ``left`` and stays ``left``, while the graph vocabulary carries ``left``, ``inner``
1033
+ and ``anti``. That is a settled ruling rather than an omission or a pending edit, and
1034
+ ``docs/TRANSFORMS.md`` records it. ``pipeline._join`` refuses the first unmatched left row and
1035
+ any row multiplier other than one, unconditionally, because a v1 plan has no way to declare
1036
+ anything else.
1037
+
1038
+ Under that rule ``inner`` cannot differ from ``left`` by a single row, so it would be a second
1039
+ spelling of one behaviour: two names an author would have to choose between with nothing to
1040
+ choose on, and no capability behind either. Admitting it would touch no digest, which makes it
1041
+ cheap rather than worth doing.
1042
+
1043
+ ``anti`` is the stronger case, and it starts from the fact that nothing here dispatches on
1044
+ ``kind`` at all. ``pipeline._join`` and both backends ignore it; its only reader in the whole
1045
+ software is ``ux.approve._join_sentence``, which renders it into the sentence a person is shown
1046
+ before approving. So widening this set alone would neither produce an anti join nor refuse one:
1047
+ the run would be the left join it always was, the word would be sealed into the plan bytes, and
1048
+ the approval sentence would say ``(anti join)`` about a result that is nothing of the kind.
1049
+ That is fail-open, and it lands on the one surface a person is asked to read.
1050
+
1051
+ Making the kind mean what it says requires withdrawing the v1 rule, because an anti join's
1052
+ output is exactly the rows that rule refuses. That rule is named in the ``mandatory_gates``
1053
+ list ``validation_policy_digest`` hashes, which every approved v1 Recipe is bound to.
1054
+ Withdrawing a name retires all of them; leaving the name in place while changing what it
1055
+ governs would silently change what they meant, which is worse.
1056
+
1057
+ Plans that need those kinds use the ``local-graph-table-plan.v1`` graph, where the join is a
1058
+ node carrying its own declared ``unmatched_policy`` and the same question is answered per node.
1059
+ """
1060
+
1061
+ left: str
1062
+ right: str
1063
+ on: tuple[str, ...]
1064
+ kind: str
1065
+ cardinality: str
1066
+
1067
+ def __post_init__(self) -> None:
1068
+ _identifier(self.left, "join.left")
1069
+ _identifier(self.right, "join.right")
1070
+ if self.left == self.right:
1071
+ raise ContractError("join.right", "SAME_SOURCE", "must differ from join.left")
1072
+ _column_tuple(self.on, "join.on", nonempty=True, maximum=16)
1073
+ _choice(self.kind, JOIN_KINDS, "join.kind")
1074
+ _choice(self.cardinality, _JOIN_CARDINALITIES, "join.cardinality")
1075
+
1076
+ def to_dict(self) -> dict[str, Any]:
1077
+ return {
1078
+ "left": self.left,
1079
+ "right": self.right,
1080
+ "on": list(self.on),
1081
+ "kind": self.kind,
1082
+ "cardinality": self.cardinality,
1083
+ }
1084
+
1085
+
1086
+ @dataclass(frozen=True)
1087
+ class QualitySpec:
1088
+ """The gates the joined output must pass.
1089
+
1090
+ A minimum row count, plus the columns that must contain no nulls.
1091
+ """
1092
+
1093
+ min_rows: int
1094
+ not_null: tuple[str, ...]
1095
+
1096
+ def __post_init__(self) -> None:
1097
+ _bounded_int(
1098
+ self.min_rows,
1099
+ "quality.min_rows",
1100
+ minimum=1,
1101
+ maximum=GRAPH_MAX_OUTPUT_ROWS,
1102
+ )
1103
+ _column_tuple(
1104
+ self.not_null,
1105
+ "quality.not_null",
1106
+ nonempty=False,
1107
+ maximum=MAX_COLUMN_COUNT,
1108
+ )
1109
+
1110
+ def to_dict(self) -> dict[str, Any]:
1111
+ return {"min_rows": self.min_rows, "not_null": list(self.not_null)}
1112
+
1113
+
1114
+ @dataclass(frozen=True)
1115
+ class SemanticEvidenceRef:
1116
+ """A bounded pointer to one source sealed alongside a semantic claim."""
1117
+
1118
+ kind: str
1119
+ ref: str
1120
+
1121
+ def __post_init__(self) -> None:
1122
+ _choice(self.kind, SEMANTIC_EVIDENCE_KINDS, "semantics.columns.evidence_refs.kind")
1123
+ _text(
1124
+ self.ref,
1125
+ "semantics.columns.evidence_refs.ref",
1126
+ minimum=1,
1127
+ maximum=128,
1128
+ )
1129
+ _identifier(self.ref, "semantics.columns.evidence_refs.ref")
1130
+
1131
+ def to_dict(self) -> dict[str, str]:
1132
+ return {"kind": self.kind, "ref": self.ref}
1133
+
1134
+
1135
+ @dataclass(frozen=True)
1136
+ class CategoryDefinition:
1137
+ """One bounded code-to-label definition from declared source semantics."""
1138
+
1139
+ code: str
1140
+ label: str
1141
+
1142
+ def __post_init__(self) -> None:
1143
+ _text(self.code, "semantics.columns.categories.code", minimum=1, maximum=64)
1144
+ _text(self.label, "semantics.columns.categories.label", minimum=1, maximum=120)
1145
+ if not is_single_plain_line(self.code) or not is_single_plain_line(self.label):
1146
+ raise ContractError(
1147
+ "semantics.columns.categories",
1148
+ "NONCANONICAL_TEXT",
1149
+ "codes and labels must each be one plain line",
1150
+ )
1151
+
1152
+ def to_dict(self) -> dict[str, str]:
1153
+ return {"code": self.code, "label": self.label}
1154
+
1155
+
1156
+ @dataclass(frozen=True)
1157
+ class SemanticCoverage:
1158
+ """Declared dataset coverage without inferred temporal or population facts."""
1159
+
1160
+ start_inclusive: str | None
1161
+ end_exclusive: str | None
1162
+ population: str
1163
+ completeness_note: str
1164
+
1165
+ def __post_init__(self) -> None:
1166
+ if (self.start_inclusive is None) != (self.end_exclusive is None):
1167
+ raise ContractError(
1168
+ "semantics.coverage.time_range",
1169
+ "FIELDS",
1170
+ "start_inclusive and end_exclusive must both be set or both be null",
1171
+ )
1172
+ for name, value in (
1173
+ ("start_inclusive", self.start_inclusive),
1174
+ ("end_exclusive", self.end_exclusive),
1175
+ ):
1176
+ if value is not None:
1177
+ _text(value, f"semantics.coverage.time_range.{name}", minimum=1, maximum=128)
1178
+ if not is_single_plain_line(value):
1179
+ raise ContractError(
1180
+ f"semantics.coverage.time_range.{name}",
1181
+ "NONCANONICAL_TEXT",
1182
+ "must be one plain line",
1183
+ )
1184
+ _text(self.population, "semantics.coverage.population", minimum=1, maximum=2_000)
1185
+ _text(
1186
+ self.completeness_note,
1187
+ "semantics.coverage.completeness_note",
1188
+ minimum=1,
1189
+ maximum=1_000,
1190
+ )
1191
+
1192
+ def to_dict(self) -> dict[str, Any]:
1193
+ return {
1194
+ "time_range": None
1195
+ if self.start_inclusive is None
1196
+ else {
1197
+ "start_inclusive": self.start_inclusive,
1198
+ "end_exclusive": self.end_exclusive,
1199
+ },
1200
+ "population": self.population,
1201
+ "completeness_note": self.completeness_note,
1202
+ }
1203
+
1204
+
1205
+ @dataclass(frozen=True)
1206
+ class ColumnSemantics:
1207
+ """Sealed human-readable meaning of exactly one selected physical column."""
1208
+
1209
+ name: str
1210
+ display_label: str
1211
+ description: str
1212
+ semantic_type: str
1213
+ unit: str
1214
+ unit_state: str
1215
+ normalization: str | None
1216
+ reversible: bool | None
1217
+ categories: tuple[CategoryDefinition, ...] | None
1218
+ categories_status: str
1219
+ evidence_refs: tuple[SemanticEvidenceRef, ...]
1220
+
1221
+ def __post_init__(self) -> None:
1222
+ _column_name(self.name, "semantics.columns.name")
1223
+ _text(self.display_label, "semantics.columns.display_label", minimum=1, maximum=120)
1224
+ if not is_single_plain_line(self.display_label):
1225
+ raise ContractError(
1226
+ "semantics.columns.display_label",
1227
+ "NONCANONICAL_TEXT",
1228
+ "must be one plain line",
1229
+ )
1230
+ _text(self.description, "semantics.columns.description", minimum=1, maximum=1_000)
1231
+ _choice(self.semantic_type, SEMANTIC_TYPES, "semantics.columns.semantic_type")
1232
+ resolved = _declared_unit(self.unit, "semantics.columns.unit")
1233
+ _choice(self.unit_state, UNIT_STATES, "semantics.columns.unit_state")
1234
+ if self.normalization is not None:
1235
+ _choice(
1236
+ self.normalization,
1237
+ NORMALIZATIONS,
1238
+ "semantics.columns.normalization",
1239
+ )
1240
+ if self.reversible is not None and type(self.reversible) is not bool:
1241
+ raise ContractError("semantics.columns.reversible", "TYPE", "must be a boolean or null")
1242
+
1243
+ if self.unit_state == "declared":
1244
+ if (
1245
+ self.unit == NO_UNIT
1246
+ or self.normalization is not None
1247
+ or self.reversible is not None
1248
+ ):
1249
+ raise ContractError(
1250
+ "semantics.columns.unit_state",
1251
+ "ENUM",
1252
+ "declared units require a physical unit and no normalization fields",
1253
+ )
1254
+ elif self.unit_state == "normalized":
1255
+ if (
1256
+ self.unit != NO_UNIT
1257
+ or self.normalization is None
1258
+ or type(self.reversible) is not bool
1259
+ ):
1260
+ raise ContractError(
1261
+ "semantics.columns.unit_state",
1262
+ "ENUM",
1263
+ "normalized values require unit none, a normalization, and reversible boolean",
1264
+ )
1265
+ elif self.unit != NO_UNIT or self.normalization is not None or self.reversible is not None:
1266
+ raise ContractError(
1267
+ "semantics.columns.unit_state",
1268
+ "ENUM",
1269
+ "not_applicable and unknown require unit none and no normalization fields",
1270
+ )
1271
+ if self.semantic_type == "measure" and self.unit_state == "not_applicable":
1272
+ raise ContractError(
1273
+ "semantics.columns.unit_state",
1274
+ "ENUM",
1275
+ "measure fields require a declared, normalized, or unknown unit state",
1276
+ )
1277
+ # The rule is about the unit rather than about the spelling of it. ``count`` and
1278
+ # ``{count}`` are one unit under two names, and a dimensionless ratio is a different one,
1279
+ # so the question asked here is which unit the code resolved to.
1280
+ if (
1281
+ self.semantic_type not in {"measure", "target"}
1282
+ and self.unit_state == "declared"
1283
+ and resolved is not None
1284
+ and not is_count_unit(resolved)
1285
+ ):
1286
+ raise ContractError(
1287
+ "semantics.columns.unit",
1288
+ "ENUM",
1289
+ "non-measure and non-target fields may declare only count units",
1290
+ )
1291
+
1292
+ _choice(
1293
+ self.categories_status,
1294
+ CATEGORY_STATUSES,
1295
+ "semantics.columns.categories_status",
1296
+ )
1297
+ if self.categories is not None:
1298
+ _typed_tuple(
1299
+ self.categories,
1300
+ CategoryDefinition,
1301
+ "semantics.columns.categories",
1302
+ nonempty=True,
1303
+ maximum=64,
1304
+ )
1305
+ _require_unique(
1306
+ (item.code for item in self.categories),
1307
+ "semantics.columns.categories.code",
1308
+ )
1309
+ if self.categories_status in {"complete", "partial"} and not self.categories:
1310
+ raise ContractError(
1311
+ "semantics.columns.categories",
1312
+ "ENUM",
1313
+ "complete or partial category status requires definitions",
1314
+ )
1315
+ if self.categories_status in {"not_applicable", "unknown"} and self.categories is not None:
1316
+ raise ContractError(
1317
+ "semantics.columns.categories",
1318
+ "ENUM",
1319
+ "not_applicable or unknown category status cannot carry definitions",
1320
+ )
1321
+ _typed_tuple(
1322
+ self.evidence_refs,
1323
+ SemanticEvidenceRef,
1324
+ "semantics.columns.evidence_refs",
1325
+ nonempty=True,
1326
+ maximum=8,
1327
+ )
1328
+ _require_unique(
1329
+ ((item.kind, item.ref) for item in self.evidence_refs),
1330
+ "semantics.columns.evidence_refs",
1331
+ )
1332
+
1333
+ def to_dict(self) -> dict[str, Any]:
1334
+ return {
1335
+ "name": self.name,
1336
+ "display_label": self.display_label,
1337
+ "description": self.description,
1338
+ "semantic_type": self.semantic_type,
1339
+ "unit": self.unit,
1340
+ "unit_state": self.unit_state,
1341
+ "normalization": self.normalization,
1342
+ "reversible": self.reversible,
1343
+ "categories": None
1344
+ if self.categories is None
1345
+ else [item.to_dict() for item in sorted(self.categories, key=lambda item: item.code)],
1346
+ "categories_status": self.categories_status,
1347
+ "evidence_refs": [
1348
+ item.to_dict()
1349
+ for item in sorted(self.evidence_refs, key=lambda item: (item.kind, item.ref))
1350
+ ],
1351
+ }
1352
+
1353
+
1354
+ @dataclass(frozen=True)
1355
+ class TableSemantics(_CanonicalContract):
1356
+ """One complete, digestable semantic document for a selected Table shape."""
1357
+
1358
+ summary: str
1359
+ grain_statement: str
1360
+ coverage: SemanticCoverage
1361
+ limitations: tuple[str, ...]
1362
+ columns: tuple[ColumnSemantics, ...]
1363
+ schema_version: str = TABLE_SEMANTICS_VERSION
1364
+
1365
+ EXPECTED_SCHEMA_VERSION: ClassVar[str] = TABLE_SEMANTICS_VERSION
1366
+
1367
+ def __post_init__(self) -> None:
1368
+ _choice_version(self.schema_version, TABLE_SEMANTICS_VERSIONS, "semantics.schema_version")
1369
+ _text(self.summary, "semantics.summary", minimum=1, maximum=1_000)
1370
+ _text(self.grain_statement, "semantics.grain_statement", minimum=1, maximum=400)
1371
+ if not isinstance(self.coverage, SemanticCoverage):
1372
+ raise ContractError("semantics.coverage", "TYPE", "must be SemanticCoverage")
1373
+ _text_tuple(
1374
+ self.limitations,
1375
+ "semantics.limitations",
1376
+ nonempty=False,
1377
+ maximum=32,
1378
+ item_maximum=1_000,
1379
+ )
1380
+ _typed_tuple(
1381
+ self.columns,
1382
+ ColumnSemantics,
1383
+ "semantics.columns",
1384
+ nonempty=True,
1385
+ maximum=MAX_COLUMN_COUNT,
1386
+ )
1387
+ _require_unique((item.name for item in self.columns), "semantics.columns.name")
1388
+ if self.schema_version == TABLE_SEMANTICS_V1:
1389
+ # The first generation's vocabulary is closed and stays closed. Widening it in place
1390
+ # would rewrite what a sealed document meant, which is the one thing a version is for.
1391
+ for index, column in enumerate(self.columns):
1392
+ if column.unit not in UNITS:
1393
+ raise ContractError(
1394
+ f"semantics.columns[{index}].unit",
1395
+ "UNIT_UNSUPPORTED_IN_VERSION",
1396
+ f"{TABLE_SEMANTICS_V1} carries only {sorted(UNITS)}; a unit code "
1397
+ f"requires {TABLE_SEMANTICS_V2}",
1398
+ )
1399
+
1400
+ def validate_for_select(self, select: tuple[str, ...]) -> None:
1401
+ if tuple(item.name for item in self.columns) != select:
1402
+ raise ContractError(
1403
+ "semantics.columns",
1404
+ "REFERENCE_MISMATCH",
1405
+ "must follow the exact selected-column order",
1406
+ )
1407
+
1408
+ def validate_for_sources(self, source_ids: tuple[str, ...]) -> None:
1409
+ """Bind every semantic citation to a source in the sealed plan/Recipe."""
1410
+
1411
+ allowed = frozenset(source_ids)
1412
+ for column in self.columns:
1413
+ for evidence in column.evidence_refs:
1414
+ if evidence.ref not in allowed:
1415
+ raise ContractError(
1416
+ "semantics.columns.evidence_refs.ref",
1417
+ "REFERENCE_MISMATCH",
1418
+ "must reference a source sealed in the same plan or Recipe",
1419
+ )
1420
+
1421
+ def to_dict(self) -> dict[str, Any]:
1422
+ return {
1423
+ "schema_version": self.schema_version,
1424
+ "summary": self.summary,
1425
+ "grain_statement": self.grain_statement,
1426
+ "coverage": self.coverage.to_dict(),
1427
+ "limitations": list(self.limitations),
1428
+ "columns": [item.to_dict() for item in self.columns],
1429
+ }
1430
+
1431
+
1432
+ @dataclass(frozen=True)
1433
+ class TablePlan(_CanonicalContract):
1434
+ """The complete plan a build executes.
1435
+
1436
+ Names the question, the output intent, the output grain, at least two sources, the
1437
+ cleaning steps, exactly one join, the selected output columns, and the quality gates.
1438
+ The output intent is one of ``local_use``, ``redistribute``, or ``host``, and it gates
1439
+ the rights check: a build refuses unless every source permits that intent. The plan is
1440
+ closed: a build derives every derived member from this value and the raw source bytes
1441
+ alone, which is what lets verification replay the derivation and compare bytes.
1442
+ """
1443
+
1444
+ question: str
1445
+ output_intent: str
1446
+ grain: tuple[str, ...]
1447
+ sources: tuple[SourceSpec, ...]
1448
+ cleaning: tuple[CleaningStep, ...]
1449
+ join: JoinSpec
1450
+ select: tuple[str, ...]
1451
+ quality: QualitySpec
1452
+ schema_version: str = PLAN_SCHEMA_VERSION
1453
+
1454
+ EXPECTED_SCHEMA_VERSION: ClassVar[str] = PLAN_SCHEMA_VERSION
1455
+
1456
+ def __post_init__(self) -> None:
1457
+ _version(self.schema_version, PLAN_SCHEMA_VERSION, "plan.schema_version")
1458
+ _text(self.question, "plan.question", minimum=1, maximum=MAX_TEXT_LENGTH)
1459
+ _choice(self.output_intent, _OUTPUT_INTENTS, "plan.output_intent")
1460
+ _column_tuple(self.grain, "plan.grain", nonempty=True, maximum=16)
1461
+ _typed_tuple(
1462
+ self.sources,
1463
+ SourceSpec,
1464
+ "plan.sources",
1465
+ nonempty=True,
1466
+ minimum=2,
1467
+ maximum=MAX_SOURCE_COUNT,
1468
+ )
1469
+ source_ids = tuple(source.source_id for source in self.sources)
1470
+ _require_unique(source_ids, "plan.sources.id")
1471
+ _typed_tuple(
1472
+ self.cleaning,
1473
+ CleaningStep,
1474
+ "plan.cleaning",
1475
+ nonempty=False,
1476
+ maximum=MAX_OPERATION_COUNT,
1477
+ )
1478
+ unknown_cleaning_sources = sorted(
1479
+ {step.source_id for step in self.cleaning} - set(source_ids)
1480
+ )
1481
+ if unknown_cleaning_sources:
1482
+ raise ContractError(
1483
+ "plan.cleaning",
1484
+ "UNKNOWN_SOURCE",
1485
+ f"references unknown sources {unknown_cleaning_sources}",
1486
+ )
1487
+ if not isinstance(self.join, JoinSpec):
1488
+ raise ContractError("plan.join", "TYPE", "must be a JoinSpec")
1489
+ if self.join.left not in source_ids:
1490
+ raise ContractError("plan.join.left", "UNKNOWN_SOURCE", "is not in plan.sources")
1491
+ if self.join.right not in source_ids:
1492
+ raise ContractError("plan.join.right", "UNKNOWN_SOURCE", "is not in plan.sources")
1493
+ _column_tuple(
1494
+ self.select,
1495
+ "plan.select",
1496
+ nonempty=True,
1497
+ maximum=MAX_COLUMN_COUNT,
1498
+ )
1499
+ missing_grain = sorted(set(self.grain) - set(self.select))
1500
+ if missing_grain:
1501
+ raise ContractError(
1502
+ "plan.grain",
1503
+ "GRAIN_NOT_SELECTED",
1504
+ f"contains columns absent from plan.select: {missing_grain}",
1505
+ )
1506
+ if not isinstance(self.quality, QualitySpec):
1507
+ raise ContractError("plan.quality", "TYPE", "must be a QualitySpec")
1508
+ _bounded_int(
1509
+ self.quality.min_rows,
1510
+ "plan.quality.min_rows",
1511
+ minimum=1,
1512
+ maximum=MAX_OUTPUT_ROWS,
1513
+ )
1514
+ if not isinstance(self.quality, QualitySpec):
1515
+ raise ContractError("plan.quality", "TYPE", "must be a QualitySpec")
1516
+ missing_quality = sorted(set(self.quality.not_null) - set(self.select))
1517
+ if missing_quality:
1518
+ raise ContractError(
1519
+ "plan.quality.not_null",
1520
+ "QUALITY_COLUMN_NOT_SELECTED",
1521
+ f"contains columns absent from plan.select: {missing_quality}",
1522
+ )
1523
+
1524
+ def to_dict(self) -> dict[str, Any]:
1525
+ return {
1526
+ "schema_version": self.schema_version,
1527
+ "question": self.question,
1528
+ "output_intent": self.output_intent,
1529
+ "grain": list(self.grain),
1530
+ "sources": [source.to_dict() for source in self.sources],
1531
+ "cleaning": [step.to_dict() for step in self.cleaning],
1532
+ "join": self.join.to_dict(),
1533
+ "select": list(self.select),
1534
+ "quality": self.quality.to_dict(),
1535
+ }
1536
+
1537
+ @classmethod
1538
+ def from_value(cls, value: Any) -> TablePlan:
1539
+ return parse_table_plan(value)
1540
+
1541
+
1542
+ @dataclass(frozen=True)
1543
+ class GraphNode:
1544
+ """One ordered node in a closed table-plan graph."""
1545
+
1546
+ node_id: str
1547
+ operation: str
1548
+ operation_version: str
1549
+ inputs: tuple[str, ...]
1550
+ parameters: tuple[tuple[str, Any], ...]
1551
+
1552
+ def __post_init__(self) -> None:
1553
+ _identifier(self.node_id, "plan.nodes.id")
1554
+ _text(self.operation, "plan.nodes.operation", minimum=1, maximum=64)
1555
+ _text(self.operation_version, "plan.nodes.operation_version", minimum=1, maximum=32)
1556
+ _text_tuple(
1557
+ self.inputs,
1558
+ "plan.nodes.inputs",
1559
+ nonempty=False,
1560
+ maximum=MAX_GRAPH_NODE_COUNT,
1561
+ item_maximum=MAX_IDENTIFIER_LENGTH,
1562
+ )
1563
+ for index, value in enumerate(self.inputs):
1564
+ _identifier(value, f"plan.nodes.inputs[{index}]")
1565
+ if not isinstance(self.parameters, tuple) or any(
1566
+ not isinstance(item, tuple) or len(item) != 2 or not isinstance(item[0], str)
1567
+ for item in self.parameters
1568
+ ):
1569
+ raise ContractError("plan.nodes.parameters", "TYPE", "must be frozen parameter pairs")
1570
+
1571
+ def parameter(self, name: str) -> Any:
1572
+ return dict(self.parameters)[name]
1573
+
1574
+ def to_dict(self) -> dict[str, Any]:
1575
+ return {
1576
+ "id": self.node_id,
1577
+ "operation": self.operation,
1578
+ "operation_version": self.operation_version,
1579
+ "inputs": list(self.inputs),
1580
+ "parameters": thaw_parameter(self.parameters),
1581
+ }
1582
+
1583
+
1584
+ @dataclass(frozen=True)
1585
+ class GraphTablePlan(_CanonicalContract):
1586
+ """A versioned ordered DAG with one declared terminal Table node."""
1587
+
1588
+ operation_contract_version: str
1589
+ question: str
1590
+ output_intent: str
1591
+ grain: tuple[str, ...]
1592
+ sources: tuple[SourceSpec, ...]
1593
+ nodes: tuple[GraphNode, ...]
1594
+ terminal: str
1595
+ output_columns: tuple[str, ...]
1596
+ quality: QualitySpec
1597
+ schema_version: str = GRAPH_PLAN_SCHEMA_VERSION
1598
+
1599
+ EXPECTED_SCHEMA_VERSION: ClassVar[str] = GRAPH_PLAN_SCHEMA_VERSION
1600
+
1601
+ def __post_init__(self) -> None:
1602
+ _version(self.schema_version, GRAPH_PLAN_SCHEMA_VERSION, "plan.schema_version")
1603
+ _version(
1604
+ self.operation_contract_version,
1605
+ GRAPH_OPERATION_CONTRACT_VERSION,
1606
+ "plan.operation_contract_version",
1607
+ )
1608
+ _text(self.question, "plan.question", minimum=1, maximum=MAX_TEXT_LENGTH)
1609
+ _choice(self.output_intent, _OUTPUT_INTENTS, "plan.output_intent")
1610
+ _column_tuple(self.grain, "plan.grain", nonempty=True, maximum=16)
1611
+ _typed_tuple(
1612
+ self.sources,
1613
+ SourceSpec,
1614
+ "plan.sources",
1615
+ nonempty=True,
1616
+ maximum=MAX_SOURCE_COUNT,
1617
+ )
1618
+ source_ids = tuple(source.source_id for source in self.sources)
1619
+ _require_unique(source_ids, "plan.sources.id")
1620
+ _typed_tuple(
1621
+ self.nodes,
1622
+ GraphNode,
1623
+ "plan.nodes",
1624
+ nonempty=True,
1625
+ maximum=MAX_GRAPH_NODE_COUNT,
1626
+ )
1627
+ node_ids = tuple(node.node_id for node in self.nodes)
1628
+ if len(set(node_ids)) != len(node_ids):
1629
+ raise ContractError("plan.nodes.id", "GRAPH_NODE_DUPLICATE", "must be unique")
1630
+ _identifier(self.terminal, "plan.terminal")
1631
+ if self.terminal not in set(node_ids):
1632
+ raise ContractError("plan.terminal", "GRAPH_TERMINAL_UNKNOWN", "is not a graph node")
1633
+ _column_tuple(
1634
+ self.output_columns,
1635
+ "plan.output_columns",
1636
+ nonempty=True,
1637
+ maximum=MAX_COLUMN_COUNT,
1638
+ )
1639
+ missing_grain = sorted(set(self.grain) - set(self.output_columns))
1640
+ if missing_grain:
1641
+ raise ContractError(
1642
+ "plan.grain",
1643
+ "GRAIN_NOT_SELECTED",
1644
+ f"contains columns absent from plan.output_columns: {missing_grain}",
1645
+ )
1646
+ if not isinstance(self.quality, QualitySpec):
1647
+ raise ContractError("plan.quality", "TYPE", "must be a QualitySpec")
1648
+ missing_quality = sorted(set(self.quality.not_null) - set(self.output_columns))
1649
+ if missing_quality:
1650
+ raise ContractError(
1651
+ "plan.quality.not_null",
1652
+ "QUALITY_COLUMN_NOT_SELECTED",
1653
+ f"contains columns absent from plan.output_columns: {missing_quality}",
1654
+ )
1655
+ self._validate_graph(source_ids)
1656
+
1657
+ @property
1658
+ def select(self) -> tuple[str, ...]:
1659
+ """Compatibility name used by Recipe output-semantic validation."""
1660
+
1661
+ return self.output_columns
1662
+
1663
+ def _validate_graph(self, source_ids: tuple[str, ...]) -> None:
1664
+ seen: set[str] = set()
1665
+ source_bindings: list[str] = []
1666
+ node_by_id = {node.node_id: node for node in self.nodes}
1667
+ for index, node in enumerate(self.nodes):
1668
+ try:
1669
+ operation = resolve_operation(node.operation, node.operation_version)
1670
+ except OperationRegistryError as exc:
1671
+ raise ContractError(
1672
+ f"plan.nodes[{index}].operation", exc.code, exc.detail
1673
+ ) from None
1674
+ minimum, maximum = operation.input_arity
1675
+ if not minimum <= len(node.inputs) <= maximum:
1676
+ raise ContractError(
1677
+ f"plan.nodes[{index}].inputs",
1678
+ "OPERATION_ARITY",
1679
+ f"requires {minimum}..{maximum} inputs",
1680
+ )
1681
+ unknown = [input_id for input_id in node.inputs if input_id not in seen]
1682
+ if unknown:
1683
+ raise ContractError(
1684
+ f"plan.nodes[{index}].inputs",
1685
+ "GRAPH_INPUT_ORDER",
1686
+ f"must reference earlier nodes; unavailable={unknown}",
1687
+ )
1688
+ if node.operation == "source":
1689
+ source_id = node.parameter("source")
1690
+ if not isinstance(source_id, str) or source_id not in set(source_ids):
1691
+ raise ContractError(
1692
+ f"plan.nodes[{index}].parameters.source",
1693
+ "GRAPH_SOURCE_UNKNOWN",
1694
+ "must bind one declared source",
1695
+ )
1696
+ source_bindings.append(source_id)
1697
+ if node.operation == "prediction_label":
1698
+ if node.node_id != self.terminal:
1699
+ raise ContractError(
1700
+ f"plan.nodes[{index}]",
1701
+ "PREDICTION_LABEL_POSITION",
1702
+ "prediction_label must be the terminal node so its future value cannot "
1703
+ "feed another transform",
1704
+ )
1705
+ output_column = node.parameter("output_column")
1706
+ if output_column not in self.output_columns:
1707
+ raise ContractError(
1708
+ f"plan.nodes[{index}].parameters.output_column",
1709
+ "PREDICTION_LABEL_OUTPUT",
1710
+ "must be selected as an output column",
1711
+ )
1712
+ seen.add(node.node_id)
1713
+ if len(set(source_bindings)) != len(source_bindings):
1714
+ raise ContractError(
1715
+ "plan.nodes.parameters.source",
1716
+ "GRAPH_SOURCE_DUPLICATE",
1717
+ "a declared source may have only one source node",
1718
+ )
1719
+ if set(source_bindings) != set(source_ids):
1720
+ raise ContractError(
1721
+ "plan.nodes.parameters.source",
1722
+ "GRAPH_SOURCE_SET",
1723
+ "source nodes must bind every declared source exactly once",
1724
+ )
1725
+ reachable: set[str] = set()
1726
+
1727
+ def visit(node_id: str) -> None:
1728
+ if node_id in reachable:
1729
+ return
1730
+ reachable.add(node_id)
1731
+ for input_id in node_by_id[node_id].inputs:
1732
+ visit(input_id)
1733
+
1734
+ visit(self.terminal)
1735
+ unreachable = sorted(set(node_by_id) - reachable)
1736
+ if unreachable:
1737
+ raise ContractError(
1738
+ "plan.nodes",
1739
+ "GRAPH_NODE_UNREACHABLE",
1740
+ f"nodes are outside terminal ancestry: {unreachable}",
1741
+ )
1742
+
1743
+ def to_dict(self) -> dict[str, Any]:
1744
+ return {
1745
+ "schema_version": self.schema_version,
1746
+ "operation_contract_version": self.operation_contract_version,
1747
+ "question": self.question,
1748
+ "output_intent": self.output_intent,
1749
+ "grain": list(self.grain),
1750
+ "sources": [source.to_dict() for source in self.sources],
1751
+ "nodes": [node.to_dict() for node in self.nodes],
1752
+ "terminal": self.terminal,
1753
+ "output_columns": list(self.output_columns),
1754
+ "quality": self.quality.to_dict(),
1755
+ }
1756
+
1757
+ @classmethod
1758
+ def from_value(cls, value: Any) -> GraphTablePlan:
1759
+ return parse_graph_table_plan(value)
1760
+
1761
+
1762
+ def is_single_plain_line(value: object) -> bool:
1763
+ """Return whether text contains no boundary that renders as another display line.
1764
+
1765
+ This deliberately answers only the structural question. A tab remains part of one line, and
1766
+ U+009B and bidi controls remain outside this rule pending an explicit terminal-safety policy.
1767
+ """
1768
+
1769
+ return isinstance(value, str) and not any(
1770
+ boundary in value for boundary in _DISPLAY_LINE_BOUNDARIES
1771
+ )
1772
+
1773
+
1774
+ def parse_question(value: Any) -> LocalQuestion:
1775
+ return LocalQuestion.from_value(value)
1776
+
1777
+
1778
+ def parse_requirements(value: Any) -> LocalRequirements:
1779
+ return LocalRequirements.from_value(value)
1780
+
1781
+
1782
+ def parse_source_proposal(value: Any) -> SourceProposal:
1783
+ return SourceProposal.from_value(value)
1784
+
1785
+
1786
+ def parse_table_semantics(value: Any) -> TableSemantics:
1787
+ """Parse the closed human-readable semantics document without inferring any claim."""
1788
+
1789
+ path = "semantics"
1790
+ data = _object(value, path)
1791
+ _exact_fields(
1792
+ data,
1793
+ {"schema_version", "summary", "grain_statement", "coverage", "limitations", "columns"},
1794
+ path,
1795
+ )
1796
+ coverage_data = _object(data["coverage"], f"{path}.coverage")
1797
+ _exact_fields(
1798
+ coverage_data,
1799
+ {"time_range", "population", "completeness_note"},
1800
+ f"{path}.coverage",
1801
+ )
1802
+ time_range = coverage_data["time_range"]
1803
+ start_inclusive: str | None = None
1804
+ end_exclusive: str | None = None
1805
+ if time_range is not None:
1806
+ time_data = _object(time_range, f"{path}.coverage.time_range")
1807
+ _exact_fields(
1808
+ time_data,
1809
+ {"start_inclusive", "end_exclusive"},
1810
+ f"{path}.coverage.time_range",
1811
+ )
1812
+ start_inclusive = _required_text(
1813
+ time_data["start_inclusive"],
1814
+ f"{path}.coverage.time_range.start_inclusive",
1815
+ maximum=128,
1816
+ )
1817
+ end_exclusive = _required_text(
1818
+ time_data["end_exclusive"],
1819
+ f"{path}.coverage.time_range.end_exclusive",
1820
+ maximum=128,
1821
+ )
1822
+
1823
+ columns: list[ColumnSemantics] = []
1824
+ for index, raw_column in enumerate(
1825
+ _array(data["columns"], f"{path}.columns", nonempty=True, maximum=MAX_COLUMN_COUNT)
1826
+ ):
1827
+ column_path = f"{path}.columns[{index}]"
1828
+ column = _object(raw_column, column_path)
1829
+ _exact_fields(
1830
+ column,
1831
+ {
1832
+ "name",
1833
+ "display_label",
1834
+ "description",
1835
+ "semantic_type",
1836
+ "unit",
1837
+ "unit_state",
1838
+ "normalization",
1839
+ "reversible",
1840
+ "categories",
1841
+ "categories_status",
1842
+ "evidence_refs",
1843
+ },
1844
+ column_path,
1845
+ )
1846
+ categories: tuple[CategoryDefinition, ...] | None = None
1847
+ if column["categories"] is not None:
1848
+ parsed_categories: list[CategoryDefinition] = []
1849
+ for category_index, raw_category in enumerate(
1850
+ _array(
1851
+ column["categories"],
1852
+ f"{column_path}.categories",
1853
+ nonempty=True,
1854
+ maximum=64,
1855
+ )
1856
+ ):
1857
+ category_path = f"{column_path}.categories[{category_index}]"
1858
+ category = _object(raw_category, category_path)
1859
+ _exact_fields(category, {"code", "label"}, category_path)
1860
+ parsed_categories.append(
1861
+ CategoryDefinition(
1862
+ code=_required_text(category["code"], f"{category_path}.code", maximum=64),
1863
+ label=_required_text(
1864
+ category["label"], f"{category_path}.label", maximum=120
1865
+ ),
1866
+ )
1867
+ )
1868
+ categories = tuple(parsed_categories)
1869
+
1870
+ evidence_refs: list[SemanticEvidenceRef] = []
1871
+ for evidence_index, raw_evidence in enumerate(
1872
+ _array(
1873
+ column["evidence_refs"],
1874
+ f"{column_path}.evidence_refs",
1875
+ nonempty=False,
1876
+ maximum=8,
1877
+ )
1878
+ ):
1879
+ evidence_path = f"{column_path}.evidence_refs[{evidence_index}]"
1880
+ evidence = _object(raw_evidence, evidence_path)
1881
+ _exact_fields(evidence, {"kind", "ref"}, evidence_path)
1882
+ evidence_refs.append(
1883
+ SemanticEvidenceRef(
1884
+ kind=_required_text(evidence["kind"], f"{evidence_path}.kind", maximum=64),
1885
+ ref=_required_text(evidence["ref"], f"{evidence_path}.ref", maximum=128),
1886
+ )
1887
+ )
1888
+
1889
+ reversible = column["reversible"]
1890
+ if reversible is not None and type(reversible) is not bool:
1891
+ raise ContractError(f"{column_path}.reversible", "TYPE", "must be a boolean or null")
1892
+ normalization = column["normalization"]
1893
+ if normalization is not None and not isinstance(normalization, str):
1894
+ raise ContractError(f"{column_path}.normalization", "TYPE", "must be a string or null")
1895
+ columns.append(
1896
+ ColumnSemantics(
1897
+ name=_required_text(column["name"], f"{column_path}.name", maximum=64),
1898
+ display_label=_required_text(
1899
+ column["display_label"], f"{column_path}.display_label", maximum=120
1900
+ ),
1901
+ description=_required_text(
1902
+ column["description"], f"{column_path}.description", maximum=1_000
1903
+ ),
1904
+ semantic_type=_required_text(
1905
+ column["semantic_type"], f"{column_path}.semantic_type", maximum=64
1906
+ ),
1907
+ unit=_required_text(column["unit"], f"{column_path}.unit", maximum=64),
1908
+ unit_state=_required_text(
1909
+ column["unit_state"], f"{column_path}.unit_state", maximum=64
1910
+ ),
1911
+ normalization=normalization,
1912
+ reversible=reversible,
1913
+ categories=categories,
1914
+ categories_status=_required_text(
1915
+ column["categories_status"],
1916
+ f"{column_path}.categories_status",
1917
+ maximum=64,
1918
+ ),
1919
+ evidence_refs=tuple(evidence_refs),
1920
+ )
1921
+ )
1922
+
1923
+ return TableSemantics(
1924
+ schema_version=_required_text(data["schema_version"], f"{path}.schema_version", maximum=64),
1925
+ summary=_required_text(data["summary"], f"{path}.summary", maximum=1_000),
1926
+ grain_statement=_required_text(
1927
+ data["grain_statement"], f"{path}.grain_statement", maximum=400
1928
+ ),
1929
+ coverage=SemanticCoverage(
1930
+ start_inclusive=start_inclusive,
1931
+ end_exclusive=end_exclusive,
1932
+ population=_required_text(
1933
+ coverage_data["population"], f"{path}.coverage.population", maximum=2_000
1934
+ ),
1935
+ completeness_note=_required_text(
1936
+ coverage_data["completeness_note"],
1937
+ f"{path}.coverage.completeness_note",
1938
+ maximum=1_000,
1939
+ ),
1940
+ ),
1941
+ limitations=_parse_text_array(
1942
+ data["limitations"],
1943
+ f"{path}.limitations",
1944
+ nonempty=False,
1945
+ maximum=32,
1946
+ item_maximum=1_000,
1947
+ ),
1948
+ columns=tuple(columns),
1949
+ )
1950
+
1951
+
1952
+ def parse_table_plan(value: Any) -> TablePlan:
1953
+ """Parse an untrusted JSON-compatible value into the closed execution plan."""
1954
+
1955
+ path = "plan"
1956
+ data = _object(value, path)
1957
+ _exact_fields(
1958
+ data,
1959
+ {
1960
+ "schema_version",
1961
+ "question",
1962
+ "output_intent",
1963
+ "grain",
1964
+ "sources",
1965
+ "cleaning",
1966
+ "join",
1967
+ "select",
1968
+ "quality",
1969
+ },
1970
+ path,
1971
+ )
1972
+ sources = tuple(
1973
+ _parse_source(item, f"{path}.sources[{index}]")
1974
+ for index, item in enumerate(
1975
+ _array(
1976
+ data["sources"],
1977
+ f"{path}.sources",
1978
+ nonempty=True,
1979
+ minimum=2,
1980
+ maximum=MAX_SOURCE_COUNT,
1981
+ )
1982
+ )
1983
+ )
1984
+ _require_unique((source.source_id for source in sources), f"{path}.sources.id")
1985
+ source_ids = frozenset(source.source_id for source in sources)
1986
+ cleaning = tuple(
1987
+ _parse_cleaning(item, f"{path}.cleaning[{index}]", source_ids)
1988
+ for index, item in enumerate(
1989
+ _array(
1990
+ data["cleaning"],
1991
+ f"{path}.cleaning",
1992
+ nonempty=False,
1993
+ maximum=MAX_OPERATION_COUNT,
1994
+ )
1995
+ )
1996
+ )
1997
+ return TablePlan(
1998
+ schema_version=_required_text(
1999
+ data["schema_version"],
2000
+ f"{path}.schema_version",
2001
+ maximum=64,
2002
+ ),
2003
+ question=_required_text(
2004
+ data["question"],
2005
+ f"{path}.question",
2006
+ maximum=MAX_TEXT_LENGTH,
2007
+ ),
2008
+ output_intent=_required_text(
2009
+ data["output_intent"],
2010
+ f"{path}.output_intent",
2011
+ maximum=64,
2012
+ ),
2013
+ grain=_parse_column_array(
2014
+ data["grain"],
2015
+ f"{path}.grain",
2016
+ nonempty=True,
2017
+ maximum=16,
2018
+ ),
2019
+ sources=sources,
2020
+ cleaning=cleaning,
2021
+ join=_parse_join(data["join"], f"{path}.join", source_ids),
2022
+ select=_parse_column_array(
2023
+ data["select"],
2024
+ f"{path}.select",
2025
+ nonempty=True,
2026
+ maximum=MAX_COLUMN_COUNT,
2027
+ ),
2028
+ quality=_parse_quality(data["quality"], f"{path}.quality"),
2029
+ )
2030
+
2031
+
2032
+ def parse_graph_table_plan(value: Any) -> GraphTablePlan:
2033
+ """Parse an untrusted v2 graph without executing any node."""
2034
+
2035
+ path = "plan"
2036
+ data = _object(value, path)
2037
+ _exact_fields(
2038
+ data,
2039
+ {
2040
+ "schema_version",
2041
+ "operation_contract_version",
2042
+ "question",
2043
+ "output_intent",
2044
+ "grain",
2045
+ "sources",
2046
+ "nodes",
2047
+ "terminal",
2048
+ "output_columns",
2049
+ "quality",
2050
+ },
2051
+ path,
2052
+ )
2053
+ operation_contract = _required_text(
2054
+ data["operation_contract_version"],
2055
+ f"{path}.operation_contract_version",
2056
+ maximum=64,
2057
+ )
2058
+ if operation_contract != GRAPH_OPERATION_CONTRACT_VERSION:
2059
+ raise ContractError(
2060
+ f"{path}.operation_contract_version",
2061
+ "PLAN_OPERATION_PAIR",
2062
+ f"{GRAPH_PLAN_SCHEMA_VERSION} requires {GRAPH_OPERATION_CONTRACT_VERSION}",
2063
+ )
2064
+ sources = tuple(
2065
+ _parse_source(item, f"{path}.sources[{index}]")
2066
+ for index, item in enumerate(
2067
+ _array(data["sources"], f"{path}.sources", nonempty=True, maximum=MAX_SOURCE_COUNT)
2068
+ )
2069
+ )
2070
+ nodes: list[GraphNode] = []
2071
+ for index, item in enumerate(
2072
+ _array(data["nodes"], f"{path}.nodes", nonempty=True, maximum=MAX_GRAPH_NODE_COUNT)
2073
+ ):
2074
+ node_path = f"{path}.nodes[{index}]"
2075
+ node_data = _object(item, node_path)
2076
+ _exact_fields(
2077
+ node_data,
2078
+ {"id", "operation", "operation_version", "inputs", "parameters"},
2079
+ node_path,
2080
+ )
2081
+ operation_name = _required_text(
2082
+ node_data["operation"], f"{node_path}.operation", maximum=64
2083
+ )
2084
+ operation_version = _required_text(
2085
+ node_data["operation_version"], f"{node_path}.operation_version", maximum=32
2086
+ )
2087
+ try:
2088
+ operation = resolve_operation(operation_name, operation_version)
2089
+ parameters = operation.validate_parameters(
2090
+ node_data["parameters"], path=f"{node_path}.parameters"
2091
+ )
2092
+ except OperationRegistryError as exc:
2093
+ raise ContractError(exc.path, exc.code, exc.detail) from None
2094
+ nodes.append(
2095
+ GraphNode(
2096
+ node_id=_required_text(node_data["id"], f"{node_path}.id", maximum=64),
2097
+ operation=operation_name,
2098
+ operation_version=operation_version,
2099
+ inputs=tuple(
2100
+ _required_text(value, f"{node_path}.inputs[{input_index}]", maximum=64)
2101
+ for input_index, value in enumerate(
2102
+ _array(
2103
+ node_data["inputs"],
2104
+ f"{node_path}.inputs",
2105
+ nonempty=False,
2106
+ maximum=MAX_GRAPH_NODE_COUNT,
2107
+ )
2108
+ )
2109
+ ),
2110
+ parameters=parameters,
2111
+ )
2112
+ )
2113
+ return GraphTablePlan(
2114
+ schema_version=_required_text(data["schema_version"], f"{path}.schema_version", maximum=64),
2115
+ operation_contract_version=operation_contract,
2116
+ question=_required_text(data["question"], f"{path}.question", maximum=MAX_TEXT_LENGTH),
2117
+ output_intent=_required_text(data["output_intent"], f"{path}.output_intent", maximum=64),
2118
+ grain=_parse_column_array(data["grain"], f"{path}.grain", nonempty=True, maximum=16),
2119
+ sources=sources,
2120
+ nodes=tuple(nodes),
2121
+ terminal=_required_text(data["terminal"], f"{path}.terminal", maximum=64),
2122
+ output_columns=_parse_column_array(
2123
+ data["output_columns"],
2124
+ f"{path}.output_columns",
2125
+ nonempty=True,
2126
+ maximum=MAX_COLUMN_COUNT,
2127
+ ),
2128
+ quality=_parse_quality(data["quality"], f"{path}.quality"),
2129
+ )
2130
+
2131
+
2132
+ def parse_plan_document(value: Any) -> TablePlan | GraphTablePlan:
2133
+ """Dispatch exactly one accepted plan generation by its schema version.
2134
+
2135
+ This is the reader for a plan of any age, including one read back out of a sealed Build, so it
2136
+ understands every operation the registry still holds. :func:`parse_fresh_plan_document` is the
2137
+ entry point for newly authored bytes.
2138
+ """
2139
+
2140
+ data = _object(value, "plan")
2141
+ schema_version = data.get("schema_version")
2142
+ if schema_version == PLAN_SCHEMA_VERSION:
2143
+ return parse_table_plan(value)
2144
+ if schema_version == GRAPH_PLAN_SCHEMA_VERSION:
2145
+ return parse_graph_table_plan(value)
2146
+ if schema_version == _LEGACY_PLAN_SCHEMA_VERSION:
2147
+ detail = (
2148
+ f"found {schema_version!r}, supported {PLAN_SCHEMA_VERSION!r} (renamed in V3); "
2149
+ f"set plan.schema_version to {PLAN_SCHEMA_VERSION!r} and retry"
2150
+ )
2151
+ else:
2152
+ detail = (
2153
+ f"found {schema_version!r}, supported one of "
2154
+ f"{[PLAN_SCHEMA_VERSION, GRAPH_PLAN_SCHEMA_VERSION]}"
2155
+ )
2156
+ raise ContractError("plan.schema_version", "VERSION", detail)
2157
+
2158
+
2159
+ def refuse_retired_operations(plan: TablePlan | GraphTablePlan) -> None:
2160
+ """Refuse a plan that names an operation the graph vocabulary has retired.
2161
+
2162
+ A retired operation is one the admission contract in ``docs/TRANSFORMS.md`` no longer admits.
2163
+ It stays in the registry because a Build sealed with it, and a Recipe approved with it, both
2164
+ re-parse their stored plan through that table to verify -- dropping the entry would make
2165
+ already-sealed bytes unreadable, which is a worse thing than the operation continuing to
2166
+ exist. So the line is drawn by age of the bytes rather than by the table: this is called on
2167
+ newly authored bytes and on nothing else, exactly as
2168
+ :func:`mostlyright.data_harness.recipe.parse_fresh_recipe` calls
2169
+ ``require_resolvable_reader`` for a Reader pin whose implementation has since been retired.
2170
+ """
2171
+
2172
+ if not isinstance(plan, GraphTablePlan):
2173
+ return
2174
+ for index, node in enumerate(plan.nodes):
2175
+ if node.operation in RETIRED_OPERATIONS:
2176
+ raise ContractError(
2177
+ f"plan.nodes[{index}].operation",
2178
+ "OPERATION_RETIRED",
2179
+ f"operation {node.operation!r} is retired and cannot be named by a new plan",
2180
+ )
2181
+
2182
+
2183
+ def refuse_retired_operation_names(value: Any) -> None:
2184
+ """Say "retired" about a node before saying anything else about it.
2185
+
2186
+ The structural rules a retired operation still carries -- where it may sit, what it must
2187
+ select -- fire during the parse, ahead of the retirement check that runs after it. Their advice
2188
+ is to correct the offending field and run the command again, which for a node that cannot be in
2189
+ a new plan at all sends the author round a loop that ends here anyway, and the author is
2190
+ frequently an agent that will act on it. So the retirement is read off the raw document first.
2191
+
2192
+ Anything that is not a well-formed node list is left alone, so a malformed document still gets
2193
+ the ordinary parse error that describes its actual shape rather than this one.
2194
+ """
2195
+
2196
+ if not isinstance(value, Mapping):
2197
+ return
2198
+ nodes = value.get("nodes")
2199
+ if not isinstance(nodes, list):
2200
+ return
2201
+ for index, node in enumerate(nodes):
2202
+ if not isinstance(node, Mapping):
2203
+ continue
2204
+ operation = node.get("operation")
2205
+ if isinstance(operation, str) and operation in RETIRED_OPERATIONS:
2206
+ raise ContractError(
2207
+ f"plan.nodes[{index}].operation",
2208
+ "OPERATION_RETIRED",
2209
+ f"operation {operation!r} is retired and cannot be named by a new plan",
2210
+ )
2211
+
2212
+
2213
+ def refuse_inconsistent_units(plan: TablePlan | GraphTablePlan) -> None:
2214
+ """Refuse a plan whose declared units contradict each other or the arithmetic on them.
2215
+
2216
+ This is on the authoring path and on nothing else, for the same reason the retirement check is.
2217
+ The analysis reads a declaration that did not exist before it, so nothing already sealed can
2218
+ trip it today -- but it is an analysis rather than a table, and a later pass that sees more will
2219
+ refuse more. Running it where sealed bytes are read would make that later pass retroactive:
2220
+ a plan sealed under this version would stop verifying under the next one, which is the outcome
2221
+ the age-of-the-bytes line exists to prevent.
2222
+ """
2223
+
2224
+ if not isinstance(plan, GraphTablePlan):
2225
+ return
2226
+ from mostlyright.data_harness.unit_flow import UnitFlowError
2227
+ from mostlyright.data_harness.unit_flow import refuse_inconsistent_units as _refuse
2228
+
2229
+ try:
2230
+ _refuse(plan)
2231
+ except UnitFlowError as exc:
2232
+ raise ContractError(exc.path, exc.code, exc.detail) from None
2233
+
2234
+
2235
+ def parse_fresh_plan_document(value: Any) -> TablePlan | GraphTablePlan:
2236
+ """Parse a plan entering the authoring lifecycle, and hold it to the checks age decides.
2237
+
2238
+ Three checks, in this order. The first reads retired names off the raw document so retirement
2239
+ is what the author is told; the second holds for a caller that passes an already-parsed plan
2240
+ through :func:`refuse_retired_operations` instead; the third is the two unit gates, which read
2241
+ a declaration and the arithmetic on it. All three are refusals for newly authored bytes and for
2242
+ nothing else.
2243
+ """
2244
+
2245
+ refuse_retired_operation_names(value)
2246
+ plan = parse_plan_document(value)
2247
+ refuse_retired_operations(plan)
2248
+ refuse_inconsistent_units(plan)
2249
+ return plan
2250
+
2251
+
2252
+ def _parse_source(value: Any, path: str) -> SourceSpec:
2253
+ data = _object(value, path)
2254
+ _exact_fields(
2255
+ data,
2256
+ {"id", "path", "format", "origin", "acquisition_method", "rights"},
2257
+ path,
2258
+ )
2259
+ rights_path = f"{path}.rights"
2260
+ rights_data = _object(data["rights"], rights_path)
2261
+ _exact_fields(rights_data, {"status", "evidence", "permissions"}, rights_path)
2262
+ try:
2263
+ rights = RightsSpec(
2264
+ status=_required_text(
2265
+ rights_data["status"],
2266
+ f"{rights_path}.status",
2267
+ maximum=64,
2268
+ ),
2269
+ evidence=_required_text(
2270
+ rights_data["evidence"],
2271
+ f"{rights_path}.evidence",
2272
+ maximum=MAX_TEXT_LENGTH,
2273
+ ),
2274
+ permissions=_parse_choice_array(
2275
+ rights_data["permissions"],
2276
+ f"{rights_path}.permissions",
2277
+ _PERMISSIONS,
2278
+ nonempty=True,
2279
+ maximum=len(_PERMISSIONS),
2280
+ ),
2281
+ )
2282
+ except ContractError as exc:
2283
+ if exc.path.startswith("rights."):
2284
+ suffix = exc.path.removeprefix("rights.")
2285
+ raise ContractError(f"{rights_path}.{suffix}", exc.code, exc.detail) from None
2286
+ raise
2287
+ try:
2288
+ return SourceSpec(
2289
+ source_id=_required_text(
2290
+ data["id"],
2291
+ f"{path}.id",
2292
+ maximum=MAX_IDENTIFIER_LENGTH,
2293
+ ),
2294
+ path=_required_text(data["path"], f"{path}.path", maximum=MAX_PATH_LENGTH),
2295
+ format=_required_text(data["format"], f"{path}.format", maximum=16),
2296
+ origin=_required_text(data["origin"], f"{path}.origin", maximum=4_000),
2297
+ acquisition_method=_required_text(
2298
+ data["acquisition_method"],
2299
+ f"{path}.acquisition_method",
2300
+ maximum=64,
2301
+ ),
2302
+ rights=rights,
2303
+ )
2304
+ except ContractError as exc:
2305
+ if exc.path.startswith("source."):
2306
+ suffix = exc.path.removeprefix("source.")
2307
+ raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
2308
+ raise
2309
+
2310
+
2311
+ def _parse_cleaning(
2312
+ value: Any,
2313
+ path: str,
2314
+ source_ids: frozenset[str],
2315
+ ) -> CleaningStep:
2316
+ data = _object(value, path)
2317
+ _exact_fields(data, {"source", "operation", "columns"}, path)
2318
+ source_id = _required_text(
2319
+ data["source"],
2320
+ f"{path}.source",
2321
+ maximum=MAX_IDENTIFIER_LENGTH,
2322
+ )
2323
+ if source_id not in source_ids:
2324
+ raise ContractError(
2325
+ f"{path}.source",
2326
+ "UNKNOWN_SOURCE",
2327
+ f"must be one of {sorted(source_ids)}",
2328
+ )
2329
+ operation = _required_text(
2330
+ data["operation"],
2331
+ f"{path}.operation",
2332
+ maximum=64,
2333
+ )
2334
+ _choice(operation, _CLEANING_OPERATIONS, f"{path}.operation")
2335
+ if operation in {"trim", "empty_to_null"}:
2336
+ columns: tuple[str, ...] | tuple[tuple[str, str], ...] = _parse_column_array(
2337
+ data["columns"],
2338
+ f"{path}.columns",
2339
+ nonempty=True,
2340
+ maximum=MAX_COLUMN_COUNT,
2341
+ )
2342
+ else:
2343
+ mapping = _object(data["columns"], f"{path}.columns")
2344
+ if not mapping:
2345
+ raise ContractError(
2346
+ f"{path}.columns",
2347
+ "EMPTY_COLLECTION",
2348
+ "cannot be empty",
2349
+ )
2350
+ if len(mapping) > MAX_COLUMN_COUNT:
2351
+ raise ContractError(
2352
+ f"{path}.columns",
2353
+ "COLLECTION_LIMIT",
2354
+ f"exceeds {MAX_COLUMN_COUNT} entries",
2355
+ )
2356
+ pairs: list[tuple[str, str]] = []
2357
+ for key, item in mapping.items():
2358
+ source_column = _column_name(key, f"{path}.columns.<key>")
2359
+ target = _required_text(
2360
+ item,
2361
+ f"{path}.columns[{key!r}]",
2362
+ maximum=MAX_IDENTIFIER_LENGTH,
2363
+ )
2364
+ if operation == "rename":
2365
+ _column_name(target, f"{path}.columns[{key!r}]")
2366
+ else:
2367
+ _choice(target, _CAST_TYPES, f"{path}.columns[{key!r}]")
2368
+ pairs.append((source_column, target))
2369
+ columns = tuple(pairs)
2370
+ try:
2371
+ return CleaningStep(source_id=source_id, operation=operation, columns=columns)
2372
+ except ContractError as exc:
2373
+ if exc.path.startswith("cleaning."):
2374
+ suffix = exc.path.removeprefix("cleaning.")
2375
+ raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
2376
+ raise
2377
+
2378
+
2379
+ def _parse_join(value: Any, path: str, source_ids: frozenset[str]) -> JoinSpec:
2380
+ data = _object(value, path)
2381
+ _exact_fields(data, {"left", "right", "on", "kind", "cardinality"}, path)
2382
+ left = _required_text(data["left"], f"{path}.left", maximum=MAX_IDENTIFIER_LENGTH)
2383
+ right = _required_text(data["right"], f"{path}.right", maximum=MAX_IDENTIFIER_LENGTH)
2384
+ if left not in source_ids:
2385
+ raise ContractError(
2386
+ f"{path}.left",
2387
+ "UNKNOWN_SOURCE",
2388
+ f"must be one of {sorted(source_ids)}",
2389
+ )
2390
+ if right not in source_ids:
2391
+ raise ContractError(
2392
+ f"{path}.right",
2393
+ "UNKNOWN_SOURCE",
2394
+ f"must be one of {sorted(source_ids)}",
2395
+ )
2396
+ try:
2397
+ return JoinSpec(
2398
+ left=left,
2399
+ right=right,
2400
+ on=_parse_column_array(
2401
+ data["on"],
2402
+ f"{path}.on",
2403
+ nonempty=True,
2404
+ maximum=16,
2405
+ ),
2406
+ kind=_required_text(data["kind"], f"{path}.kind", maximum=32),
2407
+ cardinality=_required_text(
2408
+ data["cardinality"],
2409
+ f"{path}.cardinality",
2410
+ maximum=32,
2411
+ ),
2412
+ )
2413
+ except ContractError as exc:
2414
+ if exc.path.startswith("join."):
2415
+ suffix = exc.path.removeprefix("join.")
2416
+ raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
2417
+ raise
2418
+
2419
+
2420
+ def _parse_quality(value: Any, path: str) -> QualitySpec:
2421
+ data = _object(value, path)
2422
+ _exact_fields(data, {"min_rows", "not_null"}, path)
2423
+ try:
2424
+ return QualitySpec(
2425
+ min_rows=_integer(data["min_rows"], f"{path}.min_rows"),
2426
+ not_null=_parse_column_array(
2427
+ data["not_null"],
2428
+ f"{path}.not_null",
2429
+ nonempty=False,
2430
+ maximum=MAX_COLUMN_COUNT,
2431
+ ),
2432
+ )
2433
+ except ContractError as exc:
2434
+ if exc.path.startswith("quality."):
2435
+ suffix = exc.path.removeprefix("quality.")
2436
+ raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
2437
+ raise
2438
+
2439
+
2440
+ def _object(value: Any, path: str) -> Mapping[str, Any]:
2441
+ if not isinstance(value, Mapping) or any(not isinstance(key, str) for key in value):
2442
+ raise ContractError(path, "TYPE", "must be an object with string keys")
2443
+ return value
2444
+
2445
+
2446
+ def _array(
2447
+ value: Any,
2448
+ path: str,
2449
+ *,
2450
+ nonempty: bool,
2451
+ maximum: int,
2452
+ minimum: int = 0,
2453
+ ) -> Sequence[Any]:
2454
+ if not isinstance(value, list):
2455
+ raise ContractError(path, "TYPE", "must be an array")
2456
+ if nonempty and not value:
2457
+ raise ContractError(path, "EMPTY_COLLECTION", "cannot be empty")
2458
+ if len(value) < minimum:
2459
+ raise ContractError(path, "COLLECTION_MINIMUM", f"must contain at least {minimum} entries")
2460
+ if len(value) > maximum:
2461
+ raise ContractError(path, "COLLECTION_LIMIT", f"exceeds {maximum} entries")
2462
+ return value
2463
+
2464
+
2465
+ def _exact_fields(data: Mapping[str, Any], expected: Iterable[str], path: str) -> None:
2466
+ expected_set = frozenset(expected)
2467
+ missing = sorted(expected_set - data.keys())
2468
+ extra = sorted(data.keys() - expected_set)
2469
+ if missing or extra:
2470
+ raise ContractError(
2471
+ path,
2472
+ "FIELDS",
2473
+ f"fields invalid; missing={missing}, unknown={extra}",
2474
+ )
2475
+
2476
+
2477
+ def _required_text(value: Any, path: str, *, maximum: int) -> str:
2478
+ if not isinstance(value, str):
2479
+ raise ContractError(path, "TYPE", "must be a string")
2480
+ _text(value, path, minimum=1, maximum=maximum)
2481
+ return value
2482
+
2483
+
2484
+ def _text(value: Any, path: str, *, minimum: int, maximum: int) -> str:
2485
+ if not isinstance(value, str):
2486
+ raise ContractError(path, "TYPE", "must be a string")
2487
+ if value != value.strip():
2488
+ raise ContractError(path, "NONCANONICAL_TEXT", "must not have surrounding whitespace")
2489
+ if len(value) < minimum:
2490
+ raise ContractError(path, "STRING_MINIMUM", f"must contain at least {minimum} characters")
2491
+ if len(value) > maximum:
2492
+ raise ContractError(path, "STRING_LIMIT", f"exceeds {maximum} characters")
2493
+ for char in value:
2494
+ if ord(char) < 0x20 and char not in {"\t", "\n", "\r"}:
2495
+ raise ContractError(path, "CONTROL_CHARACTER", "contains a control character")
2496
+ return value
2497
+
2498
+
2499
+ def _identifier(value: Any, path: str) -> str:
2500
+ text = _text(
2501
+ value,
2502
+ path,
2503
+ minimum=1,
2504
+ maximum=MAX_IDENTIFIER_LENGTH,
2505
+ )
2506
+ if not _IDENTIFIER.fullmatch(text):
2507
+ raise ContractError(
2508
+ path,
2509
+ "IDENTIFIER",
2510
+ "must be a canonical lowercase identifier",
2511
+ )
2512
+ return text
2513
+
2514
+
2515
+ def _column_name(value: Any, path: str) -> str:
2516
+ text = _text(
2517
+ value,
2518
+ path,
2519
+ minimum=1,
2520
+ maximum=MAX_IDENTIFIER_LENGTH,
2521
+ )
2522
+ if not _COLUMN.fullmatch(text):
2523
+ raise ContractError(
2524
+ path,
2525
+ "COLUMN_IDENTIFIER",
2526
+ "must be a canonical lowercase snake-case column name",
2527
+ )
2528
+ return text
2529
+
2530
+
2531
+ def _version(value: Any, expected: str, path: str) -> str:
2532
+ text = _text(value, path, minimum=1, maximum=64)
2533
+ if text != expected:
2534
+ raise ContractError(path, "VERSION", f"must be {expected!r}")
2535
+ return text
2536
+
2537
+
2538
+ def _choice_version(value: Any, accepted: tuple[str, ...], path: str) -> str:
2539
+ """Pin a schema version to one of the generations still read."""
2540
+
2541
+ text = _text(value, path, minimum=1, maximum=64)
2542
+ if text not in accepted:
2543
+ raise ContractError(path, "VERSION", f"must be one of {list(accepted)}")
2544
+ return text
2545
+
2546
+
2547
+ def _choice(value: Any, choices: frozenset[str], path: str) -> str:
2548
+ if not isinstance(value, str) or value not in choices:
2549
+ raise ContractError(path, "ENUM", f"must be one of {sorted(choices)}")
2550
+ return value
2551
+
2552
+
2553
+ def _declared_unit(value: Any, path: str) -> Unit | None:
2554
+ """Accept ``none`` or one unit code the grammar resolves, and refuse anything else by name.
2555
+
2556
+ Returning the resolved unit rather than the string is what lets the rules above ask what the
2557
+ code means instead of how it was spelled. ``none`` resolves to nothing on purpose: it is the
2558
+ absence of a unit, not a dimensionless one.
2559
+ """
2560
+
2561
+ text = _text(value, path, minimum=1, maximum=64)
2562
+ if text == NO_UNIT:
2563
+ return None
2564
+ try:
2565
+ return resolve_declared_unit(text)
2566
+ except UnitError as exc:
2567
+ raise ContractError(
2568
+ path, "UNIT_CODE", f"{text!r} is not a unit code: {exc.detail}"
2569
+ ) from None
2570
+
2571
+
2572
+ def _integer(value: Any, path: str) -> int:
2573
+ if type(value) is not int:
2574
+ raise ContractError(path, "INTEGER", "must be an integer (booleans are not integers)")
2575
+ return value
2576
+
2577
+
2578
+ def _bounded_int(value: Any, path: str, *, minimum: int, maximum: int) -> int:
2579
+ integer = _integer(value, path)
2580
+ if integer < minimum or integer > maximum:
2581
+ raise ContractError(
2582
+ path,
2583
+ "INTEGER_RANGE",
2584
+ f"must be between {minimum} and {maximum}",
2585
+ )
2586
+ return integer
2587
+
2588
+
2589
+ def _utc_timestamp(value: Any, path: str) -> datetime:
2590
+ text = _text(value, path, minimum=1, maximum=32)
2591
+ if not _UTC_TIMESTAMP.fullmatch(text):
2592
+ raise ContractError(
2593
+ path,
2594
+ "UTC_TIMESTAMP",
2595
+ "must be a zero-padded RFC 3339 UTC timestamp ending in Z",
2596
+ )
2597
+ try:
2598
+ parsed = datetime.fromisoformat(text.removesuffix("Z") + "+00:00")
2599
+ except ValueError:
2600
+ raise ContractError(path, "UTC_TIMESTAMP", "must be a valid UTC timestamp") from None
2601
+ if parsed.utcoffset() is None or parsed.utcoffset().total_seconds() != 0:
2602
+ raise ContractError(path, "UTC_TIMESTAMP", "must use UTC Z")
2603
+ return parsed
2604
+
2605
+
2606
+ def _sha256(value: Any, path: str) -> str:
2607
+ if not isinstance(value, str) or not _SHA256.fullmatch(value):
2608
+ raise ContractError(path, "SHA256", "must be a lowercase 64-character SHA-256 digest")
2609
+ return value
2610
+
2611
+
2612
+ def _relative_posix_path(value: Any, path: str) -> str:
2613
+ text = _text(value, path, minimum=1, maximum=MAX_PATH_LENGTH)
2614
+ pure = PurePosixPath(text)
2615
+ if (
2616
+ "\\" in text
2617
+ or "\x00" in text
2618
+ or pure.is_absolute()
2619
+ or not pure.parts
2620
+ or any(part in {"", ".", ".."} for part in pure.parts)
2621
+ or pure.as_posix() != text
2622
+ or text.endswith("/")
2623
+ or (pure.parts and ":" in pure.parts[0])
2624
+ ):
2625
+ raise ContractError(
2626
+ path,
2627
+ "POSIX_PATH",
2628
+ "must be canonical normalized relative POSIX text",
2629
+ )
2630
+ return text
2631
+
2632
+
2633
+ def _decoded_query_component(value: str) -> str:
2634
+ """Bound repeated decoding so concealed credential keys cannot enter durable evidence."""
2635
+
2636
+ decoded = value
2637
+ for _ in range(3):
2638
+ replacement = unquote_plus(decoded)
2639
+ if replacement == decoded:
2640
+ break
2641
+ decoded = replacement
2642
+ return decoded
2643
+
2644
+
2645
+ def validate_source_locator(value: Any, kind: str, path: str) -> str:
2646
+ """Validate a durable source locator without admitting credentials or signed URLs."""
2647
+
2648
+ text = _text(value, path, minimum=1, maximum=MAX_PATH_LENGTH)
2649
+ if kind == "relative_path":
2650
+ return _relative_posix_path(text, path)
2651
+ if kind == "artifact_reference":
2652
+ if not text.startswith("artifact://") or len(text) <= len("artifact://"):
2653
+ raise ContractError(path, "ARTIFACT_REFERENCE", "must be an artifact:// reference")
2654
+ if any(char.isspace() for char in text):
2655
+ raise ContractError(path, "ARTIFACT_REFERENCE", "must not contain whitespace")
2656
+ try:
2657
+ parsed_artifact = urlsplit(text)
2658
+ except ValueError as error:
2659
+ raise ContractError(
2660
+ path, "ARTIFACT_REFERENCE", "must be a valid artifact:// reference"
2661
+ ) from error
2662
+ if (
2663
+ parsed_artifact.username is not None
2664
+ or parsed_artifact.password is not None
2665
+ or parsed_artifact.fragment
2666
+ ):
2667
+ raise ContractError(
2668
+ path,
2669
+ "ARTIFACT_REFERENCE",
2670
+ "must not contain embedded credentials or a fragment",
2671
+ )
2672
+ for query_key, query_value in parse_qsl(parsed_artifact.query, keep_blank_values=True):
2673
+ if _SENSITIVE_LOCATOR_QUERY_KEY.search(
2674
+ _decoded_query_component(query_key)
2675
+ ) or _SECRET_LOCATOR_VALUE.search(_decoded_query_component(query_value)):
2676
+ raise ContractError(
2677
+ path,
2678
+ "ARTIFACT_REFERENCE",
2679
+ "must not contain credentials or signed/secret query material",
2680
+ )
2681
+ return text
2682
+ try:
2683
+ parsed = urlsplit(text)
2684
+ except ValueError as error:
2685
+ raise ContractError(path, "HTTPS_URL", "must be a valid canonical HTTPS URL") from error
2686
+ try:
2687
+ port = parsed.port
2688
+ except ValueError:
2689
+ port = None
2690
+ invalid_port = True
2691
+ else:
2692
+ invalid_port = False
2693
+ if (
2694
+ parsed.scheme != "https"
2695
+ or not parsed.hostname
2696
+ or parsed.username is not None
2697
+ or parsed.password is not None
2698
+ or parsed.fragment
2699
+ or parsed.hostname != parsed.hostname.lower()
2700
+ or invalid_port
2701
+ or port == 443
2702
+ ):
2703
+ raise ContractError(
2704
+ path,
2705
+ "HTTPS_URL",
2706
+ "must be a credential-free canonical HTTPS URL without a fragment",
2707
+ )
2708
+ for query_key, query_value in parse_qsl(parsed.query, keep_blank_values=True):
2709
+ decoded_key = _decoded_query_component(query_key)
2710
+ decoded_value = _decoded_query_component(query_value)
2711
+ if _SENSITIVE_LOCATOR_QUERY_KEY.search(decoded_key) or _SECRET_LOCATOR_VALUE.search(
2712
+ decoded_value
2713
+ ):
2714
+ raise ContractError(
2715
+ path,
2716
+ "HTTPS_URL",
2717
+ "must not contain credentials or signed/secret query material",
2718
+ )
2719
+ return text
2720
+
2721
+
2722
+ def source_locator_identity(kind: str, locator: str) -> tuple[str, str]:
2723
+ """Return a stable identity for already-validated proposal locators."""
2724
+
2725
+ validate_source_locator(locator, kind, "source_proposal.locator")
2726
+ if kind != "https_url":
2727
+ return kind, locator
2728
+ parsed = urlsplit(locator)
2729
+ normalized = urlunsplit((parsed.scheme, parsed.netloc, parsed.path or "/", parsed.query, ""))
2730
+ return kind, normalized
2731
+
2732
+
2733
+ def _evidence_uri(value: Any, path: str) -> str:
2734
+ text = _text(value, path, minimum=1, maximum=2_048)
2735
+ parsed = urlsplit(text)
2736
+ if parsed.scheme not in {"https", "fixture"} or not parsed.netloc:
2737
+ raise ContractError(path, "EVIDENCE_URI", "must be an https:// or fixture:// URI")
2738
+ if parsed.username is not None or parsed.password is not None:
2739
+ raise ContractError(path, "EVIDENCE_URI", "must not embed credentials")
2740
+ return text
2741
+
2742
+
2743
+ def _require_unique(values: Iterable[Any], path: str) -> None:
2744
+ items = tuple(values)
2745
+ if len(items) != len(set(items)):
2746
+ raise ContractError(path, "DUPLICATE", "must contain unique entries")
2747
+
2748
+
2749
+ def _column_tuple(
2750
+ values: Any,
2751
+ path: str,
2752
+ *,
2753
+ nonempty: bool,
2754
+ maximum: int,
2755
+ ) -> tuple[str, ...]:
2756
+ if not isinstance(values, tuple):
2757
+ raise ContractError(path, "TYPE", "must be an immutable tuple")
2758
+ if nonempty and not values:
2759
+ raise ContractError(path, "EMPTY_COLLECTION", "cannot be empty")
2760
+ if len(values) > maximum:
2761
+ raise ContractError(path, "COLLECTION_LIMIT", f"exceeds {maximum} entries")
2762
+ for index, item in enumerate(values):
2763
+ _column_name(item, f"{path}[{index}]")
2764
+ _require_unique(values, path)
2765
+ return values
2766
+
2767
+
2768
+ def _text_tuple(
2769
+ values: Any,
2770
+ path: str,
2771
+ *,
2772
+ nonempty: bool,
2773
+ maximum: int,
2774
+ item_maximum: int,
2775
+ ) -> tuple[str, ...]:
2776
+ if not isinstance(values, tuple):
2777
+ raise ContractError(path, "TYPE", "must be an immutable tuple")
2778
+ if nonempty and not values:
2779
+ raise ContractError(path, "EMPTY_COLLECTION", "cannot be empty")
2780
+ if len(values) > maximum:
2781
+ raise ContractError(path, "COLLECTION_LIMIT", f"exceeds {maximum} entries")
2782
+ for index, item in enumerate(values):
2783
+ _text(item, f"{path}[{index}]", minimum=1, maximum=item_maximum)
2784
+ _require_unique(values, path)
2785
+ return values
2786
+
2787
+
2788
+ def _choice_tuple(
2789
+ values: Any,
2790
+ choices: frozenset[str],
2791
+ path: str,
2792
+ *,
2793
+ maximum: int,
2794
+ nonempty: bool = False,
2795
+ ) -> tuple[str, ...]:
2796
+ if not isinstance(values, tuple):
2797
+ raise ContractError(path, "TYPE", "must be an immutable tuple")
2798
+ if nonempty and not values:
2799
+ raise ContractError(path, "EMPTY_COLLECTION", "cannot be empty")
2800
+ if len(values) > maximum:
2801
+ raise ContractError(path, "COLLECTION_LIMIT", f"exceeds {maximum} entries")
2802
+ for index, item in enumerate(values):
2803
+ _choice(item, choices, f"{path}[{index}]")
2804
+ _require_unique(values, path)
2805
+ return values
2806
+
2807
+
2808
+ def _typed_tuple(
2809
+ values: Any,
2810
+ expected_type: type[Any],
2811
+ path: str,
2812
+ *,
2813
+ nonempty: bool,
2814
+ maximum: int,
2815
+ minimum: int = 0,
2816
+ ) -> tuple[Any, ...]:
2817
+ if not isinstance(values, tuple):
2818
+ raise ContractError(path, "TYPE", "must be an immutable tuple")
2819
+ if nonempty and not values:
2820
+ raise ContractError(path, "EMPTY_COLLECTION", "cannot be empty")
2821
+ if len(values) < minimum:
2822
+ raise ContractError(path, "COLLECTION_MINIMUM", f"must contain at least {minimum} entries")
2823
+ if len(values) > maximum:
2824
+ raise ContractError(path, "COLLECTION_LIMIT", f"exceeds {maximum} entries")
2825
+ for index, item in enumerate(values):
2826
+ if not isinstance(item, expected_type):
2827
+ raise ContractError(
2828
+ f"{path}[{index}]",
2829
+ "TYPE",
2830
+ f"must be {expected_type.__name__}",
2831
+ )
2832
+ return values
2833
+
2834
+
2835
+ def _parse_column_array(
2836
+ value: Any,
2837
+ path: str,
2838
+ *,
2839
+ nonempty: bool,
2840
+ maximum: int,
2841
+ ) -> tuple[str, ...]:
2842
+ items = _array(value, path, nonempty=nonempty, maximum=maximum)
2843
+ result = tuple(_column_name(item, f"{path}[{index}]") for index, item in enumerate(items))
2844
+ _require_unique(result, path)
2845
+ return result
2846
+
2847
+
2848
+ def _parse_text_array(
2849
+ value: Any,
2850
+ path: str,
2851
+ *,
2852
+ nonempty: bool,
2853
+ maximum: int,
2854
+ item_maximum: int,
2855
+ ) -> tuple[str, ...]:
2856
+ items = _array(value, path, nonempty=nonempty, maximum=maximum)
2857
+ result = tuple(
2858
+ _required_text(item, f"{path}[{index}]", maximum=item_maximum)
2859
+ for index, item in enumerate(items)
2860
+ )
2861
+ _require_unique(result, path)
2862
+ return result
2863
+
2864
+
2865
+ def _parse_choice_array(
2866
+ value: Any,
2867
+ path: str,
2868
+ choices: frozenset[str],
2869
+ *,
2870
+ nonempty: bool,
2871
+ maximum: int,
2872
+ ) -> tuple[str, ...]:
2873
+ items = _array(value, path, nonempty=nonempty, maximum=maximum)
2874
+ result: list[str] = []
2875
+ for index, item in enumerate(items):
2876
+ text = _required_text(item, f"{path}[{index}]", maximum=64)
2877
+ _choice(text, choices, f"{path}[{index}]")
2878
+ result.append(text)
2879
+ _require_unique(result, path)
2880
+ return tuple(result)