mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,1710 @@
1
+ """Deterministic nbformat-4.5 visual sidecar for a sealed candidate.
2
+
3
+ The generated ``table.ipynb`` is useful without a kernel: it embeds a data preview and an
4
+ all-column analytics inspector as normal ``display_data`` outputs. The inspector's versioned rich
5
+ MIME payload includes logical and physical types, declared units, missingness, cardinality,
6
+ type-appropriate summaries and distributions, lineage, and transformations. Portable
7
+ ``text/html`` and ``text/plain`` fallbacks keep the notebook useful in ordinary Jupyter viewers.
8
+
9
+ Runnable pandas cells remain available for local exploration. Generated code cells carry
10
+ nbclient's ``skip-execution`` tag, preserving the deterministic pre-filled outputs during headless
11
+ execution while remaining manually runnable in an interactive kernel.
12
+
13
+ The generator is deterministic: every cell ``id`` is the first 12 hex of ``sha256(cell_source)``
14
+ and every value is a native Python scalar, so two renders of the same sealed candidate are
15
+ byte-identical. The provenance block is derived the same way — no clock, no randomness.
16
+
17
+ It adds NO new dependency: the notebook is nbformat 4.5 JSON built with the stdlib ``json``, and
18
+ the head-N Parquet sample is read through the already-pinned ``pyarrow`` (imported lazily). It never
19
+ imports ``nbformat``/``nbconvert``.
20
+
21
+ The sidecar is written as a run-dir SIBLING of ``candidate/`` (the run dir stays owner-writable
22
+ ``0o700``; only ``candidate/`` is sealed ``0o555``/``0o444``), so it never enters the sealed member
23
+ set, never moves a digest, and is invisible to ``verify_candidate`` (which polices only
24
+ ``candidate/``).
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import hashlib
30
+ import html
31
+ import json
32
+ import math
33
+ import os
34
+ import secrets
35
+ import stat
36
+ from collections import Counter
37
+ from copy import deepcopy
38
+ from decimal import Decimal
39
+ from pathlib import Path
40
+ from typing import Any
41
+
42
+ try:
43
+ import fcntl
44
+ except ImportError: # pragma: no cover - secure candidate verification already refuses Windows
45
+ fcntl = None # type: ignore[assignment]
46
+
47
+ from mostlyright.data_harness import pipeline
48
+ from mostlyright.data_harness import recipe as recipe_contracts
49
+ from mostlyright.data_harness.local_contracts import UNIT_STATES
50
+ from mostlyright.data_harness.preparation.contracts import PHYSICAL_TYPES
51
+ from mostlyright.data_harness.recipe import LOGICAL_TYPES, UNITS
52
+ from mostlyright.data_harness.units import is_declarable_unit, legacy_aliases
53
+
54
+ # The one name this module writes the generated notebook under. It is a name other lanes have
55
+ # to be able to say without writing it themselves -- the hosted handoff reports the path it
56
+ # installed and never opens it -- so it is a constant rather than four literals.
57
+ TABLE_NOTEBOOK_NAME = "table.ipynb"
58
+ # Private compatibility import for the hosted handoff lane added before the V3 vocabulary cutover.
59
+ # It names the same Table sidecar and never crosses a wire or UI boundary.
60
+ DATASET_NOTEBOOK_NAME = TABLE_NOTEBOOK_NAME
61
+ _SAMPLE_ROWS = 10
62
+ _NBFORMAT_MAJOR = 4
63
+ _NBFORMAT_MINOR = 5
64
+ _PROVENANCE_DIGEST_PREFIX = 12
65
+ # ``nbclient``'s default ``skip_cells_with_tag``. Every generated code cell carries it, so the
66
+ # headless execute path skips them and the hand-built display outputs survive untouched.
67
+ _SKIP_EXECUTION_TAG = "skip-execution"
68
+ _COLUMN_PROFILE_MIME = "application/vnd.mostlyright.column-profile.v1+json"
69
+ _PROFILE_SAMPLE_ROWS = 100_000
70
+ _MAX_CATEGORY_BARS = 12
71
+ _FALLBACK_MAX_COLUMNS = 64
72
+ _FALLBACK_MAX_LIMITATIONS = 16
73
+ _FALLBACK_MAX_CATEGORIES = 16
74
+ _FALLBACK_TEXT_LIMIT = 512
75
+
76
+ _UNIT_DISPLAY: dict[str, tuple[str | None, str]] = {
77
+ "none": (None, "Not applicable"),
78
+ "count": ("count", "count"),
79
+ "ratio": ("ratio", "unitless ratio"),
80
+ "percent": ("%", "percent"),
81
+ "celsius": ("°C", "degrees Celsius"),
82
+ "fahrenheit": ("°F", "degrees Fahrenheit"),
83
+ "kelvin": ("K", "kelvin"),
84
+ "meter": ("m", "meters"),
85
+ "kilometer": ("km", "kilometers"),
86
+ "mile": ("mi", "miles"),
87
+ "microgram_per_cubic_meter": ("µg/m³", "micrograms per cubic meter"),
88
+ "second": ("s", "seconds"),
89
+ "minute": ("min", "minutes"),
90
+ "hour": ("h", "hours"),
91
+ "hectopascal": ("hPa", "hectopascals"),
92
+ }
93
+ # These are the written-out labels for the fifteen tokens the vocabulary began with, not a second
94
+ # vocabulary. Import-time equality holds them honest: an alias added to the pinned table without a
95
+ # label here is an explicit implementation failure rather than a silent gap.
96
+ assert set(_UNIT_DISPLAY) == set(UNITS)
97
+ # The same fifteen labels, reached from the codes those names stand for. ``celsius`` and ``Cel``
98
+ # are one unit, so the page a person reads has to say so: without this, adopting the spelling the
99
+ # unit grammar promotes would silently downgrade "degrees Celsius" to "Cel".
100
+ _CODE_DISPLAY: dict[str, tuple[str | None, str]] = {
101
+ code: _UNIT_DISPLAY[token] for token, code in legacy_aliases().items()
102
+ }
103
+ _UNIT_STATE_DISPLAY = {
104
+ "declared": "Declared physical unit",
105
+ "normalized": "Normalized (dimensionless)",
106
+ "not_applicable": "Not applicable",
107
+ "unknown": "Unit not recoverable",
108
+ }
109
+ assert set(_UNIT_STATE_DISPLAY) == set(UNIT_STATES)
110
+ assert set(PHYSICAL_TYPES) == set(LOGICAL_TYPES)
111
+
112
+
113
+ # --- nbformat 4.5 cell primitives --------------------------------------------------------------
114
+
115
+
116
+ def _cell_id(source: str) -> str:
117
+ """First 12 hex of ``sha256(source)`` — a content-derived, stable, deterministic cell id."""
118
+
119
+ return hashlib.sha256(source.encode("utf-8")).hexdigest()[:12]
120
+
121
+
122
+ def _markdown_cell(source: str) -> dict[str, Any]:
123
+ return {
124
+ "cell_type": "markdown",
125
+ "id": _cell_id(source),
126
+ "metadata": {},
127
+ "source": source,
128
+ }
129
+
130
+
131
+ def _code_cell(source: str, outputs: tuple[dict[str, Any], ...] = ()) -> dict[str, Any]:
132
+ """A generated code cell, tagged ``skip-execution`` so a headless execute leaves it alone.
133
+
134
+ The tag is cell METADATA, not cell source, so the content-derived ``id`` is unaffected and the
135
+ render stays byte-deterministic.
136
+ """
137
+
138
+ return {
139
+ "cell_type": "code",
140
+ "execution_count": None,
141
+ "id": _cell_id(source),
142
+ "metadata": {"tags": [_SKIP_EXECUTION_TAG]},
143
+ "outputs": list(outputs),
144
+ "source": source,
145
+ }
146
+
147
+
148
+ def _display_html(markup: str) -> dict[str, Any]:
149
+ return {"output_type": "display_data", "data": {"text/html": markup}, "metadata": {}}
150
+
151
+
152
+ def _fallback_text(value: Any, *, limit: int = _FALLBACK_TEXT_LIMIT) -> tuple[str, bool]:
153
+ """Return deterministic bounded text and whether an explicit truncation marker was added."""
154
+
155
+ rendered = "" if value is None else str(value)
156
+ if len(rendered) <= limit:
157
+ return rendered, False
158
+ marker = " [truncated]"
159
+ return rendered[: max(0, limit - len(marker))] + marker, True
160
+
161
+
162
+ def _bounded_items(value: Any, maximum: int) -> tuple[list[Any], int]:
163
+ items = value if isinstance(value, list) else []
164
+ return items[:maximum], max(0, len(items) - maximum)
165
+
166
+
167
+ def _fallback_semantic_sections(payload: dict[str, Any]) -> tuple[str, list[str]]:
168
+ """Render bounded deterministic dataset and column semantics for portable fallbacks."""
169
+
170
+ semantics = payload.get("semantics") if isinstance(payload.get("semantics"), dict) else {}
171
+ message, _ = _fallback_text(
172
+ semantics.get(
173
+ "message",
174
+ "This Build declares no column meanings. Nothing here is inferred from column names.",
175
+ )
176
+ )
177
+ summary, _ = _fallback_text(semantics.get("summary") or "Not declared")
178
+ grain, _ = _fallback_text(semantics.get("grain_statement") or "Not declared")
179
+ coverage = semantics.get("coverage") if isinstance(semantics.get("coverage"), dict) else {}
180
+ time_range = (
181
+ coverage.get("time_range") if isinstance(coverage.get("time_range"), dict) else None
182
+ )
183
+ if time_range is None:
184
+ time_text = "Not declared"
185
+ else:
186
+ start, _ = _fallback_text(time_range.get("start_inclusive") or "Not declared")
187
+ end, _ = _fallback_text(time_range.get("end_exclusive") or "Not declared")
188
+ time_text = f"{start} (inclusive) to {end} (exclusive)"
189
+ population, _ = _fallback_text(coverage.get("population") or "Not declared")
190
+ completeness, _ = _fallback_text(coverage.get("completeness_note") or "Not declared")
191
+
192
+ limitations, omitted_limitations = _bounded_items(
193
+ semantics.get("limitations"), _FALLBACK_MAX_LIMITATIONS
194
+ )
195
+ limitation_texts = [_fallback_text(item)[0] for item in limitations]
196
+ if not limitation_texts:
197
+ limitation_texts = ["None declared"]
198
+
199
+ html_parts = [
200
+ '<section class="mostlyright-table-semantics">',
201
+ "<h3>Table semantics</h3>",
202
+ f"<p>{html.escape(message)}</p>",
203
+ f"<p><strong>Summary:</strong> {html.escape(summary)}</p>",
204
+ f"<p><strong>One row:</strong> {html.escape(grain)}</p>",
205
+ f"<p><strong>Time coverage:</strong> {html.escape(time_text)}</p>",
206
+ f"<p><strong>Population:</strong> {html.escape(population)}</p>",
207
+ f"<p><strong>Completeness:</strong> {html.escape(completeness)}</p>",
208
+ "<h4>Known limitations</h4><ul>",
209
+ *(f"<li>{html.escape(item)}</li>" for item in limitation_texts),
210
+ ]
211
+ plain = [
212
+ "Table semantics",
213
+ message,
214
+ f"Summary: {summary}",
215
+ f"One row: {grain}",
216
+ f"Time coverage: {time_text}",
217
+ f"Population: {population}",
218
+ f"Completeness: {completeness}",
219
+ "Known limitations:",
220
+ *(f"- {item}" for item in limitation_texts),
221
+ ]
222
+ if omitted_limitations:
223
+ notice = f"[truncated: {omitted_limitations} additional limitations omitted]"
224
+ html_parts.append(f"<li>{html.escape(notice)}</li>")
225
+ plain.append(notice)
226
+ html_parts.extend(["</ul>", "</section>"])
227
+
228
+ columns = payload.get("columns") if isinstance(payload.get("columns"), list) else []
229
+ displayed_columns = columns[:_FALLBACK_MAX_COLUMNS]
230
+ omitted_columns = max(0, len(columns) - _FALLBACK_MAX_COLUMNS)
231
+ header = (
232
+ "<tr><th>column</th><th>label</th><th>meaning</th><th>type</th><th>unit / scaling</th>"
233
+ "<th>categorical definitions</th><th>evidence</th></tr>"
234
+ )
235
+ rows: list[str] = []
236
+ plain.extend(["", "Column semantics"])
237
+ for column in displayed_columns:
238
+ if not isinstance(column, dict):
239
+ continue
240
+ name, _ = _fallback_text(column.get("name", ""), limit=120)
241
+ label, _ = _fallback_text(column.get("display_label") or "Not declared", limit=120)
242
+ description, _ = _fallback_text(column.get("description") or "Not declared")
243
+ logical, _ = _fallback_text(column.get("logical_type", "unknown"), limit=64)
244
+ physical, _ = _fallback_text(column.get("physical_type", "unknown"), limit=64)
245
+ unit = column.get("unit") if isinstance(column.get("unit"), dict) else {}
246
+ unit_text, _ = _fallback_text(unit.get("symbol") or unit.get("label") or "not declared")
247
+ if unit.get("state") == "normalized":
248
+ normalization, _ = _fallback_text(unit.get("normalization") or "unknown", limit=64)
249
+ reversible = "reversible" if unit.get("reversible") else "not reversible"
250
+ unit_text = f"{unit_text}; {normalization}; {reversible}"
251
+
252
+ categories, omitted_categories = _bounded_items(
253
+ column.get("categories"), _FALLBACK_MAX_CATEGORIES
254
+ )
255
+ category_parts = []
256
+ for item in categories:
257
+ if isinstance(item, dict):
258
+ code, _ = _fallback_text(item.get("code"), limit=64)
259
+ category_label, _ = _fallback_text(item.get("label"), limit=120)
260
+ category_parts.append(f"{code} = {category_label}")
261
+ category_status = str(column.get("categories_status") or "not declared").replace("_", " ")
262
+ category_text = (
263
+ f"{category_status}: {'; '.join(category_parts)}" if category_parts else category_status
264
+ )
265
+ if omitted_categories:
266
+ category_text += f"; [truncated: {omitted_categories} additional definitions omitted]"
267
+
268
+ refs, omitted_refs = _bounded_items(column.get("evidence_refs"), 8)
269
+ evidence_parts = [
270
+ f"{item.get('kind')}: {item.get('ref')}" for item in refs if isinstance(item, dict)
271
+ ]
272
+ evidence_text = "; ".join(evidence_parts) or "No evidence reference declared"
273
+ if omitted_refs:
274
+ evidence_text += f"; [truncated: {omitted_refs} additional references omitted]"
275
+
276
+ rows.append(
277
+ "<tr>"
278
+ f"<td>{html.escape(name)}</td><td>{html.escape(label)}</td>"
279
+ f"<td>{html.escape(description)}</td>"
280
+ f"<td>{html.escape(logical)} / {html.escape(physical)}</td>"
281
+ f"<td>{html.escape(unit_text)}</td>"
282
+ f"<td>{html.escape(category_text)}</td>"
283
+ f"<td>{html.escape(evidence_text)}</td></tr>"
284
+ )
285
+ plain.extend(
286
+ [
287
+ f"- {name} — {label}",
288
+ f" Meaning: {description}",
289
+ f" Type: {logical} / {physical}",
290
+ f" Unit / scaling: {unit_text}",
291
+ f" Categorical definitions: {category_text}",
292
+ f" Evidence: {evidence_text}",
293
+ ]
294
+ )
295
+ if omitted_columns:
296
+ notice = f"[truncated: {omitted_columns} additional columns omitted]"
297
+ rows.append(f'<tr><td colspan="7">{html.escape(notice)}</td></tr>')
298
+ plain.append(notice)
299
+ html_parts.extend(
300
+ [
301
+ '<section class="mostlyright-column-semantics">',
302
+ "<h3>Column semantics</h3>",
303
+ '<table border="1" class="dataframe"><thead>',
304
+ header,
305
+ f"</thead><tbody>{''.join(rows)}</tbody></table>",
306
+ "</section>",
307
+ ]
308
+ )
309
+ return "".join(html_parts), plain
310
+
311
+
312
+ def _display_column_profiles(
313
+ payload: dict[str, Any], *, fallback_payload: dict[str, Any] | None = None
314
+ ) -> dict[str, Any]:
315
+ """Portable custom output with bounded, complete semantic fallbacks."""
316
+
317
+ raw_columns = payload.get("columns", [])
318
+ columns = raw_columns[:_FALLBACK_MAX_COLUMNS] if isinstance(raw_columns, list) else []
319
+ semantic_html, plain = _fallback_semantic_sections(fallback_payload or payload)
320
+ header = (
321
+ "<tr><th>column</th><th>label</th><th>logical</th><th>physical</th><th>unit state</th>"
322
+ "<th>missing</th><th>distinct</th></tr>"
323
+ )
324
+ rows: list[str] = []
325
+ plain.extend(["", "Column inspector"])
326
+ for column in columns if isinstance(columns, list) else []:
327
+ if not isinstance(column, dict):
328
+ continue
329
+ unit = column.get("unit") if isinstance(column.get("unit"), dict) else {}
330
+ unit_text = unit.get("symbol") or unit.get("label") or "not declared"
331
+ name = str(column.get("name", ""))
332
+ label = str(column.get("display_label") or "Not declared")
333
+ logical = str(column.get("logical_type", "unknown"))
334
+ physical = str(column.get("physical_type", "unknown"))
335
+ missing = int(column.get("missing_count", 0) or 0)
336
+ distinct = int(column.get("distinct_count", 0) or 0)
337
+ rows.append(
338
+ "<tr>"
339
+ f"<td>{html.escape(name)}</td><td>{html.escape(label)}</td>"
340
+ f"<td>{html.escape(logical)}</td>"
341
+ f"<td>{html.escape(physical)}</td><td>{html.escape(str(unit_text))}</td>"
342
+ f"<td>{missing}</td><td>{distinct}</td></tr>"
343
+ )
344
+ plain.append(
345
+ f"- {name} ({label}): {logical} / {physical}; unit {unit_text}; "
346
+ f"{missing} missing; {distinct} distinct"
347
+ )
348
+ omitted_inspector_columns = max(
349
+ 0, len(raw_columns) - _FALLBACK_MAX_COLUMNS if isinstance(raw_columns, list) else 0
350
+ )
351
+ if omitted_inspector_columns:
352
+ notice = f"[truncated: {omitted_inspector_columns} additional profiles omitted]"
353
+ rows.append(f'<tr><td colspan="7">{html.escape(notice)}</td></tr>')
354
+ plain.append(notice)
355
+ fallback = (
356
+ '<div class="mostlyright-column-profile">'
357
+ f"{semantic_html}"
358
+ "<p><strong>Column inspector</strong> — open this notebook in Mostly Right for charts "
359
+ "and interactive column selection.</p>"
360
+ '<table border="1" class="dataframe"><thead>'
361
+ f"{header}</thead><tbody>{''.join(rows)}</tbody></table></div>"
362
+ )
363
+ return {
364
+ "output_type": "display_data",
365
+ "data": {
366
+ _COLUMN_PROFILE_MIME: payload,
367
+ "text/html": fallback,
368
+ "text/plain": "\n".join(plain),
369
+ },
370
+ "metadata": {},
371
+ }
372
+
373
+
374
+ # --- hand-built, kernel-free display outputs ---------------------------------------------------
375
+
376
+
377
+ def _cell_text(value: Any) -> str:
378
+ return "" if value is None else str(value)
379
+
380
+
381
+ def _html_table(rows: list[dict[str, Any]], columns: list[str]) -> dict[str, Any]:
382
+ """A hand-built HTML ``<table>`` of the head-N sample rows (every cell HTML-escaped)."""
383
+
384
+ # Match pandas' structural index signal: a blank leading header and a row-header ``th``. The
385
+ # renderer can then distinguish the index from real data columns without guessing from values.
386
+ header = "<th></th>" + "".join(f"<th>{html.escape(str(column))}</th>" for column in columns)
387
+ body_rows: list[str] = []
388
+ for index, row in enumerate(rows):
389
+ cells = "".join(
390
+ f"<td>{html.escape(_cell_text(row.get(column)))}</td>" for column in columns
391
+ )
392
+ body_rows.append(f"<tr><th>{index}</th>{cells}</tr>")
393
+ if not body_rows:
394
+ body_rows.append(
395
+ f'<tr><th></th><td colspan="{max(1, len(columns))}"><em>no rows</em></td></tr>'
396
+ )
397
+ markup = (
398
+ '<table border="1" class="dataframe">'
399
+ f"<thead><tr>{header}</tr></thead>"
400
+ f"<tbody>{''.join(body_rows)}</tbody>"
401
+ "</table>"
402
+ )
403
+ return _display_html(markup)
404
+
405
+
406
+ # --- markdown evidence sections ----------------------------------------------------------------
407
+
408
+
409
+ def _esc(value: Any) -> str:
410
+ """HTML-escape an evidence-derived value before interpolating it into a markdown cell.
411
+
412
+ These cells are rendered to HTML by the viewer's ``HTMLExporter``, which passes raw HTML in
413
+ markdown through live — so every evidence-derived value (question, profiled types, lineage ops,
414
+ quality rows, join facts, source origin/rights) is escaped here, exactly as the sample-row table
415
+ and SVG labels already escape. Escaping is safe in every markdown context (code span, table
416
+ cell, or bold flow): an escaped ``&lt;`` never re-opens as a tag even if the value breaks a
417
+ backtick.
418
+ """
419
+
420
+ return html.escape("" if value is None else str(value))
421
+
422
+
423
+ def _header_markdown(question: str, row_count: int, columns: int) -> str:
424
+ title = question.strip().rstrip("?") or "Research dataset"
425
+ return f"# {_esc(title)}\n\n{row_count:,} rows · {columns} columns."
426
+
427
+
428
+ def _overview_markdown(semantics: dict[str, Any]) -> str:
429
+ lines = [
430
+ "## Dataset contract",
431
+ "",
432
+ _esc(semantics.get("message", "Semantic status unavailable.")),
433
+ ]
434
+ summary = semantics.get("summary")
435
+ grain = semantics.get("grain_statement")
436
+ coverage = semantics.get("coverage")
437
+ if isinstance(summary, str):
438
+ lines.extend(["", "### Definition", "", _esc(summary)])
439
+ else:
440
+ lines.extend(["", "### Definition", "", "No dataset-level summary was declared."])
441
+ if isinstance(grain, str):
442
+ lines.append(f"**Row grain:** {_esc(grain)}")
443
+ else:
444
+ lines.append("**Row grain:** Not declared.")
445
+ lines.extend(["", "### Coverage", ""])
446
+ if isinstance(coverage, dict):
447
+ time_range = coverage.get("time_range")
448
+ if isinstance(time_range, dict):
449
+ lines.append(
450
+ "**Time:** "
451
+ f"{_esc(time_range.get('start_inclusive'))} (inclusive) to "
452
+ f"{_esc(time_range.get('end_exclusive'))} (exclusive)."
453
+ )
454
+ else:
455
+ lines.append("**Time:** Not declared.")
456
+ lines.append(f"**Population:** {_esc(coverage.get('population'))}")
457
+ lines.append(f"**Completeness:** {_esc(coverage.get('completeness_note'))}")
458
+ limitations = semantics.get("limitations")
459
+ lines.extend(["", "### Known limitations", ""])
460
+ if isinstance(limitations, list) and limitations:
461
+ lines.extend(f"- {_esc(item)}" for item in limitations)
462
+ elif isinstance(limitations, list):
463
+ lines.append("No limitations were declared.")
464
+ else:
465
+ lines.append("No limitations statement was declared.")
466
+ return "\n".join(lines)
467
+
468
+
469
+ def _dictionary_markdown(
470
+ column_names: list[str],
471
+ semantics_by_name: dict[str, dict[str, Any]],
472
+ logical_by_name: dict[str, str],
473
+ ) -> str:
474
+ headings = ("Column", "Label", "Meaning", "Type", "Unit / scaling", "Values", "Evidence")
475
+ rows: list[str] = []
476
+
477
+ def table_value(value: Any) -> str:
478
+ return (
479
+ " ".join(("" if value is None else str(value)).split())
480
+ .replace("\\", r"\\")
481
+ .replace("|", r"\|")
482
+ )
483
+
484
+ for name in column_names:
485
+ record = semantics_by_name.get(name)
486
+ unit = _unit_metadata(record)
487
+ unit_text = unit["label"]
488
+ if unit.get("state") == "normalized":
489
+ reversible = "reversible" if unit.get("reversible") else "not reversible"
490
+ unit_text = f"{unit_text}: {unit.get('normalization')} ({reversible})"
491
+ categories = record.get("categories") if isinstance(record, dict) else None
492
+ category_status = (
493
+ record.get("categories_status", "unknown") if isinstance(record, dict) else "unknown"
494
+ )
495
+ category_text = str(category_status).replace("_", " ")
496
+ if isinstance(categories, list) and categories:
497
+ definitions = "; ".join(
498
+ f"{item.get('code')} = {item.get('label')}"
499
+ for item in categories
500
+ if isinstance(item, dict)
501
+ )
502
+ category_text = f"{category_text}: {definitions}"
503
+ refs = record.get("evidence_refs") if isinstance(record, dict) else None
504
+ evidence = "; ".join(
505
+ f"{item.get('kind')}: {item.get('ref')}"
506
+ for item in refs or []
507
+ if isinstance(item, dict)
508
+ )
509
+ values = (
510
+ name,
511
+ record.get("display_label", "Not declared")
512
+ if isinstance(record, dict)
513
+ else "Not declared",
514
+ record.get("description", "Not declared")
515
+ if isinstance(record, dict)
516
+ else "Not declared",
517
+ logical_by_name.get(name, "unknown"),
518
+ unit_text,
519
+ category_text,
520
+ evidence or "No evidence reference declared",
521
+ )
522
+ rows.append("| " + " | ".join(table_value(value) for value in values) + " |")
523
+ header = "| " + " | ".join(headings) + " |"
524
+ rule = "| " + " | ".join("---" for _heading in headings) + " |"
525
+ return "\n".join(("### Data dictionary", "", header, rule, *rows))
526
+
527
+
528
+ def _quality_markdown(quality: dict[str, Any]) -> str:
529
+ checks = quality.get("checks", []) if isinstance(quality, dict) else []
530
+ passed = sum(bool(check.get("passed")) for check in checks if isinstance(check, dict))
531
+ failed = [check for check in checks if isinstance(check, dict) and not check.get("passed")]
532
+ lines = [
533
+ "## Validation",
534
+ "",
535
+ "### Build checks",
536
+ "",
537
+ f"{passed} of {len(checks)} checks passed.",
538
+ ]
539
+ for check in failed:
540
+ lines.append(f"- {_esc(check.get('check_id', 'Check'))}: needs attention")
541
+ return "\n".join(lines)
542
+
543
+
544
+ def _join_markdown(join: dict[str, Any]) -> str:
545
+ keys = ", ".join(f"`{_esc(key)}`" for key in join.get("keys", [])) or "—"
546
+ lines = [
547
+ "### Join",
548
+ "",
549
+ f"- **cardinality:** {_esc(join.get('cardinality', '—'))}",
550
+ f"- **keys:** {keys}",
551
+ f"- **left:** `{_esc(join.get('left_source', '—'))}` "
552
+ f"({_esc(join.get('left_rows', '—'))} rows)",
553
+ f"- **right:** `{_esc(join.get('right_source', '—'))}` "
554
+ f"({_esc(join.get('right_rows', '—'))} rows)",
555
+ f"- **output rows:** {_esc(join.get('output_rows', '—'))}",
556
+ f"- **unmatched left rows:** {_esc(join.get('unmatched_left_rows', '—'))}",
557
+ ]
558
+ return "\n".join(lines)
559
+
560
+
561
+ def _operations_markdown(evidence: dict[str, Any]) -> str:
562
+ nodes = evidence.get("nodes", []) if isinstance(evidence, dict) else []
563
+ lines = ["## How this dataset was built", "", "### Deterministic operations", ""]
564
+ for node in nodes if isinstance(nodes, list) else []:
565
+ if not isinstance(node, dict):
566
+ continue
567
+ inputs = ", ".join(f"`{_esc(item)}`" for item in node.get("inputs", [])) or "source"
568
+ lines.append(
569
+ f"- **{_esc(node.get('node_id', 'node'))}** — "
570
+ f"`{_esc(node.get('operation', 'operation'))}` from {inputs}; "
571
+ f"{_esc(node.get('input_rows', '—'))} → {_esc(node.get('output_rows', '—'))} rows"
572
+ )
573
+ parameters = node.get("parameters")
574
+ if isinstance(parameters, dict):
575
+ rendered = json.dumps(
576
+ parameters,
577
+ ensure_ascii=False,
578
+ sort_keys=True,
579
+ separators=(",", ":"),
580
+ )
581
+ markdown_safe = rendered.replace("`", "\\u0060")
582
+ lines.append(f" - parameters: `{markdown_safe}`")
583
+ inspectable = evidence.get("schema_version") == pipeline.GRAPH_OPERATION_EVIDENCE_VERSION
584
+ lines.extend(
585
+ [
586
+ "",
587
+ (
588
+ "Exact parameters, ordering, lineage, engine identity, and logical digests are "
589
+ "sealed in `evidence/operations.json`."
590
+ if inspectable
591
+ else "Parameter digests, ordering, lineage, engine identity, and logical digests "
592
+ "are sealed in `evidence/operations.json`."
593
+ ),
594
+ ]
595
+ )
596
+ return "\n".join(lines)
597
+
598
+
599
+ def _sources_markdown(sources: tuple[dict[str, Any], ...]) -> str:
600
+ lines = ["## Provenance", "", "### Sources", ""]
601
+ for source in sources:
602
+ source = source if isinstance(source, dict) else {}
603
+ rights = source.get("rights", {})
604
+ status = rights.get("status", "unspecified") if isinstance(rights, dict) else "unspecified"
605
+ origin = source.get("origin", "")
606
+ lines.append(
607
+ f"- **{_esc(source.get('source_id', 'source'))}** — {_esc(origin) or 'local source'} "
608
+ f"({_esc(status).replace('_', ' ')})"
609
+ )
610
+ return "\n".join(lines)
611
+
612
+
613
+ def _provenance_metadata(inspection: Any) -> dict[str, str]:
614
+ """Build ``metadata.mostlyright.provenance`` from the rendered evidence members.
615
+
616
+ Four string fields in renderer order — ``source``, ``join``, ``revision``, ``cache`` — each
617
+ derived from the already-read build reports (``evidence/sources.json``, ``evidence/join.json``,
618
+ the verified manifest identity). No clock and no randomness: two renders of the same candidate
619
+ produce the same block, exactly like the cells. Values are raw text, not HTML-escaped — a
620
+ renderer escapes metadata at interpolation, so escaping here would double-encode.
621
+ """
622
+
623
+ source_ids = [
624
+ str(source.get("source_id", ""))
625
+ for source in inspection.sources
626
+ if isinstance(source, dict) and source.get("source_id")
627
+ ]
628
+ join = inspection.join if isinstance(inspection.join, dict) else {}
629
+ if join.get("schema_version") in {
630
+ pipeline.GRAPH_OPERATION_EVIDENCE_V1,
631
+ pipeline.GRAPH_OPERATION_EVIDENCE_VERSION,
632
+ }:
633
+ nodes = join.get("nodes", [])
634
+ join_text = f"{len(nodes) if isinstance(nodes, list) else 0} graph operations"
635
+ else:
636
+ keys = [str(key) for key in (join.get("keys") or [])]
637
+ cardinality = str(join.get("cardinality") or "")
638
+ if cardinality and keys:
639
+ join_text = f"{cardinality} on {', '.join(keys)}"
640
+ else:
641
+ join_text = cardinality or "none"
642
+ snapshots = len(source_ids)
643
+ noun = "snapshot" if snapshots == 1 else "snapshots"
644
+ digest = inspection.candidate_digest[:_PROVENANCE_DIGEST_PREFIX]
645
+ return {
646
+ "source": ", ".join(source_ids) or "none",
647
+ "join": join_text,
648
+ # Digest first: a renderer dims the parenthetical tail, and the identity is the digest —
649
+ # the manifest version is the format constant.
650
+ "revision": f"{digest} ({inspection.manifest_version})",
651
+ "cache": f"{snapshots} raw {noun} in raw/",
652
+ }
653
+
654
+
655
+ # --- all-column analytics ---------------------------------------------------------------------
656
+
657
+
658
+ def _number(value: int | float | Decimal) -> str:
659
+ """Format a number without first narrowing an integer through binary64."""
660
+
661
+ if isinstance(value, int):
662
+ return f"{value:,}"
663
+ if isinstance(value, Decimal):
664
+ if not value.is_finite():
665
+ return "Not finite"
666
+ # Keep exact fixed-point rendering across the complete int64 range, while avoiding a
667
+ # hundreds-of-characters label if a binary64-derived Decimal reaches an extreme exponent.
668
+ if value and (value.adjusted() >= 20 or value.adjusted() <= -5):
669
+ return _trim_scientific(format(value, ".4g"))
670
+ if value == value.to_integral_value():
671
+ return f"{int(value):,}"
672
+ return format(value, ",f")
673
+ if not math.isfinite(value):
674
+ return "Not finite"
675
+ # ``float.is_integer`` is also true for 1e308. Expanding that value through ``int`` creates a
676
+ # 309-character chart label, so reserve exact-looking grouped integers for ordinary magnitudes.
677
+ if value and (abs(value) >= 1e20 or abs(value) < 1e-4):
678
+ return _trim_scientific(f"{value:.4g}")
679
+ if math.isfinite(value) and value.is_integer():
680
+ return f"{int(value):,}"
681
+ return f"{value:,.4g}"
682
+
683
+
684
+ def _trim_scientific(value: str) -> str:
685
+ """Normalize a compact general-format number without changing its significant digits."""
686
+
687
+ rendered = value.lower()
688
+ if "e" not in rendered:
689
+ return rendered
690
+ mantissa, exponent = rendered.split("e", 1)
691
+ mantissa = mantissa.rstrip("0").rstrip(".")
692
+ return f"{mantissa}e{exponent}"
693
+
694
+
695
+ def _unit_metadata(semantics: dict[str, Any] | None) -> dict[str, Any]:
696
+ if semantics is None:
697
+ return {
698
+ "value": None,
699
+ "symbol": None,
700
+ "label": "Not declared",
701
+ "status": "unknown",
702
+ "state": "unknown",
703
+ "normalization": None,
704
+ "reversible": None,
705
+ }
706
+ raw = str(semantics.get("unit", "none"))
707
+ semantic_type = str(semantics.get("semantic_type", ""))
708
+ state = semantics.get("unit_state")
709
+ if state not in UNIT_STATES:
710
+ state = (
711
+ "declared"
712
+ if raw != "none"
713
+ else ("unknown" if semantic_type == "measure" else "not_applicable")
714
+ )
715
+ if state == "normalized":
716
+ return {
717
+ "value": raw,
718
+ "symbol": None,
719
+ "label": _UNIT_STATE_DISPLAY[state],
720
+ "status": state,
721
+ "state": state,
722
+ "normalization": semantics.get("normalization"),
723
+ "reversible": semantics.get("reversible"),
724
+ }
725
+ if state == "not_applicable":
726
+ return {
727
+ "value": raw,
728
+ "symbol": None,
729
+ "label": _UNIT_STATE_DISPLAY[state],
730
+ "status": "not_applicable",
731
+ "state": state,
732
+ "normalization": None,
733
+ "reversible": None,
734
+ }
735
+ if state == "unknown":
736
+ return {
737
+ "value": raw,
738
+ "symbol": None,
739
+ "label": _UNIT_STATE_DISPLAY[state],
740
+ "status": "unknown",
741
+ "state": state,
742
+ "normalization": None,
743
+ "reversible": None,
744
+ }
745
+ # A declared unit that is neither one of the written-out names nor a code the grammar resolves
746
+ # cannot have got past Recipe validation. Keep this branch fail-visible for a caller building a
747
+ # fixture without going through that validation.
748
+ if raw not in _UNIT_DISPLAY and raw not in _CODE_DISPLAY and not is_declarable_unit(raw):
749
+ return {
750
+ "value": raw,
751
+ "symbol": None,
752
+ "label": "Unknown unit",
753
+ "status": "unknown",
754
+ "state": "unknown",
755
+ "normalization": None,
756
+ "reversible": None,
757
+ }
758
+ symbol, label = _UNIT_DISPLAY.get(raw) or _CODE_DISPLAY.get(raw) or (raw, raw)
759
+ return {
760
+ "value": raw,
761
+ "symbol": symbol,
762
+ "label": label,
763
+ "status": "declared",
764
+ "state": "declared",
765
+ "normalization": None,
766
+ "reversible": None,
767
+ }
768
+
769
+
770
+ def _value_text(value: Any) -> str:
771
+ if value is None:
772
+ return "null"
773
+ if isinstance(value, float):
774
+ if math.isnan(value):
775
+ return "NaN"
776
+ # ``repr`` is the shortest round-trippable spelling. Category labels therefore cannot
777
+ # merge near-equal binary64 values merely because the chart uses compact typography.
778
+ if value == 0.0:
779
+ return "0.0"
780
+ return repr(value)
781
+ return str(value)
782
+
783
+
784
+ def _analysis_note(analysis_rows: int, total_rows: int) -> str:
785
+ if analysis_rows >= total_rows:
786
+ return f"{analysis_rows:,} observations · all rows"
787
+ return f"{analysis_rows:,} observations · first {analysis_rows:,} of {total_rows:,} rows"
788
+
789
+
790
+ def _histogram(
791
+ values: list[int | float],
792
+ *,
793
+ unit_symbol: str | None,
794
+ analysis_rows: int,
795
+ total_rows: int,
796
+ ) -> dict[str, Any]:
797
+ if not values:
798
+ return {"title": "Distribution", "note": "No non-missing values", "items": []}
799
+ minimum, maximum = min(values), max(values)
800
+ suffix = f" {unit_symbol}" if unit_symbol else ""
801
+ if minimum == maximum:
802
+ return {
803
+ "title": "Distribution",
804
+ "note": _analysis_note(analysis_rows, total_rows),
805
+ "items": [{"label": f"{_number(minimum)}{suffix}", "count": len(values)}],
806
+ }
807
+ bins = min(8, max(4, round(math.sqrt(len(values)))))
808
+ counts = [0] * bins
809
+ items = []
810
+ if all(isinstance(value, int) for value in values):
811
+ # Integer arithmetic throughout: neither +/-2**53 nor int64 extrema are rounded through a
812
+ # float. Inclusive integer buckets are gap-free and cover both extrema exactly.
813
+ integer_minimum = int(minimum)
814
+ integer_maximum = int(maximum)
815
+ span = integer_maximum - integer_minimum + 1
816
+ bins = min(bins, span)
817
+ counts = [0] * bins
818
+ for value in values:
819
+ index = min(bins - 1, ((int(value) - integer_minimum) * bins) // span)
820
+ counts[index] += 1
821
+ for index, count in enumerate(counts):
822
+ low = integer_minimum + ((span * index + bins - 1) // bins)
823
+ high = integer_minimum + ((span * (index + 1) + bins - 1) // bins) - 1
824
+ items.append({"label": f"{_number(low)} to {_number(high)}{suffix}", "count": count})
825
+ else:
826
+ float_minimum = float(minimum)
827
+ float_maximum = float(maximum)
828
+ crosses_zero = float_minimum < 0.0 < float_maximum
829
+ if crosses_zero:
830
+ # Scaling first keeps the normalized span in [1, 2]. Directly subtracting valid
831
+ # binary64 extrema (for example -1e308 and 1e308) would overflow to infinity.
832
+ scale = max(-float_minimum, float_maximum)
833
+ scaled_minimum = float_minimum / scale
834
+ scaled_maximum = float_maximum / scale
835
+ scaled_span = scaled_maximum - scaled_minimum
836
+
837
+ def position(value: int | float) -> float:
838
+ return ((float(value) / scale) - scaled_minimum) / scaled_span
839
+
840
+ def boundary(fraction: float) -> float:
841
+ # A convex combination never exceeds either finite endpoint. Unlike
842
+ # ``minimum + fraction * (maximum - minimum)``, it needs no overflowing span.
843
+ return float_minimum * (1.0 - fraction) + float_maximum * fraction
844
+
845
+ else:
846
+ span = float_maximum - float_minimum
847
+
848
+ def position(value: int | float) -> float:
849
+ return (float(value) - float_minimum) / span
850
+
851
+ def boundary(fraction: float) -> float:
852
+ return float_minimum + span * fraction
853
+
854
+ for value in values:
855
+ index = min(bins - 1, max(0, int(position(value) * bins)))
856
+ counts[index] += 1
857
+ for index, count in enumerate(counts):
858
+ low = boundary(index / bins)
859
+ high = boundary((index + 1) / bins)
860
+ items.append({"label": f"{_number(low)} to {_number(high)}{suffix}", "count": count})
861
+ return {
862
+ "title": "Distribution",
863
+ "note": _analysis_note(analysis_rows, total_rows),
864
+ "items": items,
865
+ }
866
+
867
+
868
+ def _is_missing(value: Any) -> bool:
869
+ """Notebook policy: null and every NaN payload are missing; signed zero is observed."""
870
+
871
+ return value is None or (isinstance(value, float) and math.isnan(value))
872
+
873
+
874
+ def _distinct_key(value: Any) -> tuple[str, Any]:
875
+ """Typed exact identity for cardinality/counting, separate from display formatting.
876
+
877
+ All NaNs are excluded by ``_is_missing``. IEEE signed zeros intentionally form one value,
878
+ matching Arrow/Python equality semantics for an analytical distinct count.
879
+ """
880
+
881
+ if _is_missing(value): # pragma: no cover - guarded by callers, retained as an invariant
882
+ raise ValueError("missing values do not have a distinct key")
883
+ if isinstance(value, bool):
884
+ return ("boolean", value)
885
+ if isinstance(value, int):
886
+ return ("int64", value)
887
+ if isinstance(value, float):
888
+ return ("float64", 0.0 if value == 0.0 else value)
889
+ return (type(value).__qualname__, value)
890
+
891
+
892
+ def _value_counts(
893
+ values: list[Any],
894
+ *,
895
+ title: str = "Value counts",
896
+ analysis_rows: int,
897
+ total_rows: int,
898
+ ) -> dict[str, Any]:
899
+ keyed = Counter(_distinct_key(value) for value in values if not _is_missing(value))
900
+ counts = [(_value_text(key[1]), count) for key, count in keyed.items()]
901
+ ordered = sorted(counts, key=lambda item: (-item[1], item[0]))
902
+ visible = ordered[:_MAX_CATEGORY_BARS]
903
+ hidden = max(0, len(ordered) - len(visible))
904
+ note = _analysis_note(analysis_rows, total_rows)
905
+ if hidden:
906
+ note += f" · top {_MAX_CATEGORY_BARS} of {len(ordered):,} values"
907
+ return {
908
+ "title": title,
909
+ "note": note,
910
+ "items": [{"label": label, "count": count} for label, count in visible],
911
+ }
912
+
913
+
914
+ def _column_summary(
915
+ values: list[Any],
916
+ logical: str,
917
+ unit: dict[str, Any],
918
+ *,
919
+ analysis_rows: int,
920
+ total_rows: int,
921
+ ) -> tuple[dict[str, str], dict[str, Any]]:
922
+ non_null = [value for value in values if not _is_missing(value)]
923
+ if not non_null:
924
+ return (
925
+ {"label": "Observed", "value": "0", "detail": "No non-missing values"},
926
+ {"title": "Distribution", "note": "No non-missing values", "items": []},
927
+ )
928
+ symbol = unit.get("symbol") if isinstance(unit.get("symbol"), str) else None
929
+ suffix = f" {symbol}" if symbol else ""
930
+ if logical in {"int64", "float64"}:
931
+ numeric = [
932
+ value
933
+ for value in non_null
934
+ if isinstance(value, (int, float))
935
+ and not isinstance(value, bool)
936
+ and (isinstance(value, int) or math.isfinite(value))
937
+ ]
938
+ if numeric:
939
+ if logical == "int64" and all(isinstance(value, int) for value in numeric):
940
+ ordered = sorted(numeric)
941
+ middle = len(ordered) // 2
942
+ if len(ordered) % 2:
943
+ median: int | float | Decimal = ordered[middle]
944
+ else:
945
+ total = ordered[middle - 1] + ordered[middle]
946
+ median = total // 2 if total % 2 == 0 else Decimal(total) / Decimal(2)
947
+ else:
948
+ ordered_floats = sorted(float(value) for value in numeric)
949
+ middle = len(ordered_floats) // 2
950
+ if len(ordered_floats) % 2:
951
+ median = ordered_floats[middle]
952
+ else:
953
+ low = ordered_floats[middle - 1]
954
+ high = ordered_floats[middle]
955
+ if low < 0.0 < high:
956
+ # Opposite signs make the sum safe even when the direct span overflows.
957
+ median = (low + high) / 2.0
958
+ else:
959
+ # Same-sign subtraction is finite; this also preserves repeated extrema.
960
+ median = low + (high - low) / 2.0
961
+ summary = {
962
+ "label": "Median",
963
+ "value": f"{_number(median)}{suffix}",
964
+ "detail": f"{_number(min(numeric))} → {_number(max(numeric))}{suffix}",
965
+ }
966
+ return summary, _histogram(
967
+ numeric,
968
+ unit_symbol=symbol,
969
+ analysis_rows=analysis_rows,
970
+ total_rows=total_rows,
971
+ )
972
+ if logical in {"date", "timestamp_utc"}:
973
+ rendered = sorted(_value_text(value) for value in non_null)
974
+ summary = {
975
+ "label": "Range",
976
+ "value": rendered[0],
977
+ "detail": f"through {rendered[-1]}",
978
+ }
979
+ return summary, _value_counts(
980
+ non_null,
981
+ title="Observations over time",
982
+ analysis_rows=analysis_rows,
983
+ total_rows=total_rows,
984
+ )
985
+ keyed_counts = Counter(_distinct_key(value) for value in non_null)
986
+ mode_key, count = sorted(
987
+ keyed_counts.items(), key=lambda item: (-item[1], _value_text(item[0][1]))
988
+ )[0]
989
+ mode = _value_text(mode_key[1])
990
+ summary = {
991
+ "label": "Most common",
992
+ "value": mode,
993
+ "detail": f"{count:,} of {len(non_null):,} observed",
994
+ }
995
+ title = "Boolean counts" if logical == "boolean" else "Value counts"
996
+ return summary, _value_counts(
997
+ non_null,
998
+ title=title,
999
+ analysis_rows=analysis_rows,
1000
+ total_rows=total_rows,
1001
+ )
1002
+
1003
+
1004
+ def _semantics_context(
1005
+ candidate_dir: Path,
1006
+ members: set[str],
1007
+ column_names: list[str],
1008
+ *,
1009
+ member_bytes: dict[str, bytes] | None = None,
1010
+ verified_recipe: dict[str, Any] | None = None,
1011
+ ) -> tuple[dict[str, Any], dict[str, dict[str, Any]]]:
1012
+ """Derive semantic authority only from verified bindings or sealed local declarations."""
1013
+
1014
+ document: dict[str, Any] | None = None
1015
+ records: list[Any] = []
1016
+ if "recipe.json" in members:
1017
+ authority = "approved_recipe" if isinstance(verified_recipe, dict) else "none"
1018
+ raw_document = verified_recipe.get("table_semantics") if verified_recipe else None
1019
+ if isinstance(raw_document, dict):
1020
+ document = raw_document
1021
+ records = raw_document.get("columns", [])
1022
+ elif verified_recipe and isinstance(verified_recipe.get("output_semantics"), list):
1023
+ records = verified_recipe["output_semantics"]
1024
+ elif "semantics.json" in members:
1025
+ authority = "declared_local_plan"
1026
+ raw_document = _read_member_json(
1027
+ candidate_dir,
1028
+ "semantics.json",
1029
+ members,
1030
+ member_bytes=member_bytes,
1031
+ )
1032
+ if isinstance(raw_document, dict):
1033
+ document = raw_document
1034
+ records = raw_document.get("columns", [])
1035
+ else:
1036
+ authority = "none"
1037
+
1038
+ by_name = {
1039
+ str(item.get("name")): item
1040
+ for item in records
1041
+ if isinstance(item, dict) and isinstance(item.get("name"), str)
1042
+ }
1043
+ missing = sum(
1044
+ 1
1045
+ for name in column_names
1046
+ if not isinstance(by_name.get(name, {}).get("display_label"), str)
1047
+ or not by_name[name].get("display_label")
1048
+ or not isinstance(by_name[name].get("description"), str)
1049
+ or not by_name[name].get("description")
1050
+ or _unit_metadata(by_name[name]).get("state") == "unknown"
1051
+ )
1052
+ complete = document is not None and missing == 0
1053
+ completeness = "absent" if authority == "none" else "complete" if complete else "partial"
1054
+ total = len(column_names)
1055
+ if completeness == "absent":
1056
+ message = (
1057
+ "This Build declares no column meanings. Nothing here is inferred from column names."
1058
+ )
1059
+ elif authority == "approved_recipe" and complete:
1060
+ message = "Column meanings are declared in an approved Recipe and sealed with this Build."
1061
+ elif authority == "approved_recipe":
1062
+ message = (
1063
+ "Some column meanings are declared in the approved Recipe. "
1064
+ f"{missing} of {total} columns have no declared meaning."
1065
+ )
1066
+ elif complete:
1067
+ message = (
1068
+ "Column meanings were declared with this local plan and sealed with this Build. "
1069
+ "They are not human-approved."
1070
+ )
1071
+ else:
1072
+ message = (
1073
+ "Some column meanings were declared with this local plan. "
1074
+ f"{missing} of {total} columns have no declared meaning. They are not human-approved."
1075
+ )
1076
+ context = {
1077
+ "schema_version": "mostlyright.column-semantics.v1",
1078
+ "authority": authority,
1079
+ "completeness": completeness,
1080
+ "message": message,
1081
+ "missing_column_count": missing if authority != "none" else total,
1082
+ "column_count": total,
1083
+ "render_scope": "full",
1084
+ }
1085
+ if document is not None:
1086
+ context.update(
1087
+ {
1088
+ "summary": document.get("summary"),
1089
+ "grain_statement": document.get("grain_statement"),
1090
+ "coverage": document.get("coverage"),
1091
+ "limitations": document.get("limitations"),
1092
+ }
1093
+ )
1094
+ return context, by_name
1095
+
1096
+
1097
+ def _fit_column_profile_payload(payload: dict[str, Any]) -> dict[str, Any]:
1098
+ """Reduce optional semantic detail deterministically until the portable MIME is bounded."""
1099
+
1100
+ from mostlyright.data_harness.nbrender.outputs_data import is_valid_column_profiles_payload
1101
+
1102
+ if is_valid_column_profiles_payload(payload):
1103
+ return payload
1104
+ reduced = deepcopy(payload)
1105
+ semantics = reduced.get("semantics")
1106
+ if isinstance(semantics, dict):
1107
+ semantics["render_scope"] = "reduced"
1108
+ for key in ("coverage", "limitations"):
1109
+ semantics.pop(key, None)
1110
+ for column in reduced.get("columns", []):
1111
+ if isinstance(column, dict):
1112
+ for key in ("description", "categories", "evidence_refs"):
1113
+ column.pop(key, None)
1114
+ if is_valid_column_profiles_payload(reduced):
1115
+ return reduced
1116
+ minimal = deepcopy(reduced)
1117
+ semantics = minimal.get("semantics")
1118
+ if isinstance(semantics, dict):
1119
+ semantics["render_scope"] = "minimal"
1120
+ for key in ("summary", "grain_statement"):
1121
+ semantics.pop(key, None)
1122
+ for column in minimal.get("columns", []):
1123
+ if isinstance(column, dict):
1124
+ for key in ("display_label", "categories_status"):
1125
+ column.pop(key, None)
1126
+ if is_valid_column_profiles_payload(minimal):
1127
+ return minimal
1128
+ legacy = deepcopy(minimal)
1129
+ for column in legacy.get("columns", []):
1130
+ if isinstance(column, dict):
1131
+ column["unit"] = {
1132
+ key: column.get("unit", {}).get(key)
1133
+ for key in ("value", "symbol", "label", "status")
1134
+ }
1135
+ if is_valid_column_profiles_payload(legacy):
1136
+ return legacy
1137
+
1138
+ # A verified string column may legitimately contain a very large value. Summary and chart
1139
+ # labels derived from ten such columns can exceed the portable 1 MiB rich-MIME ceiling even
1140
+ # after semantic detail is reduced. Keep the analytical shape and semantic authority, but
1141
+ # deterministically bound presentation strings before considering refusal. Complete values
1142
+ # never belonged in this MIME contract; the portable fallbacks describe the dataset rather
1143
+ # than repeating profiled cell contents.
1144
+ bounded_rich = deepcopy(legacy)
1145
+ for column in bounded_rich.get("columns", []):
1146
+ if not isinstance(column, dict):
1147
+ continue
1148
+ for field in ("summary", "source"):
1149
+ value = column.get(field)
1150
+ if isinstance(value, dict):
1151
+ for key, item in tuple(value.items()):
1152
+ if isinstance(item, str):
1153
+ value[key] = _bounded_rich_label(item)
1154
+ chart = column.get("chart")
1155
+ if isinstance(chart, dict):
1156
+ for key in ("title", "note"):
1157
+ if isinstance(chart.get(key), str):
1158
+ chart[key] = _bounded_rich_label(chart[key])
1159
+ items = chart.get("items")
1160
+ if isinstance(items, list):
1161
+ for item in items:
1162
+ if isinstance(item, dict) and isinstance(item.get("label"), str):
1163
+ item["label"] = _bounded_rich_label(item["label"])
1164
+ unit = column.get("unit")
1165
+ if isinstance(unit, dict):
1166
+ for key, item in tuple(unit.items()):
1167
+ if isinstance(item, str):
1168
+ unit[key] = _bounded_rich_label(item)
1169
+ operations = column.get("transformations")
1170
+ if isinstance(operations, list):
1171
+ column["transformations"] = [
1172
+ _bounded_rich_label(item) if isinstance(item, str) else item for item in operations
1173
+ ]
1174
+ if not is_valid_column_profiles_payload(bounded_rich):
1175
+ raise ValueError("generated column-profile MIME exceeds its bounded contract")
1176
+ return bounded_rich
1177
+
1178
+
1179
+ def _bounded_rich_label(value: str, *, limit: int = 128) -> str:
1180
+ """Bound one derived rich-MIME label with an explicit deterministic marker."""
1181
+
1182
+ if len(value) <= limit:
1183
+ return value
1184
+ marker = " [truncated]"
1185
+ return value[: limit - len(marker)] + marker
1186
+
1187
+
1188
+ def _column_profiles(
1189
+ candidate_dir: Path,
1190
+ table: Any,
1191
+ inspection: Any,
1192
+ profile: dict[str, Any],
1193
+ members: set[str],
1194
+ *,
1195
+ member_bytes: dict[str, bytes] | None = None,
1196
+ verified_recipe: dict[str, Any] | None = None,
1197
+ fit_payload: bool = True,
1198
+ ) -> dict[str, Any]:
1199
+ """Compute bounded visual analytics over the verified Parquet and bind declared units."""
1200
+
1201
+ semantics_context, semantics_by_name = _semantics_context(
1202
+ candidate_dir,
1203
+ members,
1204
+ list(table.column_names),
1205
+ member_bytes=member_bytes,
1206
+ verified_recipe=verified_recipe,
1207
+ )
1208
+ logical_by_name = {
1209
+ str(item.get("name")): str(item.get("logical_type", "unknown"))
1210
+ for item in profile.get("logical_schema", [])
1211
+ if isinstance(item, dict)
1212
+ }
1213
+ type_by_name = (
1214
+ {str(name): str(value) for name, value in profile.get("types", {}).items()}
1215
+ if isinstance(profile.get("types"), dict)
1216
+ else {}
1217
+ )
1218
+ nulls_by_name = (
1219
+ profile.get("null_counts", {}) if isinstance(profile.get("null_counts"), dict) else {}
1220
+ )
1221
+ rows = int(profile.get("row_count", table.num_rows) or 0)
1222
+ columns: list[dict[str, Any]] = []
1223
+ for name in table.column_names:
1224
+ semantics = semantics_by_name.get(name)
1225
+ logical = str(
1226
+ logical_by_name.get(name)
1227
+ or (semantics or {}).get("logical_type")
1228
+ or type_by_name.get(name)
1229
+ or "unknown"
1230
+ )
1231
+ physical = PHYSICAL_TYPES.get(logical, str(table.schema.field(name).type))
1232
+ unit = _unit_metadata(semantics)
1233
+ array = table[name]
1234
+ sample_size = min(table.num_rows, _PROFILE_SAMPLE_ROWS)
1235
+ sample = array.slice(0, sample_size).to_pylist()
1236
+ null_count = int(nulls_by_name.get(name, array.null_count) or 0)
1237
+ nan_count = 0
1238
+ if logical == "float64":
1239
+ import pyarrow.compute as pc
1240
+
1241
+ # Arrow's sealed profile counts nulls. The notebook's visual missingness policy also
1242
+ # treats every NaN payload as missing, so calculate that second exact count over the
1243
+ # same verified Arrow snapshot and state both parts in the payload.
1244
+ nan_count = int(pc.sum(pc.is_nan(array)).as_py() or 0)
1245
+ missing = null_count + nan_count
1246
+ non_missing_sample = [value for value in sample if not _is_missing(value)]
1247
+ distinct_sample = len({_distinct_key(value) for value in non_missing_sample})
1248
+ # Exact for small/medium columns; high-cardinality columns are clearly labelled sampled.
1249
+ if table.num_rows <= _PROFILE_SAMPLE_ROWS:
1250
+ distinct = distinct_sample
1251
+ distinct_status = "exact"
1252
+ else:
1253
+ distinct = distinct_sample
1254
+ distinct_status = f"sampled from first {sample_size:,} rows"
1255
+ summary, chart = _column_summary(
1256
+ sample,
1257
+ logical,
1258
+ unit,
1259
+ analysis_rows=sample_size,
1260
+ total_rows=table.num_rows,
1261
+ )
1262
+ detail = inspection.lineage.get(name, {}) if isinstance(inspection.lineage, dict) else {}
1263
+ detail = detail if isinstance(detail, dict) else {}
1264
+ column = {
1265
+ "name": name,
1266
+ "logical_type": logical,
1267
+ "physical_type": physical,
1268
+ "semantic_type": (semantics or {}).get("semantic_type"),
1269
+ "unit": unit,
1270
+ "row_count": rows,
1271
+ "null_count": null_count,
1272
+ "nan_count": nan_count,
1273
+ "missing_count": missing,
1274
+ "missing_rate": (missing / rows) if rows else 0.0,
1275
+ "missing_policy": "null_or_nan",
1276
+ "distinct_count": distinct,
1277
+ "distinct_rate": (
1278
+ distinct / max(1, rows if distinct_status == "exact" else sample_size)
1279
+ ),
1280
+ "distinct_status": distinct_status,
1281
+ "distinct_policy": "typed_exact; null_and_nan_excluded; signed_zero_collapsed",
1282
+ "sample_size": sample_size,
1283
+ "summary": summary,
1284
+ "chart": chart,
1285
+ "source": {
1286
+ "source_id": detail.get("source_id"),
1287
+ "column": detail.get("source_column", name),
1288
+ },
1289
+ "transformations": list(detail.get("operations", []))
1290
+ if isinstance(detail.get("operations"), list)
1291
+ else [],
1292
+ }
1293
+ if semantics is not None:
1294
+ column.update(
1295
+ {
1296
+ "display_label": semantics.get("display_label"),
1297
+ "description": semantics.get("description"),
1298
+ "categories": semantics.get("categories"),
1299
+ "categories_status": semantics.get("categories_status"),
1300
+ "evidence_refs": semantics.get("evidence_refs"),
1301
+ }
1302
+ )
1303
+ columns.append(column)
1304
+ payload = {
1305
+ "schema_version": "mostlyright.column-profile.v1",
1306
+ "table_digest": inspection.table_sha256,
1307
+ "row_count": rows,
1308
+ "column_count": len(columns),
1309
+ "analysis_scope": "all rows"
1310
+ if table.num_rows <= _PROFILE_SAMPLE_ROWS
1311
+ else f"first {_PROFILE_SAMPLE_ROWS:,} rows for charts",
1312
+ "columns": columns,
1313
+ "semantics": semantics_context,
1314
+ }
1315
+ return _fit_column_profile_payload(payload) if fit_payload else payload
1316
+
1317
+
1318
+ # --- reading the sealed candidate --------------------------------------------------------------
1319
+
1320
+
1321
+ def _read_member_json(
1322
+ candidate_dir: Path,
1323
+ relative: str,
1324
+ members: set[str],
1325
+ *,
1326
+ member_bytes: dict[str, bytes] | None = None,
1327
+ ) -> Any | None:
1328
+ """Parse ``relative`` from ``candidate_dir`` only when it is a recorded, present member."""
1329
+
1330
+ if relative not in members:
1331
+ return None
1332
+ try:
1333
+ if member_bytes is not None:
1334
+ raw = member_bytes.get(relative)
1335
+ return json.loads(raw.decode("utf-8")) if raw is not None else None
1336
+ path = candidate_dir / relative
1337
+ if not path.exists():
1338
+ return None
1339
+ return json.loads(path.read_text(encoding="utf-8"))
1340
+ except (OSError, ValueError):
1341
+ return None
1342
+
1343
+
1344
+ def _read_dataset(
1345
+ candidate_dir: Path,
1346
+ limit: int,
1347
+ *,
1348
+ member_bytes: dict[str, bytes] | None = None,
1349
+ ) -> tuple[Any, list[dict[str, Any]], list[str]]:
1350
+ """Read the verified Parquet once, returning the table and a native head sample."""
1351
+
1352
+ import pyarrow.parquet as pq # lazy — the already-pinned base dependency
1353
+
1354
+ if member_bytes is not None:
1355
+ import pyarrow as pa
1356
+
1357
+ table = pq.read_table(pa.BufferReader(member_bytes["data/table.parquet"]))
1358
+ else:
1359
+ table = pq.read_table(candidate_dir / "data" / "table.parquet")
1360
+ sample = table.slice(0, limit).to_pylist()
1361
+ return table, sample, list(table.column_names)
1362
+
1363
+
1364
+ # --- notebook assembly -------------------------------------------------------------------------
1365
+
1366
+
1367
+ def build_notebook(candidate_dir: Path) -> dict[str, Any]:
1368
+ """Render a sealed candidate from one retained replay-verified snapshot."""
1369
+
1370
+ candidate_dir = Path(candidate_dir)
1371
+ with pipeline.open_verified_snapshot(candidate_dir.parent) as verified_snapshot:
1372
+ return _build_notebook_from_snapshot(candidate_dir, verified_snapshot)
1373
+
1374
+
1375
+ def _build_notebook_from_snapshot(
1376
+ candidate_dir: Path,
1377
+ verified_snapshot: pipeline.VerifiedSnapshot,
1378
+ ) -> dict[str, Any]:
1379
+ """Derive notebook bytes without reopening any candidate member pathname."""
1380
+
1381
+ member_bytes = dict(verified_snapshot.members)
1382
+ inspection = verified_snapshot.inspection()
1383
+ members = set(inspection.member_paths)
1384
+ profile = (
1385
+ _read_member_json(
1386
+ candidate_dir,
1387
+ "evidence/profile.json",
1388
+ members,
1389
+ member_bytes=member_bytes,
1390
+ )
1391
+ or {}
1392
+ )
1393
+ table = verified_snapshot.parquet_file().read()
1394
+ sample_rows = table.slice(0, _SAMPLE_ROWS).to_pylist()
1395
+ sample_columns = list(table.column_names)
1396
+ verified_recipe: dict[str, Any] | None = None
1397
+ if "recipe.json" in members:
1398
+ try:
1399
+ bindings = recipe_contracts.verify_recipe_authority_snapshot(verified_snapshot)
1400
+ except recipe_contracts.RecipeError:
1401
+ # Generic Build verification proves bytes, not human approval. A malformed or
1402
+ # unbound Recipe candidate remains inspectable, but it must not gain approved copy.
1403
+ verified_recipe = None
1404
+ else:
1405
+ verified_recipe = bindings["recipe"].to_dict()
1406
+
1407
+ full_column_profiles = _column_profiles(
1408
+ candidate_dir,
1409
+ table,
1410
+ inspection,
1411
+ profile,
1412
+ members,
1413
+ member_bytes=member_bytes,
1414
+ verified_recipe=verified_recipe,
1415
+ fit_payload=False,
1416
+ )
1417
+ column_profiles = _fit_column_profile_payload(full_column_profiles)
1418
+ semantics_context, semantics_by_name = _semantics_context(
1419
+ candidate_dir,
1420
+ members,
1421
+ sample_columns,
1422
+ member_bytes=member_bytes,
1423
+ verified_recipe=verified_recipe,
1424
+ )
1425
+ logical_by_name = {
1426
+ str(item.get("name")): str(item.get("logical_type", "unknown"))
1427
+ for item in profile.get("logical_schema", [])
1428
+ if isinstance(item, dict)
1429
+ }
1430
+
1431
+ cells: list[dict[str, Any]] = [
1432
+ _markdown_cell(
1433
+ _header_markdown(
1434
+ inspection.question,
1435
+ inspection.row_count,
1436
+ len(inspection.columns),
1437
+ )
1438
+ ),
1439
+ _markdown_cell(_overview_markdown(semantics_context)),
1440
+ _markdown_cell(
1441
+ "## Data\n\n### Preview\n\nThe first rows are embedded so the result is readable "
1442
+ "without starting Python. Run the cell to load the complete Parquet file."
1443
+ ),
1444
+ _code_cell(
1445
+ "import pandas as pd\n"
1446
+ # The sidecar is a run-dir SIBLING of candidate/, so every relative path a runnable
1447
+ # cell uses must go through candidate/ to resolve from the notebook's own directory.
1448
+ "df = pd.read_parquet('candidate/data/table.parquet')\n"
1449
+ f"df.head({_SAMPLE_ROWS}) # first rows",
1450
+ (_html_table(sample_rows, sample_columns),),
1451
+ ),
1452
+ _markdown_cell(_dictionary_markdown(sample_columns, semantics_by_name, logical_by_name)),
1453
+ _markdown_cell(
1454
+ "### Column profiles\n\nSelect a column to inspect its type, unit or scaling state, "
1455
+ "missingness, "
1456
+ "cardinality, distribution, source, and transformations. Charts are computed from "
1457
+ "the verified Parquet; semantic claims come only from sealed Build members."
1458
+ ),
1459
+ _code_cell(
1460
+ "# visual profile for every column\ndf.describe(include='all')",
1461
+ (
1462
+ _display_column_profiles(
1463
+ column_profiles,
1464
+ fallback_payload=full_column_profiles,
1465
+ ),
1466
+ ),
1467
+ ),
1468
+ _markdown_cell(_quality_markdown(inspection.quality)),
1469
+ ]
1470
+
1471
+ if isinstance(inspection.join, dict) and inspection.join:
1472
+ if inspection.join.get("schema_version") in {
1473
+ pipeline.GRAPH_OPERATION_EVIDENCE_V1,
1474
+ pipeline.GRAPH_OPERATION_EVIDENCE_VERSION,
1475
+ }:
1476
+ cells.append(_markdown_cell(_operations_markdown(inspection.join)))
1477
+ else:
1478
+ cells.append(_markdown_cell(_join_markdown(inspection.join)))
1479
+
1480
+ cells.append(_markdown_cell(_sources_markdown(inspection.sources)))
1481
+
1482
+ cells.extend(
1483
+ [
1484
+ _markdown_cell(
1485
+ "## Analysis\n\nThe following cells are optional local queries. They can be "
1486
+ "edited without changing the packaged Parquet data."
1487
+ ),
1488
+ _code_cell("df.describe(include='all')"),
1489
+ _code_cell("df.dtypes"),
1490
+ _code_cell("df[df.columns[0]].value_counts().head(15)"),
1491
+ ]
1492
+ )
1493
+
1494
+ notebook = {
1495
+ "cells": cells,
1496
+ "metadata": {
1497
+ "mostlyright": {
1498
+ "provenance": _provenance_metadata(inspection),
1499
+ "column_profile_mime": _COLUMN_PROFILE_MIME,
1500
+ }
1501
+ },
1502
+ "nbformat": _NBFORMAT_MAJOR,
1503
+ "nbformat_minor": _NBFORMAT_MINOR,
1504
+ }
1505
+ verified_snapshot.validate()
1506
+ return notebook
1507
+
1508
+
1509
+ def render_table_notebook(
1510
+ run_dir: Path,
1511
+ *,
1512
+ replace_existing: bool = True,
1513
+ _run_fd: int | None = None,
1514
+ _ancestor_validator: Any | None = None,
1515
+ ) -> Path:
1516
+ """Build the notebook for the candidate under ``run_dir`` and write ``table.ipynb`` beside it.
1517
+
1518
+ Resolves the candidate root via ``_candidate_run_target`` (which follows the workspace
1519
+ ``result/`` indirection) and writes ``table.ipynb`` (0o644) as a SIBLING of ``candidate/`` —
1520
+ never inside it. The sidecar is atomically replaced through the verified Build directory's
1521
+ retained descriptor, without following an existing output link or reopening its path.
1522
+
1523
+ ``replace_existing`` separates the command a person ran from the viewer's unattended recovery.
1524
+ ``mr-data notebook`` was asked for a fresh notebook and replaces whatever is there. The watcher
1525
+ may only fill an absence, and it checks that absence long before this write: pass ``False`` and
1526
+ a sidecar that appeared in between raises :class:`FileExistsError` instead of being replaced.
1527
+ """
1528
+
1529
+ # Imported inside the function to break an import cycle: ``cli`` imports this module at module
1530
+ # level for the notebook sidecar, so this module cannot import ``cli`` at module level.
1531
+ from mostlyright.data_harness.cli import _candidate_run_target
1532
+
1533
+ run_dir = Path(run_dir).absolute()
1534
+ target = run_dir if _run_fd is not None else _candidate_run_target(run_dir)
1535
+ handle = None if _run_fd is not None else pipeline._open_candidate_run_handle(target)
1536
+ active_run_fd = _run_fd if _run_fd is not None else handle.run_fd
1537
+ sidecar_lock_fd = -1
1538
+ sidecar_lock_identity: tuple[int, int, int] | None = None
1539
+
1540
+ def validate() -> None:
1541
+ if handle is not None:
1542
+ pipeline._validate_candidate_run_handle(handle)
1543
+ if sidecar_lock_fd >= 0 and sidecar_lock_identity is not None:
1544
+ _validate_table_sidecar_lock(
1545
+ active_run_fd,
1546
+ sidecar_lock_fd,
1547
+ sidecar_lock_identity,
1548
+ )
1549
+ if _ancestor_validator is not None:
1550
+ _ancestor_validator()
1551
+
1552
+ try:
1553
+ if handle is not None:
1554
+ sidecar_lock_fd, sidecar_lock_identity = _acquire_table_sidecar_lock(active_run_fd)
1555
+ with pipeline.open_verified_snapshot(
1556
+ target,
1557
+ _run_fd=active_run_fd,
1558
+ _ancestor_validator=validate,
1559
+ ) as verified_snapshot:
1560
+ notebook = _build_notebook_from_snapshot(target / "candidate", verified_snapshot)
1561
+ validate()
1562
+ raw = (json.dumps(notebook, indent=1, ensure_ascii=False) + "\n").encode("utf-8")
1563
+ _atomic_write_table_notebook(active_run_fd, raw, replace_existing=replace_existing)
1564
+ validate()
1565
+ finally:
1566
+ try:
1567
+ if sidecar_lock_fd >= 0:
1568
+ if fcntl is not None:
1569
+ fcntl.flock(sidecar_lock_fd, fcntl.LOCK_UN)
1570
+ os.close(sidecar_lock_fd)
1571
+ finally:
1572
+ if handle is not None:
1573
+ handle.close()
1574
+ return target / TABLE_NOTEBOOK_NAME
1575
+
1576
+
1577
+ def _acquire_table_sidecar_lock(run_fd: int) -> tuple[int, tuple[int, int, int]]:
1578
+ """Exclusively serialize every sidecar writer on the Build's structural activity file."""
1579
+
1580
+ if fcntl is None:
1581
+ raise pipeline.BuildError(
1582
+ "CANDIDATE_PLATFORM_UNSUPPORTED",
1583
+ "secure table notebook locking is unavailable on this platform",
1584
+ )
1585
+ lock_fd = -1
1586
+ try:
1587
+ named = os.stat(
1588
+ pipeline.RUN_BUILD_ACTIVITY_LOCK,
1589
+ dir_fd=run_fd,
1590
+ follow_symlinks=False,
1591
+ )
1592
+ if not stat.S_ISREG(named.st_mode) or named.st_nlink != 1:
1593
+ raise pipeline.BuildError(
1594
+ "CANDIDATE_MEMBER_INVALID",
1595
+ "run build activity lock is not a stable regular file",
1596
+ )
1597
+ lock_fd = os.open(
1598
+ pipeline.RUN_BUILD_ACTIVITY_LOCK,
1599
+ os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0),
1600
+ dir_fd=run_fd,
1601
+ )
1602
+ opened = os.fstat(lock_fd)
1603
+ identity = pipeline._entry_identity(opened)
1604
+ if (
1605
+ not stat.S_ISREG(opened.st_mode)
1606
+ or opened.st_nlink != 1
1607
+ or identity != pipeline._entry_identity(named)
1608
+ ):
1609
+ raise pipeline.BuildError(
1610
+ "CANDIDATE_MEMBER_INVALID",
1611
+ "run build activity lock changed identity",
1612
+ )
1613
+ fcntl.flock(lock_fd, fcntl.LOCK_EX)
1614
+ _validate_table_sidecar_lock(run_fd, lock_fd, identity)
1615
+ return lock_fd, identity
1616
+ except BaseException:
1617
+ if lock_fd >= 0:
1618
+ os.close(lock_fd)
1619
+ raise
1620
+
1621
+
1622
+ def _validate_table_sidecar_lock(
1623
+ run_fd: int,
1624
+ lock_fd: int,
1625
+ identity: tuple[int, int, int],
1626
+ ) -> None:
1627
+ """Require the held activity lock to remain the exact named run child."""
1628
+
1629
+ opened = os.fstat(lock_fd)
1630
+ named = os.stat(
1631
+ pipeline.RUN_BUILD_ACTIVITY_LOCK,
1632
+ dir_fd=run_fd,
1633
+ follow_symlinks=False,
1634
+ )
1635
+ if (
1636
+ not stat.S_ISREG(opened.st_mode)
1637
+ or opened.st_nlink != 1
1638
+ or not stat.S_ISREG(named.st_mode)
1639
+ or named.st_nlink != 1
1640
+ or pipeline._entry_identity(opened) != identity
1641
+ or pipeline._entry_identity(named) != identity
1642
+ ):
1643
+ raise pipeline.BuildError(
1644
+ "CANDIDATE_MEMBER_INVALID",
1645
+ "run build activity lock changed identity",
1646
+ )
1647
+
1648
+
1649
+ def _atomic_write_table_notebook(run_fd: int, raw: bytes, *, replace_existing: bool = True) -> None:
1650
+ """Install the unsealed sidecar inside one descriptor-pinned Build directory.
1651
+
1652
+ Linking a complete temporary file is an atomic no-replace installation, so a caller that may
1653
+ only fill an absence never overwrites a sidecar that arrived while it was rendering.
1654
+ """
1655
+
1656
+ temporary = ""
1657
+ descriptor = -1
1658
+ try:
1659
+ for _attempt in range(100):
1660
+ temporary = f".{TABLE_NOTEBOOK_NAME}.{secrets.token_hex(12)}"
1661
+ try:
1662
+ descriptor = os.open(
1663
+ temporary,
1664
+ os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_CLOEXEC | os.O_NOFOLLOW,
1665
+ 0o600,
1666
+ dir_fd=run_fd,
1667
+ )
1668
+ break
1669
+ except FileExistsError:
1670
+ continue
1671
+ else:
1672
+ raise FileExistsError("could not allocate a table notebook temporary file")
1673
+
1674
+ offset = 0
1675
+ while offset < len(raw):
1676
+ written = os.write(descriptor, raw[offset:])
1677
+ if written < 1:
1678
+ raise OSError("table notebook write did not make progress")
1679
+ offset += written
1680
+ os.fsync(descriptor)
1681
+ os.fchmod(descriptor, 0o644)
1682
+ os.fsync(descriptor)
1683
+ os.close(descriptor)
1684
+ descriptor = -1
1685
+ if replace_existing:
1686
+ os.replace(
1687
+ temporary,
1688
+ TABLE_NOTEBOOK_NAME,
1689
+ src_dir_fd=run_fd,
1690
+ dst_dir_fd=run_fd,
1691
+ )
1692
+ else:
1693
+ os.link(
1694
+ temporary,
1695
+ TABLE_NOTEBOOK_NAME,
1696
+ src_dir_fd=run_fd,
1697
+ dst_dir_fd=run_fd,
1698
+ follow_symlinks=False,
1699
+ )
1700
+ os.unlink(temporary, dir_fd=run_fd)
1701
+ temporary = ""
1702
+ os.fsync(run_fd)
1703
+ finally:
1704
+ if descriptor >= 0:
1705
+ os.close(descriptor)
1706
+ if temporary:
1707
+ try:
1708
+ os.unlink(temporary, dir_fd=run_fd)
1709
+ except FileNotFoundError:
1710
+ pass