mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,815 @@
1
+ """Deterministic public-HTTPS collections with one certified Reader pin.
2
+
3
+ One collection is one logical Recipe source. The query either names an exact ordered member
4
+ list or a fixed inclusive page-number range. Every member crosses the existing combined
5
+ retrieve/decode/parse Clean-room operation; fetched bytes never enter this coordinator process.
6
+ Only the normalized tables return, and they are concatenated without guessing, coercion,
7
+ deduplication, or reordering.
8
+
9
+ This is intentionally not a general crawler. It performs credential-free GETs, follows only the
10
+ redirects admitted by ``PinnedHttpsRetriever``, and has no cursor, link-following, stop-on-empty,
11
+ header, method, or callback surface.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import re
17
+ import time
18
+ from collections.abc import Callable, Iterable, Mapping
19
+ from dataclasses import dataclass, field, replace
20
+ from pathlib import Path
21
+ from typing import Any
22
+ from urllib.parse import parse_qsl, urlsplit, urlunsplit
23
+
24
+ from mostlyright.data_harness.acquisition.http import (
25
+ RetrievalLimits,
26
+ reader_retrieval_limits,
27
+ seal_content_addressed_snapshot,
28
+ )
29
+ from mostlyright.data_harness.acquisition.sandbox import CrawlerSandbox, sandbox_policy_digest
30
+ from mostlyright.data_harness.canonical import sha256_bytes
31
+ from mostlyright.data_harness.readers.contracts import ReaderBudgets, ReaderError
32
+ from mostlyright.data_harness.readers.registry import TOOLBOX
33
+ from mostlyright.data_harness.readers.tabular import encode_canonical_csv
34
+ from mostlyright.data_harness.sources._adapter_steps import (
35
+ DecodeRecord,
36
+ _authorize_candidate,
37
+ _parse_limits,
38
+ _require_decode_matches_pin,
39
+ )
40
+ from mostlyright.data_harness.sources.adapters import (
41
+ AdapterGovernance,
42
+ _reader_budget_caps,
43
+ _warm_up_family,
44
+ https_reader_policy_digest,
45
+ )
46
+ from mostlyright.data_harness.sources.contracts import (
47
+ AcquisitionReceiptV3,
48
+ AcquisitionRequest,
49
+ AdapterDescriptor,
50
+ CollectionMemberEvidence,
51
+ SnapshotReference,
52
+ SourceContractError,
53
+ SourceObservation,
54
+ collection_fetched_content_sha256,
55
+ collection_member_manifest_digest,
56
+ collection_transport_evidence_digest,
57
+ decode_options_digest,
58
+ )
59
+ from mostlyright.data_harness.sources.governance import (
60
+ classify_before_persistence,
61
+ require_admissible,
62
+ )
63
+ from mostlyright.data_harness.sources.registry import AcquisitionContext, AcquisitionOutput
64
+
65
+ COLLECTION_ADAPTER_ID = "public.https.collection"
66
+ COLLECTION_ADAPTER_VERSION = "1.0.0"
67
+ MAX_COLLECTION_MEMBERS = 16
68
+ MAX_COLLECTION_MEMBER_ID_LENGTH = 128
69
+ MAX_URL_LENGTH = 2_048
70
+
71
+ _MEMBER_ID = re.compile(r"^[a-z][a-z0-9]*(?:[._-][a-z0-9]+)*$")
72
+ _PAGE_PARAMETER = re.compile(r"^[A-Za-z][A-Za-z0-9_-]{0,63}$")
73
+ _MAX_PAGE_NUMBER = 9_007_199_254_740_991
74
+
75
+
76
+ @dataclass(frozen=True)
77
+ class CollectionMember:
78
+ """One exact member address after closed query expansion."""
79
+
80
+ member_id: str
81
+ url: str
82
+
83
+ def __post_init__(self) -> None:
84
+ if (
85
+ not isinstance(self.member_id, str)
86
+ or len(self.member_id) > MAX_COLLECTION_MEMBER_ID_LENGTH
87
+ or _MEMBER_ID.fullmatch(self.member_id) is None
88
+ ):
89
+ raise SourceContractError(
90
+ "COLLECTION_MEMBER_ID",
91
+ "request.query.collection.members.member_id",
92
+ "must be a canonical lowercase identifier no longer than "
93
+ f"{MAX_COLLECTION_MEMBER_ID_LENGTH} characters",
94
+ )
95
+ _safe_public_https_url(self.url, "request.query.collection.members.url")
96
+
97
+
98
+ def collection_members(query: Mapping[str, Any]) -> tuple[CollectionMember, ...]:
99
+ """Parse the collection adapter's exact query and return its semantic member order."""
100
+
101
+ data = _exact_object(query, {"collection", "data_format"}, "request.query")
102
+ if data["data_format"] != "csv":
103
+ raise SourceContractError(
104
+ "COLLECTION_FORMAT",
105
+ "request.query.data_format",
106
+ "collection output must be canonical CSV",
107
+ )
108
+ collection = _object(data["collection"], "request.query.collection")
109
+ kind = collection.get("kind")
110
+ if kind == "explicit":
111
+ return _explicit_members(collection)
112
+ if kind == "page_number":
113
+ return _page_members(collection)
114
+ raise SourceContractError(
115
+ "COLLECTION_KIND",
116
+ "request.query.collection.kind",
117
+ "must be explicit or page_number",
118
+ )
119
+
120
+
121
+ def _explicit_members(value: Mapping[str, Any]) -> tuple[CollectionMember, ...]:
122
+ data = _exact_object(value, {"kind", "members"}, "request.query.collection")
123
+ raw_members = data["members"]
124
+ if not isinstance(raw_members, list) or not 1 <= len(raw_members) <= MAX_COLLECTION_MEMBERS:
125
+ raise SourceContractError(
126
+ "COLLECTION_MEMBER_LIMIT",
127
+ "request.query.collection.members",
128
+ f"must contain between 1 and {MAX_COLLECTION_MEMBERS} members",
129
+ )
130
+ members: list[CollectionMember] = []
131
+ for index, raw in enumerate(raw_members):
132
+ item = _exact_object(
133
+ raw,
134
+ {"member_id", "url"},
135
+ f"request.query.collection.members[{index}]",
136
+ )
137
+ members.append(CollectionMember(member_id=item["member_id"], url=item["url"]))
138
+ _require_unique_members(tuple(members))
139
+ return tuple(members)
140
+
141
+
142
+ def _page_members(value: Mapping[str, Any]) -> tuple[CollectionMember, ...]:
143
+ data = _exact_object(
144
+ value,
145
+ {"kind", "base_url", "page_parameter", "first_page", "last_page"},
146
+ "request.query.collection",
147
+ )
148
+ base_url = _safe_public_https_url(
149
+ data["base_url"],
150
+ "request.query.collection.base_url",
151
+ )
152
+ parameter = data["page_parameter"]
153
+ if not isinstance(parameter, str) or _PAGE_PARAMETER.fullmatch(parameter) is None:
154
+ raise SourceContractError(
155
+ "COLLECTION_PAGE_PARAMETER",
156
+ "request.query.collection.page_parameter",
157
+ "must be a bounded URL-query parameter name",
158
+ )
159
+ first = _page_number(data["first_page"], "request.query.collection.first_page")
160
+ last = _page_number(data["last_page"], "request.query.collection.last_page")
161
+ if last < first:
162
+ raise SourceContractError(
163
+ "COLLECTION_PAGE_RANGE",
164
+ "request.query.collection.last_page",
165
+ "must not precede first_page",
166
+ )
167
+ count = last - first + 1
168
+ if count > MAX_COLLECTION_MEMBERS:
169
+ raise SourceContractError(
170
+ "COLLECTION_MEMBER_LIMIT",
171
+ "request.query.collection",
172
+ f"page range must contain at most {MAX_COLLECTION_MEMBERS} pages",
173
+ )
174
+ parsed = urlsplit(base_url)
175
+ try:
176
+ existing_names = [name for name, _ in parse_qsl(parsed.query, keep_blank_values=True)]
177
+ except ValueError as error:
178
+ raise SourceContractError(
179
+ "COLLECTION_URL",
180
+ "request.query.collection.base_url",
181
+ "contains a malformed query string",
182
+ ) from error
183
+ if parameter in existing_names:
184
+ raise SourceContractError(
185
+ "COLLECTION_PAGE_PARAMETER",
186
+ "request.query.collection.page_parameter",
187
+ "must not already occur in base_url",
188
+ )
189
+ members = tuple(
190
+ CollectionMember(
191
+ member_id=f"page-{page}",
192
+ url=_append_page_parameter(parsed, parameter=parameter, page=page),
193
+ )
194
+ for page in range(first, last + 1)
195
+ )
196
+ _require_unique_members(members)
197
+ return members
198
+
199
+
200
+ def _append_page_parameter(parsed: Any, *, parameter: str, page: int) -> str:
201
+ suffix = f"{parameter}={page}"
202
+ query = f"{parsed.query}&{suffix}" if parsed.query else suffix
203
+ url = urlunsplit((parsed.scheme, parsed.netloc, parsed.path, query, ""))
204
+ return _safe_public_https_url(url, "request.query.collection.base_url")
205
+
206
+
207
+ def _require_unique_members(members: tuple[CollectionMember, ...]) -> None:
208
+ ids = [item.member_id for item in members]
209
+ urls = [item.url for item in members]
210
+ if len(set(ids)) != len(ids):
211
+ raise SourceContractError(
212
+ "COLLECTION_MEMBER_DUPLICATE",
213
+ "request.query.collection.members",
214
+ "member identifiers must be unique",
215
+ )
216
+ if len(set(urls)) != len(urls):
217
+ raise SourceContractError(
218
+ "COLLECTION_MEMBER_DUPLICATE",
219
+ "request.query.collection.members",
220
+ "member URLs must be unique",
221
+ )
222
+
223
+
224
+ def _page_number(value: Any, path: str) -> int:
225
+ if type(value) is not int or not 0 <= value <= _MAX_PAGE_NUMBER:
226
+ raise SourceContractError(
227
+ "COLLECTION_PAGE_RANGE",
228
+ path,
229
+ f"must be an integer in [0, {_MAX_PAGE_NUMBER}]",
230
+ )
231
+ return value
232
+
233
+
234
+ def _safe_public_https_url(value: Any, path: str) -> str:
235
+ if (
236
+ not isinstance(value, str)
237
+ or not 1 <= len(value) <= MAX_URL_LENGTH
238
+ or not value.isascii()
239
+ or any(char in value for char in "\r\n\t")
240
+ ):
241
+ raise SourceContractError(
242
+ "COLLECTION_URL",
243
+ path,
244
+ "must be bounded ASCII HTTPS text without control characters",
245
+ )
246
+ try:
247
+ parsed = urlsplit(value)
248
+ port = parsed.port
249
+ except ValueError as error:
250
+ raise SourceContractError("COLLECTION_URL", path, "is malformed") from error
251
+ if (
252
+ parsed.scheme != "https"
253
+ or not parsed.hostname
254
+ or parsed.username is not None
255
+ or parsed.password is not None
256
+ or parsed.fragment
257
+ or port not in {None, 443}
258
+ ):
259
+ raise SourceContractError(
260
+ "COLLECTION_URL",
261
+ path,
262
+ "must be public HTTPS on port 443 without user information or a fragment",
263
+ )
264
+ return value
265
+
266
+
267
+ def _object(value: Any, path: str) -> Mapping[str, Any]:
268
+ if not isinstance(value, Mapping):
269
+ raise SourceContractError("COLLECTION_QUERY", path, "must be an object")
270
+ return value
271
+
272
+
273
+ def _exact_object(value: Any, keys: set[str], path: str) -> Mapping[str, Any]:
274
+ data = _object(value, path)
275
+ if set(data) != keys:
276
+ raise SourceContractError(
277
+ "COLLECTION_QUERY",
278
+ path,
279
+ f"fields must be exactly {sorted(keys)}",
280
+ )
281
+ return data
282
+
283
+
284
+ def _descriptor() -> AdapterDescriptor:
285
+ return AdapterDescriptor(
286
+ adapter_id=COLLECTION_ADAPTER_ID,
287
+ adapter_version=COLLECTION_ADAPTER_VERSION,
288
+ source_class="external_adapter",
289
+ data_formats=("csv",),
290
+ capabilities=("snapshot", "refresh", "backfill"),
291
+ )
292
+
293
+
294
+ def _collection_retrieval_limits() -> RetrievalLimits:
295
+ # A collection member with no redirect consumes one request. The transport ceiling is 16,
296
+ # matching this adapter's member ceiling, and is one aggregate budget across the collection,
297
+ # not a fresh allowance for every member.
298
+ return RetrievalLimits(max_requests=MAX_COLLECTION_MEMBERS)
299
+
300
+
301
+ @dataclass(frozen=True)
302
+ class PublicHttpsCollectionAdapter:
303
+ """Acquire a bounded collection and seal one canonical CSV snapshot."""
304
+
305
+ sandbox: CrawlerSandbox
306
+ allowed_hostnames: tuple[str, ...]
307
+ governance: AdapterGovernance
308
+ retrieval_limits: RetrievalLimits = field(default_factory=_collection_retrieval_limits)
309
+ descriptor: AdapterDescriptor = field(default_factory=_descriptor)
310
+ monotonic: Callable[[], float] = field(default=time.monotonic, repr=False, compare=False)
311
+
312
+ def __post_init__(self) -> None:
313
+ expected = _descriptor()
314
+ if self.descriptor != expected:
315
+ raise SourceContractError(
316
+ "COLLECTION_ADAPTER_DESCRIPTOR",
317
+ "adapter.descriptor",
318
+ f"must be exactly {COLLECTION_ADAPTER_ID}@{COLLECTION_ADAPTER_VERSION}",
319
+ )
320
+ if not self.allowed_hostnames:
321
+ raise SourceContractError(
322
+ "EGRESS_ALLOWLIST",
323
+ "adapter.allowed_hostnames",
324
+ "collection adapter requires an exact non-empty host allowlist",
325
+ )
326
+ if not callable(self.monotonic):
327
+ raise SourceContractError(
328
+ "COLLECTION_CLOCK",
329
+ "adapter.monotonic",
330
+ "must be a callable monotonic clock",
331
+ )
332
+
333
+ def acquire(
334
+ self,
335
+ request: AcquisitionRequest,
336
+ context: AcquisitionContext,
337
+ ) -> AcquisitionOutput:
338
+ if (request.adapter_id, request.adapter_version) != (
339
+ self.descriptor.adapter_id,
340
+ self.descriptor.adapter_version,
341
+ ):
342
+ raise SourceContractError(
343
+ "ADAPTER_BINDING",
344
+ "request",
345
+ "request does not bind the exact collection adapter identity",
346
+ )
347
+ if request.credential_reference_id is not None:
348
+ raise SourceContractError(
349
+ "AUTHENTICATED_SOURCE_REQUIRES_TRUSTED_GATEWAY",
350
+ "request.credential_reference_id",
351
+ "public collection acquisition receives no credential",
352
+ )
353
+ if request.reader_pin is None:
354
+ raise SourceContractError(
355
+ "READER_PIN_REQUIRED",
356
+ "request.reader_pin",
357
+ "collection members require one exact common Reader pin",
358
+ )
359
+ members = collection_members(request.query)
360
+ _authorize_candidate(request, self.governance)
361
+ if context.crawler_sandbox_attestation_digest != https_reader_policy_digest():
362
+ raise SourceContractError(
363
+ "SANDBOX_ATTESTATION_MISMATCH",
364
+ "context.crawler_sandbox_attestation_digest",
365
+ "context does not pin the HTTPS retrieval-plus-decode policies",
366
+ )
367
+ expected_retrieval = sandbox_policy_digest("retrieve_decode_and_parse")
368
+ expected_decode = sandbox_policy_digest("decode_and_parse")
369
+ _warm_up_family(
370
+ self.sandbox,
371
+ pin=request.reader_pin,
372
+ context=context,
373
+ sealed_format="csv",
374
+ expected_policy_digest=expected_decode,
375
+ )
376
+ return self._acquire_members(
377
+ request,
378
+ context,
379
+ members=members,
380
+ expected_retrieval=expected_retrieval,
381
+ )
382
+
383
+ def _acquire_members(
384
+ self,
385
+ request: AcquisitionRequest,
386
+ context: AcquisitionContext,
387
+ *,
388
+ members: tuple[CollectionMember, ...],
389
+ expected_retrieval: str,
390
+ ) -> AcquisitionOutput:
391
+ assert request.reader_pin is not None
392
+ pin = request.reader_pin
393
+ fetched_bytes = 0
394
+ normalized_bytes = 0
395
+ request_count = 0
396
+ row_count = 0
397
+ declared_cells = 0
398
+ columns: tuple[str, ...] | None = None
399
+ schema_digest: str | None = None
400
+ member_rows: list[tuple[tuple[Any, ...], ...]] = []
401
+ evidence: list[CollectionMemberEvidence] = []
402
+ started = _monotonic_value(self.monotonic, "adapter.monotonic")
403
+ deadline = started + self.retrieval_limits.total_timeout_seconds
404
+ max_declared_cells = _effective_max_declared_cells(context, request)
405
+
406
+ for sequence, member in enumerate(members):
407
+ remaining_requests = self.retrieval_limits.max_requests - request_count
408
+ if remaining_requests <= 0:
409
+ raise SourceContractError(
410
+ "COLLECTION_REQUEST_LIMIT",
411
+ "request.query.collection",
412
+ "members exceed the cumulative HTTP request budget",
413
+ )
414
+ remaining_seconds = deadline - _monotonic_value(
415
+ self.monotonic,
416
+ "adapter.monotonic",
417
+ )
418
+ if remaining_seconds <= 0:
419
+ raise SourceContractError(
420
+ "COLLECTION_TIMEOUT",
421
+ "request.query.collection",
422
+ "collection exceeded its cumulative retrieval deadline",
423
+ )
424
+ remaining = context.max_source_bytes - fetched_bytes
425
+ if remaining <= 0:
426
+ raise SourceContractError(
427
+ "COLLECTION_BYTE_LIMIT",
428
+ "request.query.collection",
429
+ "members exceed the cumulative fetched-byte budget",
430
+ )
431
+ limits = reader_retrieval_limits(
432
+ self.retrieval_limits,
433
+ family_id=str(pin["family_id"]),
434
+ family_version=str(pin["family_version"]),
435
+ )
436
+ limits = replace(
437
+ limits,
438
+ max_response_bytes=min(limits.max_response_bytes, remaining),
439
+ max_aggregate_response_bytes=min(
440
+ limits.max_aggregate_response_bytes,
441
+ remaining,
442
+ ),
443
+ max_requests=remaining_requests,
444
+ # RetrievalLimits requires room for the admitted redirect chain. A nearly
445
+ # exhausted collection can still make its one terminal request, but it may not
446
+ # spend requests it no longer owns on redirects.
447
+ max_redirects=min(limits.max_redirects, remaining_requests - 1),
448
+ total_timeout_seconds=min(limits.total_timeout_seconds, remaining_seconds),
449
+ )
450
+ reader_budgets = _reader_budget_caps(context, request)
451
+ # A zero-row member is valid even after earlier members have spent the whole cell
452
+ # allowance. Reader budgets are positive integers, so retain a one-cell confined
453
+ # decode allowance and enforce the exact aggregate immediately on its parsed result.
454
+ # This never authorizes aggregate output beyond ``max_declared_cells``.
455
+ reader_budgets["max_declared_cells"] = min(
456
+ int(reader_budgets.get("max_declared_cells", max_declared_cells)),
457
+ max(max_declared_cells - declared_cells, 1),
458
+ )
459
+ result = self.sandbox.retrieve_decode_and_parse(
460
+ # The outer request id may already occupy its complete 128-byte contract.
461
+ # Derive a short, collision-resistant child id instead of appending unbounded
462
+ # member text to it. Sequence is semantic and the request digest binds the
463
+ # complete ordered member plan.
464
+ request_id=f"collection.{request.digest[:24]}.{sequence}",
465
+ url=member.url,
466
+ allowed_hostnames=self.allowed_hostnames,
467
+ reader_pin=pin,
468
+ output_format="csv",
469
+ parse_limits=_parse_limits(context),
470
+ retrieval_limits=limits,
471
+ reader_budgets=reader_budgets,
472
+ )
473
+ if result.policy_digest != expected_retrieval:
474
+ raise SourceContractError(
475
+ "SANDBOX_ATTESTATION_MISMATCH",
476
+ "sandbox.policy_digest",
477
+ "collection member does not bind the retrieval-plus-decode policy",
478
+ )
479
+ required = (
480
+ result.content,
481
+ result.parsed,
482
+ result.final_url,
483
+ result.transport_evidence_digest,
484
+ result.fetched_content_sha256,
485
+ result.fetched_content_size_bytes,
486
+ result.fetched_total_response_body_size_bytes,
487
+ result.fetched_request_count,
488
+ result.decode_family_id,
489
+ result.decode_family_version,
490
+ result.decode_options_digest,
491
+ result.decode_flags,
492
+ result.decode_declared_cell_count,
493
+ )
494
+ if any(item is None for item in required):
495
+ raise SourceContractError(
496
+ "SANDBOX_RESULT",
497
+ "sandbox",
498
+ "collection member acquisition result is incomplete",
499
+ )
500
+ assert result.content is not None
501
+ assert result.parsed is not None
502
+ assert result.final_url is not None
503
+ assert result.transport_evidence_digest is not None
504
+ assert result.fetched_content_sha256 is not None
505
+ assert result.fetched_content_size_bytes is not None
506
+ assert result.fetched_total_response_body_size_bytes is not None
507
+ assert result.fetched_request_count is not None
508
+ assert result.decode_family_id is not None
509
+ assert result.decode_family_version is not None
510
+ assert result.decode_options_digest is not None
511
+ assert result.decode_flags is not None
512
+ assert result.decode_declared_cell_count is not None
513
+ if result.parsed.input_sha256 != sha256_bytes(result.content):
514
+ raise SourceContractError(
515
+ "SANDBOX_RESULT",
516
+ "sandbox.parsed.input_sha256",
517
+ "parsed table does not bind the normalized member bytes",
518
+ )
519
+ _require_decode_matches_pin(
520
+ request,
521
+ _decode_record(
522
+ fetched_sha256=result.fetched_content_sha256,
523
+ family_id=result.decode_family_id,
524
+ family_version=result.decode_family_version,
525
+ options_digest=result.decode_options_digest,
526
+ flags=result.decode_flags,
527
+ ),
528
+ )
529
+ if (
530
+ type(result.fetched_total_response_body_size_bytes) is not int
531
+ or type(result.fetched_content_size_bytes) is not int
532
+ or not 1
533
+ <= result.fetched_content_size_bytes
534
+ <= result.fetched_total_response_body_size_bytes
535
+ ):
536
+ raise SourceContractError(
537
+ "COLLECTION_BYTE_LIMIT",
538
+ "sandbox.fetched_total_response_body_size_bytes",
539
+ "member byte accounting does not contain its terminal fetched object",
540
+ )
541
+ # Charge every retained response body, not only the terminal object recorded as this
542
+ # member's fetched identity. Redirect bodies consume the same collection-wide source
543
+ # budget and cannot buy a fresh allowance by preceding a smaller final response.
544
+ fetched_bytes += result.fetched_total_response_body_size_bytes
545
+ if (
546
+ type(result.fetched_request_count) is not int
547
+ or not 1 <= result.fetched_request_count <= remaining_requests
548
+ ):
549
+ raise SourceContractError(
550
+ "COLLECTION_REQUEST_LIMIT",
551
+ "sandbox.fetched_request_count",
552
+ "member request accounting is outside the remaining collection budget",
553
+ )
554
+ request_count += result.fetched_request_count
555
+ if _monotonic_value(self.monotonic, "adapter.monotonic") > deadline:
556
+ raise SourceContractError(
557
+ "COLLECTION_TIMEOUT",
558
+ "request.query.collection",
559
+ "collection exceeded its cumulative retrieval deadline",
560
+ )
561
+ normalized_bytes += len(result.content)
562
+ if fetched_bytes > context.max_source_bytes:
563
+ raise SourceContractError(
564
+ "COLLECTION_BYTE_LIMIT",
565
+ "request.query.collection",
566
+ "members exceed the cumulative fetched-byte budget",
567
+ )
568
+ if normalized_bytes > context.max_source_bytes:
569
+ raise SourceContractError(
570
+ "COLLECTION_BYTE_LIMIT",
571
+ "request.query.collection",
572
+ "members exceed the cumulative normalized-byte budget",
573
+ )
574
+ if len(result.parsed.columns) > context.max_columns:
575
+ raise SourceContractError(
576
+ "COLLECTION_COLUMN_LIMIT",
577
+ "request.query.collection",
578
+ "member schema exceeds the column budget",
579
+ )
580
+ if columns is None:
581
+ columns = result.parsed.columns
582
+ schema_digest = result.parsed.schema_digest
583
+ elif result.parsed.columns != columns or result.parsed.schema_digest != schema_digest:
584
+ raise SourceContractError(
585
+ "COLLECTION_SCHEMA_MISMATCH",
586
+ f"request.query.collection.members[{sequence}]",
587
+ "member schema differs from the first member; no coercion is implicit",
588
+ )
589
+ row_count += len(result.parsed.rows)
590
+ if row_count > context.max_rows:
591
+ raise SourceContractError(
592
+ "COLLECTION_ROW_LIMIT",
593
+ "request.query.collection",
594
+ "members exceed the cumulative row budget",
595
+ )
596
+ normalized_cell_count = len(result.parsed.rows) * len(result.parsed.columns)
597
+ if (
598
+ type(result.decode_declared_cell_count) is not int
599
+ or result.decode_declared_cell_count != normalized_cell_count
600
+ ):
601
+ raise SourceContractError(
602
+ "COLLECTION_CELL_EVIDENCE",
603
+ "sandbox.decode_declared_cell_count",
604
+ "collection v1 requires Reader cell accounting that is exactly rederived "
605
+ "from the normalized member geometry",
606
+ )
607
+ declared_cells += result.decode_declared_cell_count
608
+ if declared_cells > max_declared_cells:
609
+ raise SourceContractError(
610
+ "COLLECTION_CELL_LIMIT",
611
+ "request.query.collection",
612
+ "members exceed the cumulative effective Reader cell budget",
613
+ )
614
+ member_rows.append(result.parsed.rows)
615
+ evidence.append(
616
+ CollectionMemberEvidence(
617
+ member_id=member.member_id,
618
+ sequence=sequence,
619
+ source_uri=member.url,
620
+ final_uri=result.final_url,
621
+ fetched_content_sha256=result.fetched_content_sha256,
622
+ fetched_size_bytes=result.fetched_content_size_bytes,
623
+ normalized_content_sha256=sha256_bytes(result.content),
624
+ normalized_size_bytes=len(result.content),
625
+ request_count=result.fetched_request_count,
626
+ row_count=len(result.parsed.rows),
627
+ column_count=len(result.parsed.columns),
628
+ declared_cell_count=result.decode_declared_cell_count,
629
+ parsed_schema_digest=result.parsed.schema_digest,
630
+ transport_evidence_digest=result.transport_evidence_digest,
631
+ decode_flags=result.decode_flags,
632
+ )
633
+ )
634
+
635
+ assert columns is not None and schema_digest is not None
636
+ try:
637
+ content = encode_canonical_csv(
638
+ columns,
639
+ _ordered_rows(member_rows),
640
+ budgets=ReaderBudgets().narrowed_by(
641
+ {
642
+ "max_output_bytes": (
643
+ context.max_source_bytes
644
+ if context.max_normalized_bytes is None
645
+ else context.max_normalized_bytes
646
+ ),
647
+ "max_rows": context.max_rows,
648
+ "max_columns": context.max_columns,
649
+ "max_declared_cells": max_declared_cells,
650
+ }
651
+ ),
652
+ )
653
+ except ReaderError as error:
654
+ raise SourceContractError(
655
+ "COLLECTION_OUTPUT",
656
+ "request.query.collection",
657
+ "canonical aggregate output was refused by the Reader budget",
658
+ ) from error
659
+ classification = classify_before_persistence(
660
+ content,
661
+ declared_classification=self.governance.declared_classification,
662
+ lawful_basis_declared=self.governance.lawful_basis_declared,
663
+ )
664
+ require_admissible(classification)
665
+ snapshot_path = seal_content_addressed_snapshot(
666
+ context.snapshot_root,
667
+ content=content,
668
+ suffix="csv",
669
+ )
670
+ return _collection_output(
671
+ request=request,
672
+ descriptor=self.descriptor,
673
+ governance=self.governance,
674
+ snapshot_path=snapshot_path,
675
+ content=content,
676
+ classification=classification.classification,
677
+ members=tuple(evidence),
678
+ parsed_schema_digest=schema_digest,
679
+ )
680
+
681
+
682
+ def _decode_record(
683
+ *,
684
+ fetched_sha256: str,
685
+ family_id: str,
686
+ family_version: str,
687
+ options_digest: str,
688
+ flags: tuple[str, ...],
689
+ ) -> DecodeRecord:
690
+ return DecodeRecord(
691
+ fetched_content_sha256=fetched_sha256,
692
+ family_id=family_id,
693
+ family_version=family_version,
694
+ decode_options_digest=options_digest,
695
+ flags=flags,
696
+ )
697
+
698
+
699
+ def _effective_max_declared_cells(
700
+ context: AcquisitionContext,
701
+ request: AcquisitionRequest,
702
+ ) -> int:
703
+ """Return the one aggregate cell ceiling shared by every member and final encoding.
704
+
705
+ The selected family's certified defaults remain the outer Reader authority. Trusted context
706
+ row/column ceilings and an optional Recipe cap can only narrow it. Resolving the exact Toolbox
707
+ coordinate here also means the aggregate coordinator cannot accidentally enforce a generic
708
+ Reader ceiling wider than the family that actually decodes the members.
709
+ """
710
+
711
+ assert request.reader_pin is not None
712
+ family = TOOLBOX.resolve(
713
+ str(request.reader_pin["family_id"]),
714
+ str(request.reader_pin["family_version"]),
715
+ )
716
+ budgets = family.default_budgets.narrowed_by(_reader_budget_caps(context, request))
717
+ return min(
718
+ budgets.max_declared_cells,
719
+ budgets.max_rows * budgets.max_columns,
720
+ )
721
+
722
+
723
+ def _monotonic_value(clock: Callable[[], float], path: str) -> float:
724
+ value = clock()
725
+ if type(value) not in {int, float} or not 0 <= value < float("inf"):
726
+ raise SourceContractError(
727
+ "COLLECTION_CLOCK",
728
+ path,
729
+ "must return a finite non-negative monotonic value",
730
+ )
731
+ return float(value)
732
+
733
+
734
+ def _ordered_rows(
735
+ members: Iterable[tuple[tuple[Any, ...], ...]],
736
+ ) -> Iterable[tuple[Any, ...]]:
737
+ for rows in members:
738
+ yield from rows
739
+
740
+
741
+ def _collection_output(
742
+ *,
743
+ request: AcquisitionRequest,
744
+ descriptor: AdapterDescriptor,
745
+ governance: AdapterGovernance,
746
+ snapshot_path: Path,
747
+ content: bytes,
748
+ classification: str,
749
+ members: tuple[CollectionMemberEvidence, ...],
750
+ parsed_schema_digest: str,
751
+ ) -> AcquisitionOutput:
752
+ assert request.reader_pin is not None
753
+ observation = SourceObservation(
754
+ source_id=request.source_id,
755
+ observed_at=governance.observation.observed_at,
756
+ event_time_field=governance.observation.event_time_field,
757
+ available_at_field=governance.observation.available_at_field,
758
+ historical_start=governance.observation.historical_start,
759
+ historical_end=governance.observation.historical_end,
760
+ live_status=governance.observation.live_status,
761
+ publication_delay_seconds=governance.observation.publication_delay_seconds,
762
+ update_frequency_seconds=governance.observation.update_frequency_seconds,
763
+ response_status=governance.observation.response_status,
764
+ evidence=governance.observation.evidence,
765
+ )
766
+ flags = tuple(sorted({flag for member in members for flag in member.decode_flags}))
767
+ digest = sha256_bytes(content)
768
+ receipt = AcquisitionReceiptV3(
769
+ receipt_id=f"receipt.{request.digest[:24]}",
770
+ request_digest=request.digest,
771
+ source_id=request.source_id,
772
+ adapter=descriptor,
773
+ query_digest=request.query_digest,
774
+ snapshot=SnapshotReference(
775
+ content_sha256=digest,
776
+ size_bytes=len(content),
777
+ media_type="text/csv",
778
+ data_format="csv",
779
+ storage_name=snapshot_path.name,
780
+ ),
781
+ fetched_content_sha256=collection_fetched_content_sha256(members),
782
+ family_id=str(request.reader_pin["family_id"]),
783
+ family_version=str(request.reader_pin["family_version"]),
784
+ decode_options_digest=decode_options_digest(request.reader_pin["decode_options"]),
785
+ decode_flags=flags,
786
+ event_time_field=governance.observation.event_time_field,
787
+ observed_at=governance.observation.observed_at,
788
+ available_at=governance.observation.available_at,
789
+ ingested_at=request.requested_at,
790
+ classification=classification,
791
+ rights_digest=governance.rights.digest,
792
+ retention_digest=governance.retention.digest,
793
+ observation_digest=observation.digest,
794
+ transport_evidence_digest=collection_transport_evidence_digest(members),
795
+ members=members,
796
+ member_manifest_digest=collection_member_manifest_digest(members),
797
+ parsed_schema_digest=parsed_schema_digest,
798
+ )
799
+ return AcquisitionOutput(
800
+ receipt=receipt,
801
+ observation=observation,
802
+ parsed_schema_digest=parsed_schema_digest,
803
+ snapshot_path=snapshot_path,
804
+ )
805
+
806
+
807
+ __all__ = [
808
+ "COLLECTION_ADAPTER_ID",
809
+ "COLLECTION_ADAPTER_VERSION",
810
+ "MAX_COLLECTION_MEMBERS",
811
+ "MAX_COLLECTION_MEMBER_ID_LENGTH",
812
+ "CollectionMember",
813
+ "PublicHttpsCollectionAdapter",
814
+ "collection_members",
815
+ ]