mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,1214 @@
1
+ """Governed user-file and public-HTTPS source adapters.
2
+
3
+ The steps shared with the external-provider and stream adapters live in ``_adapter_steps``. The
4
+ private helpers below are local to this module.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import os
10
+ import re
11
+ import stat
12
+ from collections.abc import Callable, Mapping
13
+ from dataclasses import dataclass, field, replace
14
+ from pathlib import Path, PurePosixPath
15
+ from typing import Any
16
+ from urllib.parse import urlsplit
17
+
18
+ from mostlyright.data_harness.acquisition.http import (
19
+ RetrievalLimits,
20
+ reader_retrieval_limits,
21
+ seal_content_addressed_snapshot,
22
+ )
23
+ from mostlyright.data_harness.acquisition.parsing import row_digest_for
24
+ from mostlyright.data_harness.acquisition.sandbox import (
25
+ CrawlerSandbox,
26
+ sandbox_policy_digest,
27
+ )
28
+ from mostlyright.data_harness.acquisition.sandbox import (
29
+ FetchedMember as SandboxFetchedMember,
30
+ )
31
+ from mostlyright.data_harness.acquisition.url_policy import AcquisitionSecurityError
32
+ from mostlyright.data_harness.canonical import canonical_sha256, sha256_bytes
33
+ from mostlyright.data_harness.formats import DATA_FORMATS
34
+ from mostlyright.data_harness.readers.contracts import ReaderBudgets, ReaderPin
35
+ from mostlyright.data_harness.readers.samples import DecodedFacts, ReaderSample, warm_up
36
+ from mostlyright.data_harness.sources._adapter_steps import (
37
+ DecodeRecord,
38
+ _authorize_candidate,
39
+ _output,
40
+ _parse_limits,
41
+ _require_sandbox_binding,
42
+ )
43
+ from mostlyright.data_harness.sources.contracts import (
44
+ AcquisitionReceiptV2,
45
+ AcquisitionRequest,
46
+ AdapterDescriptor,
47
+ EvidenceReference,
48
+ RetentionPolicy,
49
+ RightsEvidence,
50
+ SnapshotReference,
51
+ SourceContractError,
52
+ SourceObservation,
53
+ fetched_source_revision_digest,
54
+ )
55
+ from mostlyright.data_harness.sources.contracts import (
56
+ FetchedMember as ReceiptFetchedMember,
57
+ )
58
+ from mostlyright.data_harness.sources.governance import (
59
+ classify_before_persistence,
60
+ require_admissible,
61
+ )
62
+ from mostlyright.data_harness.sources.range_reader import (
63
+ RangeRevisionLock,
64
+ acquire_range_reader,
65
+ default_policy_digests,
66
+ range_reader_policy_digest,
67
+ require_crawler_sandbox,
68
+ )
69
+ from mostlyright.data_harness.sources.registry import (
70
+ AcquisitionContext,
71
+ AcquisitionOutput,
72
+ )
73
+
74
+ # Ceilings for the composed address. Every one is enforced before a single character is
75
+ # joined, so a malformed composition object costs a bounded amount of work rather than an
76
+ # unbounded one.
77
+ URL_MAX_LENGTH = 2_048
78
+ URL_ORIGIN_MAX_LENGTH = 255
79
+ URL_MAX_PATH_COMPONENTS = 16
80
+ URL_MAX_COMPONENT_PARTS = 8
81
+ URL_PART_MAX_LENGTH = 64
82
+ URL_COMPONENT_MAX_LENGTH = 128
83
+
84
+
85
+ def _receipt_v2_member(member: SandboxFetchedMember) -> ReceiptFetchedMember:
86
+ """Project current transport evidence onto the immutable historical v2 receipt shape."""
87
+
88
+ document = member.to_dict()
89
+ document["peer_attempts"] = [
90
+ {
91
+ "approved_ip": attempt["approved_ip"],
92
+ "outcome": attempt["outcome"],
93
+ "failure_code": attempt["failure_code"],
94
+ }
95
+ for attempt in document["peer_attempts"]
96
+ ]
97
+ return ReceiptFetchedMember(**document)
98
+
99
+
100
+ # What a path part may be made of. The excluded characters are not a stylistic preference:
101
+ # each one is how a path part stops being a path part. ``/`` invents a component the recipe
102
+ # did not state; ``?`` and ``#`` start a query string or a fragment the recipe never
103
+ # declared; ``@`` and ``:`` can move the authority, so the address would reach a host the
104
+ # allowlist was never asked about; ``%`` re-encodes any of those past a later reader; and a
105
+ # space or a non-ASCII character is normalised differently by different readers, so what is
106
+ # checked and what is fetched can disagree. What remains cannot express any of them, which
107
+ # is why the refusal is structural rather than a regular expression run over the assembled
108
+ # result.
109
+ _URL_PART = re.compile(rf"^[A-Za-z0-9._-]{{1,{URL_PART_MAX_LENGTH}}}$")
110
+
111
+
112
+ @dataclass(frozen=True)
113
+ class ObservationPolicy:
114
+ """Recorded, digest-bound temporal/source evidence owned by the coordinator."""
115
+
116
+ observed_at: str
117
+ available_at: str
118
+ historical_start: str
119
+ historical_end: str
120
+ live_status: str
121
+ publication_delay_seconds: int
122
+ update_frequency_seconds: int
123
+ response_status: int | None
124
+ event_time_field: str | None
125
+ available_at_field: str | None
126
+ evidence: tuple[EvidenceReference, ...]
127
+
128
+
129
+ @dataclass(frozen=True)
130
+ class AdapterGovernance:
131
+ rights: RightsEvidence
132
+ retention: RetentionPolicy
133
+ declared_classification: str
134
+ lawful_basis_declared: bool
135
+ attached_license_artifact_digests: tuple[str, ...]
136
+ observation: ObservationPolicy
137
+
138
+
139
+ # Which families this process has already warmed up. One recipe execution is one refresh, and
140
+ # the harness already runs a fresh coordinator process per execution, so per-process memoization
141
+ # is per-refresh in practice -- that is why "before every refresh" and "once per process" say the
142
+ # same thing here. The consequence, stated rather than left to be inferred: a long-lived
143
+ # coordinator would warm up once and then never again, so if one is ever introduced this memo is
144
+ # the thing that must change.
145
+ _WARMED_READER_FAMILIES: set[tuple[str, str]] = set()
146
+
147
+
148
+ def _clean_room_decode_route(
149
+ sandbox: CrawlerSandbox,
150
+ *,
151
+ context: AcquisitionContext,
152
+ sealed_format: str,
153
+ expected_policy_digest: str | None = None,
154
+ ) -> Callable[[ReaderSample], DecodedFacts]:
155
+ """Build the decode route the warm-up check runs its known file through.
156
+
157
+ The route is injected rather than called from ``readers.samples`` because ``readers/`` may
158
+ not import ``acquisition/``; that back-edge would close an import cycle. It is built here,
159
+ where the sandbox already is.
160
+
161
+ It goes through the same ``decode_and_parse`` operation the real decode uses, under the same
162
+ attested policy. Warming up through a shortcut would prove the decoder works on a path
163
+ production does not take, which is the class of green test that hides a red system.
164
+
165
+ Two consumers, two routes, one expectation. The CI coverage gate
166
+ (``tests/test_reader_samples.py``) takes ``readers.samples.decode_in_process``: it runs on
167
+ every leg including ones with no sandbox, and its job is to prove the family answers
168
+ correctly. A refresh takes this route: its job is to prove the family answers correctly *on
169
+ the path production takes*, attestation included. Both read the same packaged sample through
170
+ the same loader and compare against the same recorded constants, so the two cannot drift on
171
+ expectations -- only on transport, which is the difference they exist to cover.
172
+
173
+ The row count and column names reported here are the parser's, derived from the bytes the
174
+ Reader returned, where the in-process route reports the family's own declaration. That is
175
+ deliberate and is the one place the two routes differ in substance: a family whose declared
176
+ shape disagrees with the parser's reading of its own output fails this check, and that is a
177
+ finding worth having rather than one to reconcile away.
178
+
179
+ The family's own default budgets apply, exactly as they do for the CI gate. Narrowing the
180
+ warm-up by a recipe's caps would let one recipe's tight budget turn a healthy Reader into a
181
+ halted refresh.
182
+ """
183
+
184
+ def decode(sample: ReaderSample) -> DecodedFacts:
185
+ sandboxed = sandbox.decode_and_parse(
186
+ # Lower-cased and dot-joined because a sandbox request identifier admits lowercase
187
+ # letters, digits, and ``._-`` only, while a family identifier may carry capitals
188
+ # and a coordinate carries an ``@``.
189
+ request_id=f"warmup.{sample.family_id}.{sample.family_version}".lower(),
190
+ content=sample.content,
191
+ reader_pin={
192
+ "family_id": sample.family_id,
193
+ "family_version": sample.family_version,
194
+ "decode_options": dict(sample.decode_options),
195
+ },
196
+ output_format=sealed_format,
197
+ limits=_parse_limits(context),
198
+ reader_budgets=None,
199
+ )
200
+ if expected_policy_digest is None:
201
+ _require_sandbox_binding(sandboxed.policy_digest, context)
202
+ elif sandboxed.policy_digest != expected_policy_digest:
203
+ raise SourceContractError(
204
+ "SANDBOX_ATTESTATION_MISMATCH",
205
+ "sandbox.policy_digest",
206
+ "warm-up decode does not bind the composed adapter's decode policy",
207
+ )
208
+ if sandboxed.parsed is None or sandboxed.content is None or sandboxed.decode_flags is None:
209
+ raise SourceContractError(
210
+ "SANDBOX_RESULT",
211
+ "sandbox",
212
+ "warm-up decode result is incomplete",
213
+ )
214
+ return DecodedFacts(
215
+ sha256_bytes(sandboxed.content),
216
+ len(sandboxed.parsed.rows),
217
+ tuple(sandboxed.parsed.columns),
218
+ tuple(sandboxed.decode_flags),
219
+ )
220
+
221
+ return decode
222
+
223
+
224
+ def _warm_up_family(
225
+ sandbox: CrawlerSandbox,
226
+ *,
227
+ pin: Mapping[str, Any],
228
+ context: AcquisitionContext,
229
+ sealed_format: str,
230
+ expected_policy_digest: str | None = None,
231
+ ) -> None:
232
+ """Re-open this family's one known file before it is allowed to touch source data.
233
+
234
+ Stated for the user as: before every refresh, the Reader re-opens one known file; a wrong
235
+ answer halts the refresh. A decoder that answers differently on a file whose answer is known
236
+ has changed the meaning of every dataset it has ever produced, and the datasets already
237
+ sealed cannot be un-sealed -- so the only useful moment to stop is before the next one is
238
+ built, which is why this runs before the fetch rather than beside it.
239
+ """
240
+
241
+ coordinate = (str(pin["family_id"]), str(pin["family_version"]))
242
+ if coordinate in _WARMED_READER_FAMILIES:
243
+ return
244
+ warm_up(
245
+ coordinate[0],
246
+ coordinate[1],
247
+ decode=_clean_room_decode_route(
248
+ sandbox,
249
+ context=context,
250
+ sealed_format=sealed_format,
251
+ expected_policy_digest=expected_policy_digest,
252
+ ),
253
+ )
254
+ _WARMED_READER_FAMILIES.add(coordinate)
255
+
256
+
257
+ @dataclass(frozen=True)
258
+ class _Decoded:
259
+ """What one clean-room decode handed back, in exactly the facts the adapter seals.
260
+
261
+ ``content`` is the Reader's canonical output and is what gets sealed; the fetched bytes
262
+ are never sealed and survive only as the digest inside ``record``.
263
+ """
264
+
265
+ content: bytes
266
+ media_type: str
267
+ parsed_schema_digest: str
268
+ record: DecodeRecord
269
+
270
+
271
+ def _reader_budget_caps(
272
+ context: AcquisitionContext,
273
+ request: AcquisitionRequest | None = None,
274
+ ) -> dict[str, int]:
275
+ """Every cap that reaches the decode: the coordinator's, and the recipe's own.
276
+
277
+ ``ReaderBudgets.narrowed_by`` takes a per-field minimum, so these can only tighten the
278
+ family's own defaults and can never buy more room than the family allows. This is the
279
+ Reader-side twin of ``_parse_limits``, which does the same for the parser.
280
+
281
+ This is the only function that builds ``reader_budgets``, so recipe caps are merged here.
282
+
283
+ Where the two disagree, the tighter wins on each field independently, and that is
284
+ ``narrowed_by``'s own rule applied once more here rather than a second policy: a
285
+ coordinator that caps a source and a recipe that caps itself are both entitled to their
286
+ ceiling, and neither can raise the other's.
287
+ """
288
+
289
+ caps = {
290
+ "max_input_bytes": context.max_source_bytes,
291
+ # The decode output budget, distinct from the fetched-byte budget, mirroring the
292
+ # hosted Reader budget narrowing.
293
+ "max_output_bytes": (
294
+ context.max_source_bytes
295
+ if context.max_normalized_bytes is None
296
+ else context.max_normalized_bytes
297
+ ),
298
+ "max_rows": context.max_rows,
299
+ "max_columns": context.max_columns,
300
+ }
301
+ stated = None if request is None else request.resource_caps
302
+ if stated is None:
303
+ return caps
304
+ for name, value in stated.items():
305
+ current = caps.get(name)
306
+ caps[name] = int(value) if current is None else min(current, int(value))
307
+ return caps
308
+
309
+
310
+ def https_reader_policy_digest() -> str:
311
+ """Bind the networked decode and its mandatory networkless warm-up as one policy."""
312
+
313
+ return canonical_sha256(
314
+ {
315
+ "retrieve_decode_and_parse": sandbox_policy_digest("retrieve_decode_and_parse"),
316
+ "decode_and_parse": sandbox_policy_digest("decode_and_parse"),
317
+ }
318
+ )
319
+
320
+
321
+ def _decode_in_clean_room(
322
+ sandbox: CrawlerSandbox,
323
+ *,
324
+ request: AcquisitionRequest,
325
+ context: AcquisitionContext,
326
+ content: bytes,
327
+ sealed_format: str,
328
+ ) -> _Decoded:
329
+ """Run the request's pinned Reader on fetched bytes, inside the confinement.
330
+
331
+ Two formats meet here and are deliberately not the same thing. The **sealed** format is
332
+ ``sealed_format``: what the adapter query names, what the snapshot suffix carries, and what
333
+ the receipt's ``snapshot.data_format`` must equal. The **fetched** format is described by
334
+ the Reader pin, appears in no format table, and is never sealed. Conflating them would seal
335
+ an archive under a wire-format name.
336
+
337
+ The filename the parser reads is the family's own declaration and never the fetched file's
338
+ name: it is chosen inside the worker from ``ReaderResult.filename``, so a zip fetched as
339
+ ``cities.zip`` is parsed as the CSV the Reader produced rather than refused as
340
+ ``PARSE_SUFFIX_MISMATCH``.
341
+ """
342
+
343
+ pin = request.reader_pin
344
+ if pin is None:
345
+ raise SourceContractError(
346
+ "DECODE_PIN_MISMATCH",
347
+ "request.reader_pin",
348
+ "a decode was attempted for a request that entitles no Reader family",
349
+ )
350
+ try:
351
+ sandboxed = sandbox.decode_and_parse(
352
+ request_id=request.request_id,
353
+ content=content,
354
+ reader_pin=dict(pin),
355
+ output_format=sealed_format,
356
+ limits=_parse_limits(context),
357
+ reader_budgets=_reader_budget_caps(context, request),
358
+ )
359
+ except AcquisitionSecurityError as error:
360
+ # A halted refresh names the source it halted on. The Reader cannot do this itself and
361
+ # should not be able to: it is handed bytes and no provenance at all, which is the
362
+ # property that lets its refusals be carried out of the confinement in the first place.
363
+ # This is the layer that knows the recipe's own name for the source, so this is where
364
+ # the name is attached.
365
+ raise AcquisitionSecurityError(
366
+ error.code,
367
+ f"source {request.source_id}: {error.detail}",
368
+ ) from error
369
+ # The decode operation carries its own attestation. A coordinator pinned to the plain
370
+ # ``parse`` digest does not authorize a decode and is refused here.
371
+ _require_sandbox_binding(sandboxed.policy_digest, context)
372
+ if (
373
+ sandboxed.parsed is None
374
+ or sandboxed.content is None
375
+ or sandboxed.media_type is None
376
+ or sandboxed.decode_family_id is None
377
+ or sandboxed.decode_family_version is None
378
+ or sandboxed.decode_options_digest is None
379
+ or sandboxed.decode_flags is None
380
+ ):
381
+ raise SourceContractError(
382
+ "SANDBOX_RESULT",
383
+ "sandbox",
384
+ "decode result is incomplete",
385
+ )
386
+ # The returned table must describe the returned bytes. The response boundary checks this
387
+ # too; both gates stand, so a worker that returned a table about other bytes has to defeat
388
+ # the boundary and the adapter rather than either alone.
389
+ if sandboxed.parsed.input_sha256 != sha256_bytes(sandboxed.content):
390
+ raise SourceContractError(
391
+ "SANDBOX_RESULT",
392
+ "sandbox.parsed.input_sha256",
393
+ "clean-room result does not bind the exact normalized bytes it returned",
394
+ )
395
+ return _Decoded(
396
+ content=sandboxed.content,
397
+ media_type=sandboxed.media_type,
398
+ parsed_schema_digest=sandboxed.parsed.schema_digest,
399
+ record=DecodeRecord(
400
+ fetched_content_sha256=sha256_bytes(content),
401
+ family_id=sandboxed.decode_family_id,
402
+ family_version=sandboxed.decode_family_version,
403
+ decode_options_digest=sandboxed.decode_options_digest,
404
+ flags=sandboxed.decode_flags,
405
+ ),
406
+ )
407
+
408
+
409
+ def _local_descriptor() -> AdapterDescriptor:
410
+ return AdapterDescriptor(
411
+ adapter_id="user.file",
412
+ adapter_version="1.0.0",
413
+ source_class="user_file",
414
+ data_formats=tuple(sorted(DATA_FORMATS)),
415
+ capabilities=("snapshot",),
416
+ )
417
+
418
+
419
+ @dataclass(frozen=True)
420
+ class LocalFileAdapter:
421
+ """Descriptor-confined user-file acquisition with sandboxed parsing."""
422
+
423
+ input_root: Path
424
+ sandbox: CrawlerSandbox
425
+ governance: AdapterGovernance
426
+ descriptor: AdapterDescriptor = field(default_factory=_local_descriptor)
427
+
428
+ def __post_init__(self) -> None:
429
+ if (
430
+ not self.input_root.is_absolute()
431
+ or self.input_root.is_symlink()
432
+ or not self.input_root.is_dir()
433
+ ):
434
+ raise SourceContractError(
435
+ "INPUT_ROOT",
436
+ "adapter.input_root",
437
+ "must be an existing absolute non-symlink directory",
438
+ )
439
+
440
+ def acquire(
441
+ self,
442
+ request: AcquisitionRequest,
443
+ context: AcquisitionContext,
444
+ ) -> AcquisitionOutput:
445
+ _bind_request(request, self.descriptor)
446
+ if request.credential_reference_id is not None:
447
+ raise SourceContractError(
448
+ "USER_FILE_CREDENTIAL",
449
+ "request.credential_reference_id",
450
+ "local files cannot select or receive source credentials",
451
+ )
452
+ _require_disjoint_roots(self.input_root, context.snapshot_root)
453
+ relative_path, data_format, media_type = _local_query(request.query)
454
+ _authorize_candidate(request, self.governance)
455
+ if request.reader_pin is not None:
456
+ # The Reader stage. ``data_format`` here is the **sealed** format and stays what
457
+ # it is; the query's ``media_type`` describes the **fetched** bytes and is not
458
+ # used below, because the Reader family declares the media type of what it
459
+ # produced. Everything after this branch is the direct-fetch path, unchanged.
460
+ return self._acquire_decoded(
461
+ request,
462
+ context,
463
+ relative_path=relative_path,
464
+ sealed_format=data_format,
465
+ )
466
+ content = _read_confined_file(
467
+ self.input_root,
468
+ relative_path=relative_path,
469
+ max_bytes=context.max_source_bytes,
470
+ )
471
+ classification = classify_before_persistence(
472
+ content,
473
+ declared_classification=self.governance.declared_classification,
474
+ lawful_basis_declared=self.governance.lawful_basis_declared,
475
+ )
476
+ require_admissible(classification)
477
+ sandboxed = self.sandbox.parse(
478
+ request_id=request.request_id,
479
+ content=content,
480
+ data_format=data_format,
481
+ media_type=media_type,
482
+ filename=PurePosixPath(relative_path).name,
483
+ limits=_parse_limits(context),
484
+ )
485
+ _require_sandbox_binding(sandboxed.policy_digest, context)
486
+ if (
487
+ sandboxed.parsed is None
488
+ or sandboxed.parsed.input_sha256 != classification.content_sha256
489
+ ):
490
+ raise SourceContractError(
491
+ "SANDBOX_RESULT",
492
+ "sandbox.parsed",
493
+ "sandbox result does not bind the classified exact source bytes",
494
+ )
495
+ snapshot_path = seal_content_addressed_snapshot(
496
+ context.snapshot_root,
497
+ content=content,
498
+ suffix=data_format,
499
+ )
500
+ transport_digest = canonical_sha256(
501
+ {
502
+ "kind": "descriptor-confined-local-file",
503
+ "content_sha256": classification.content_sha256,
504
+ "size_bytes": len(content),
505
+ "identity_checked_before_after": True,
506
+ "symlink_hardlink_special_file_rejected": True,
507
+ }
508
+ )
509
+ content_urn = f"urn:sha256:{classification.content_sha256}"
510
+ return _output(
511
+ request=request,
512
+ descriptor=self.descriptor,
513
+ governance=self.governance,
514
+ snapshot_path=snapshot_path,
515
+ content=content,
516
+ data_format=data_format,
517
+ media_type=media_type,
518
+ source_uri=content_urn,
519
+ final_uri=content_urn,
520
+ classification=classification.classification,
521
+ transport_evidence_digest=transport_digest,
522
+ parsed_schema_digest=sandboxed.parsed.schema_digest,
523
+ )
524
+
525
+ def _acquire_decoded(
526
+ self,
527
+ request: AcquisitionRequest,
528
+ context: AcquisitionContext,
529
+ *,
530
+ relative_path: str,
531
+ sealed_format: str,
532
+ ) -> AcquisitionOutput:
533
+ """Acquire one source whose recipe pins a Reader.
534
+
535
+ The order is the design. The family is warmed up first, so a suspect Reader never
536
+ touches source data. Then the fetched bytes are decoded before anything is classified
537
+ or sealed, so classification and the rights gate scan the normalized artifact -- the
538
+ data a build actually reads -- and the sealed snapshot is the Reader's canonical
539
+ output. The fetched bytes leave one trace, their digest, which is what gives the
540
+ derived snapshot a link back to what produced it.
541
+ """
542
+
543
+ assert request.reader_pin is not None
544
+ _warm_up_family(
545
+ self.sandbox,
546
+ pin=request.reader_pin,
547
+ context=context,
548
+ sealed_format=sealed_format,
549
+ )
550
+ content = _read_confined_file(
551
+ self.input_root,
552
+ relative_path=relative_path,
553
+ max_bytes=context.max_source_bytes,
554
+ )
555
+ decoded = _decode_in_clean_room(
556
+ self.sandbox,
557
+ request=request,
558
+ context=context,
559
+ content=content,
560
+ sealed_format=sealed_format,
561
+ )
562
+ classification = classify_before_persistence(
563
+ decoded.content,
564
+ declared_classification=self.governance.declared_classification,
565
+ lawful_basis_declared=self.governance.lawful_basis_declared,
566
+ )
567
+ require_admissible(classification)
568
+ snapshot_path = seal_content_addressed_snapshot(
569
+ context.snapshot_root,
570
+ content=decoded.content,
571
+ suffix=sealed_format,
572
+ )
573
+ transport_digest = canonical_sha256(
574
+ {
575
+ "kind": "descriptor-confined-local-file",
576
+ "content_sha256": decoded.record.fetched_content_sha256,
577
+ "size_bytes": len(content),
578
+ "identity_checked_before_after": True,
579
+ "symlink_hardlink_special_file_rejected": True,
580
+ }
581
+ )
582
+ # The URN names the bytes that were fetched, which is what this source is. The sealed
583
+ # snapshot is a derivative of them and carries its own digest on the receipt.
584
+ content_urn = f"urn:sha256:{decoded.record.fetched_content_sha256}"
585
+ return _output(
586
+ request=request,
587
+ descriptor=self.descriptor,
588
+ governance=self.governance,
589
+ snapshot_path=snapshot_path,
590
+ content=decoded.content,
591
+ data_format=sealed_format,
592
+ media_type=decoded.media_type,
593
+ source_uri=content_urn,
594
+ final_uri=content_urn,
595
+ classification=classification.classification,
596
+ transport_evidence_digest=transport_digest,
597
+ parsed_schema_digest=decoded.parsed_schema_digest,
598
+ decode=decoded.record,
599
+ )
600
+
601
+
602
+ @dataclass(frozen=True)
603
+ class RangeReaderHttpsAdapter:
604
+ """Normal registry adapter for one locked HTTPS range plus a pinned Reader."""
605
+
606
+ sandbox: CrawlerSandbox
607
+ allowed_hostnames: tuple[str, ...]
608
+ governance: AdapterGovernance
609
+ descriptor: AdapterDescriptor
610
+ retrieval_limits: RetrievalLimits = field(default_factory=RetrievalLimits)
611
+
612
+ def __post_init__(self) -> None:
613
+ require_crawler_sandbox(self.sandbox)
614
+ if self.descriptor.source_class != "external_adapter":
615
+ raise SourceContractError(
616
+ "ADAPTER_CLASS",
617
+ "adapter.descriptor.source_class",
618
+ "range Reader adapter requires external_adapter",
619
+ )
620
+ if not self.allowed_hostnames:
621
+ raise SourceContractError(
622
+ "EGRESS_ALLOWLIST",
623
+ "adapter.allowed_hostnames",
624
+ "range Reader adapter requires an exact non-empty host allowlist",
625
+ )
626
+
627
+ def acquire(
628
+ self,
629
+ request: AcquisitionRequest,
630
+ context: AcquisitionContext,
631
+ ) -> AcquisitionOutput:
632
+ _bind_request(request, self.descriptor)
633
+ if request.credential_reference_id is not None:
634
+ raise SourceContractError(
635
+ "AUTHENTICATED_SOURCE_REQUIRES_TRUSTED_GATEWAY",
636
+ "request.credential_reference_id",
637
+ "range Courier receives no credential",
638
+ )
639
+ if request.reader_pin is None:
640
+ raise SourceContractError(
641
+ "READER_PIN_REQUIRED",
642
+ "request.reader_pin",
643
+ "range Reader acquisition requires an exact Reader pin",
644
+ )
645
+ query = _exact_query(request.query, {"revision_lock"})
646
+ try:
647
+ lock = RangeRevisionLock.from_dict(query["revision_lock"])
648
+ pin = ReaderPin(
649
+ request.reader_pin["family_id"],
650
+ request.reader_pin["family_version"],
651
+ request.reader_pin["decode_options"],
652
+ )
653
+ budgets = ReaderBudgets().narrowed_by(request.resource_caps)
654
+ except (AcquisitionSecurityError, KeyError) as error:
655
+ raise SourceContractError(
656
+ "RANGE_READER_REQUEST",
657
+ "request.query",
658
+ "range Reader authority is malformed",
659
+ ) from error
660
+ expected_fetch, expected_decode = default_policy_digests()
661
+ if context.crawler_sandbox_attestation_digest != range_reader_policy_digest():
662
+ raise SourceContractError(
663
+ "SANDBOX_ATTESTATION_MISMATCH",
664
+ "context.crawler_sandbox_attestation_digest",
665
+ "context does not pin the composed fetch-plus-decode policies",
666
+ )
667
+ _authorize_candidate(request, self.governance)
668
+ _warm_up_family(
669
+ self.sandbox,
670
+ pin=request.reader_pin,
671
+ context=context,
672
+ sealed_format="csv",
673
+ expected_policy_digest=expected_decode,
674
+ )
675
+ composed = acquire_range_reader(
676
+ self.sandbox,
677
+ request_id=request.request_id,
678
+ lock=lock,
679
+ allowed_hostnames=self.allowed_hostnames,
680
+ retrieval_limits=self.retrieval_limits,
681
+ reader_pin=pin,
682
+ reader_budgets=budgets,
683
+ parse_limits=_parse_limits(context),
684
+ expected_fetch_policy_digest=expected_fetch,
685
+ expected_decode_policy_digest=expected_decode,
686
+ )
687
+ classification = classify_before_persistence(
688
+ composed.content,
689
+ declared_classification=self.governance.declared_classification,
690
+ lawful_basis_declared=self.governance.lawful_basis_declared,
691
+ )
692
+ require_admissible(classification)
693
+ snapshot_path = seal_content_addressed_snapshot(
694
+ context.snapshot_root,
695
+ content=composed.content,
696
+ suffix="csv",
697
+ )
698
+ observation = SourceObservation(
699
+ source_id=request.source_id,
700
+ observed_at=self.governance.observation.observed_at,
701
+ event_time_field=self.governance.observation.event_time_field,
702
+ available_at_field=self.governance.observation.available_at_field,
703
+ historical_start=self.governance.observation.historical_start,
704
+ historical_end=self.governance.observation.historical_end,
705
+ live_status=self.governance.observation.live_status,
706
+ publication_delay_seconds=self.governance.observation.publication_delay_seconds,
707
+ update_frequency_seconds=self.governance.observation.update_frequency_seconds,
708
+ response_status=self.governance.observation.response_status,
709
+ evidence=self.governance.observation.evidence,
710
+ )
711
+ snapshot_digest = sha256_bytes(composed.content)
712
+ receipt_members = tuple(_receipt_v2_member(member) for member in composed.fetched_members)
713
+ receipt = AcquisitionReceiptV2(
714
+ receipt_id=f"receipt.{request.digest[:24]}",
715
+ request_digest=request.digest,
716
+ source_id=request.source_id,
717
+ adapter=self.descriptor,
718
+ query_digest=request.query_digest,
719
+ snapshot=SnapshotReference(
720
+ content_sha256=snapshot_digest,
721
+ size_bytes=len(composed.content),
722
+ media_type=composed.media_type,
723
+ data_format="csv",
724
+ storage_name=snapshot_path.name,
725
+ ),
726
+ fetched_content_sha256=lock.slices[0].sha256,
727
+ family_id=pin.family_id,
728
+ family_version=pin.family_version,
729
+ decode_options_digest=pin.options_digest,
730
+ decode_flags=composed.decode_flags,
731
+ source_uri=lock.object_url,
732
+ final_uri=lock.object_url,
733
+ event_time_field=self.governance.observation.event_time_field,
734
+ observed_at=self.governance.observation.observed_at,
735
+ available_at=self.governance.observation.available_at,
736
+ ingested_at=request.requested_at,
737
+ classification=classification.classification,
738
+ rights_digest=self.governance.rights.digest,
739
+ retention_digest=self.governance.retention.digest,
740
+ observation_digest=observation.digest,
741
+ transport_evidence_digest=canonical_sha256(
742
+ [member.hop_document() for member in receipt_members]
743
+ ),
744
+ fetched_members=receipt_members,
745
+ cycle=lock.cycle,
746
+ revision_lock_digest=lock.digest,
747
+ source_revision_digest=fetched_source_revision_digest(
748
+ lock.cycle,
749
+ lock.full_object_size_bytes,
750
+ receipt_members,
751
+ ),
752
+ full_object_size_bytes=lock.full_object_size_bytes,
753
+ efficiency_numerator=composed.efficiency_numerator,
754
+ efficiency_denominator=lock.full_object_size_bytes,
755
+ fetch_policy_digest=expected_fetch,
756
+ decode_policy_digest=expected_decode,
757
+ parsed_schema_digest=composed.parsed_schema_digest,
758
+ )
759
+ return AcquisitionOutput(
760
+ receipt=receipt,
761
+ observation=observation,
762
+ parsed_schema_digest=composed.parsed_schema_digest,
763
+ snapshot_path=snapshot_path,
764
+ )
765
+
766
+
767
+ @dataclass(frozen=True)
768
+ class PublicHttpsAdapter:
769
+ """Public-only URL/API adapter using the real credential-free crawler boundary."""
770
+
771
+ sandbox: CrawlerSandbox
772
+ allowed_hostnames: tuple[str, ...]
773
+ governance: AdapterGovernance
774
+ descriptor: AdapterDescriptor
775
+ retrieval_limits: RetrievalLimits = field(default_factory=RetrievalLimits)
776
+
777
+ def __post_init__(self) -> None:
778
+ if self.descriptor.source_class not in {"user_url", "user_api", "external_adapter"}:
779
+ raise SourceContractError(
780
+ "ADAPTER_CLASS",
781
+ "adapter.descriptor.source_class",
782
+ "public HTTPS adapter requires user_url, user_api, or external_adapter",
783
+ )
784
+ if not self.allowed_hostnames:
785
+ raise SourceContractError(
786
+ "EGRESS_ALLOWLIST",
787
+ "adapter.allowed_hostnames",
788
+ "public HTTPS adapter requires an exact non-empty host allowlist",
789
+ )
790
+
791
+ def acquire(
792
+ self,
793
+ request: AcquisitionRequest,
794
+ context: AcquisitionContext,
795
+ ) -> AcquisitionOutput:
796
+ _bind_request(request, self.descriptor)
797
+ if request.credential_reference_id is not None:
798
+ raise SourceContractError(
799
+ "AUTHENTICATED_SOURCE_REQUIRES_TRUSTED_GATEWAY",
800
+ "request.credential_reference_id",
801
+ "generic crawler receives no credential; use a trusted scoped gateway",
802
+ )
803
+ url, data_format, filename = _https_query(request.query)
804
+ _authorize_candidate(request, self.governance)
805
+ retrieval_limits = replace(
806
+ self.retrieval_limits,
807
+ max_response_bytes=min(
808
+ self.retrieval_limits.max_response_bytes,
809
+ context.max_source_bytes,
810
+ ),
811
+ )
812
+ if request.reader_pin is None:
813
+ sandboxed = self.sandbox.retrieve_and_parse(
814
+ request_id=request.request_id,
815
+ url=url,
816
+ allowed_hostnames=self.allowed_hostnames,
817
+ data_format=data_format,
818
+ filename=filename,
819
+ parse_limits=_parse_limits(context),
820
+ retrieval_limits=retrieval_limits,
821
+ )
822
+ decode = None
823
+ else:
824
+ pin = request.reader_pin
825
+ expected_retrieval = sandbox_policy_digest("retrieve_decode_and_parse")
826
+ expected_decode = sandbox_policy_digest("decode_and_parse")
827
+ if context.crawler_sandbox_attestation_digest != https_reader_policy_digest():
828
+ raise SourceContractError(
829
+ "SANDBOX_ATTESTATION_MISMATCH",
830
+ "context.crawler_sandbox_attestation_digest",
831
+ "context does not pin the composed HTTPS retrieval-plus-decode policies",
832
+ )
833
+ retrieval_limits = reader_retrieval_limits(
834
+ retrieval_limits,
835
+ family_id=str(pin["family_id"]),
836
+ family_version=str(pin["family_version"]),
837
+ )
838
+ _warm_up_family(
839
+ self.sandbox,
840
+ pin=pin,
841
+ context=context,
842
+ sealed_format=data_format,
843
+ expected_policy_digest=expected_decode,
844
+ )
845
+ sandboxed = self.sandbox.retrieve_decode_and_parse(
846
+ request_id=request.request_id,
847
+ url=url,
848
+ allowed_hostnames=self.allowed_hostnames,
849
+ reader_pin=pin,
850
+ output_format=data_format,
851
+ parse_limits=_parse_limits(context),
852
+ retrieval_limits=retrieval_limits,
853
+ reader_budgets=_reader_budget_caps(context, request),
854
+ )
855
+ if (
856
+ sandboxed.fetched_content_sha256 is None
857
+ or sandboxed.decode_family_id is None
858
+ or sandboxed.decode_family_version is None
859
+ or sandboxed.decode_options_digest is None
860
+ or sandboxed.decode_flags is None
861
+ ):
862
+ raise SourceContractError(
863
+ "SANDBOX_RESULT",
864
+ "sandbox",
865
+ "network Reader acquisition result is incomplete",
866
+ )
867
+ decode = DecodeRecord(
868
+ fetched_content_sha256=sandboxed.fetched_content_sha256,
869
+ family_id=sandboxed.decode_family_id,
870
+ family_version=sandboxed.decode_family_version,
871
+ decode_options_digest=sandboxed.decode_options_digest,
872
+ flags=sandboxed.decode_flags,
873
+ )
874
+ if request.reader_pin is None:
875
+ _require_sandbox_binding(sandboxed.policy_digest, context)
876
+ elif sandboxed.policy_digest != expected_retrieval:
877
+ raise SourceContractError(
878
+ "SANDBOX_ATTESTATION_MISMATCH",
879
+ "sandbox.policy_digest",
880
+ "HTTPS Reader acquisition does not bind its retrieval-plus-decode policy",
881
+ )
882
+ if (
883
+ sandboxed.content is None
884
+ or sandboxed.parsed is None
885
+ or sandboxed.media_type is None
886
+ or sandboxed.final_url is None
887
+ or sandboxed.transport_evidence_digest is None
888
+ ):
889
+ raise SourceContractError(
890
+ "SANDBOX_RESULT",
891
+ "sandbox",
892
+ "network acquisition result is incomplete",
893
+ )
894
+ classification = classify_before_persistence(
895
+ sandboxed.content,
896
+ declared_classification=self.governance.declared_classification,
897
+ lawful_basis_declared=self.governance.lawful_basis_declared,
898
+ )
899
+ require_admissible(classification)
900
+ if sandboxed.parsed.input_sha256 != classification.content_sha256:
901
+ raise SourceContractError(
902
+ "SANDBOX_RESULT",
903
+ "sandbox.parsed.input_sha256",
904
+ "sandbox result does not bind the classified exact response bytes",
905
+ )
906
+ snapshot_path = seal_content_addressed_snapshot(
907
+ context.snapshot_root,
908
+ content=sandboxed.content,
909
+ suffix=data_format,
910
+ )
911
+ return _output(
912
+ request=request,
913
+ descriptor=self.descriptor,
914
+ governance=self.governance,
915
+ snapshot_path=snapshot_path,
916
+ content=sandboxed.content,
917
+ data_format=data_format,
918
+ media_type=sandboxed.media_type,
919
+ source_uri=url,
920
+ final_uri=sandboxed.final_url,
921
+ classification=classification.classification,
922
+ transport_evidence_digest=sandboxed.transport_evidence_digest,
923
+ parsed_schema_digest=sandboxed.parsed.schema_digest,
924
+ decode=decode,
925
+ probe_transport=sandboxed.probe_transport,
926
+ parsed_row_count=len(sandboxed.parsed.rows),
927
+ parsed_row_digest=row_digest_for(sandboxed.parsed),
928
+ )
929
+
930
+
931
+ def _bind_request(request: AcquisitionRequest, descriptor: AdapterDescriptor) -> None:
932
+ if (request.adapter_id, request.adapter_version) != (
933
+ descriptor.adapter_id,
934
+ descriptor.adapter_version,
935
+ ):
936
+ raise SourceContractError(
937
+ "ADAPTER_BINDING",
938
+ "request",
939
+ "request does not bind the exact adapter identity",
940
+ )
941
+
942
+
943
+ def _local_query(query: Any) -> tuple[str, str, str]:
944
+ data = _exact_query(query, {"relative_path", "data_format", "media_type"})
945
+ return (
946
+ _text(data["relative_path"], "query.relative_path", 1_024),
947
+ _format(data["data_format"]),
948
+ _text(data["media_type"], "query.media_type", 255),
949
+ )
950
+
951
+
952
+ def _https_query(query: Any) -> tuple[str, str, str]:
953
+ data = _exact_query(query, {"url", "data_format", "filename"})
954
+ return (
955
+ _https_url(data["url"]),
956
+ _format(data["data_format"]),
957
+ _text(data["filename"], "query.filename", 255),
958
+ )
959
+
960
+
961
+ def _https_url(value: Any) -> str:
962
+ """Admit the address as a literal string or as the closed composition object.
963
+
964
+ The key set of the query is unchanged; the shape change lives inside one value. A
965
+ string behaves exactly as it always has. The object exists so a model-run address --
966
+ ``gfs.20260806/06/atmos/gfs.t06z.pgrb2.0p25.f003`` -- can be assembled from bounded
967
+ literal parts and the rendered cycle coordinates, which whole-string substitution alone
968
+ could never express. It is deliberately not free-form interpolation: there is no slot
969
+ in the shape for a scheme, an authority, a query or a fragment, so those cannot be
970
+ reached from a recipe at all.
971
+ """
972
+
973
+ if isinstance(value, dict):
974
+ return _compose_url(value)
975
+ return _text(value, "query.url", URL_MAX_LENGTH)
976
+
977
+
978
+ def _compose_url(value: dict[str, Any]) -> str:
979
+ """Assemble ``origin`` and an ordered list of path components into one address."""
980
+
981
+ if set(value) != {"origin", "path"}:
982
+ raise SourceContractError(
983
+ "URL_COMPOSITION",
984
+ "query.url",
985
+ "composed address must contain exactly ['origin', 'path']",
986
+ )
987
+ origin = _composition_origin(value["origin"])
988
+ components = _composition_components(value["path"])
989
+ composed = origin + "/" + "/".join(components)
990
+ if len(composed) > URL_MAX_LENGTH:
991
+ raise SourceContractError(
992
+ "URL_COMPOSITION",
993
+ "query.url",
994
+ f"composed address must be no longer than {URL_MAX_LENGTH} characters",
995
+ )
996
+ return composed
997
+
998
+
999
+ def _composition_origin(origin: Any) -> str:
1000
+ """Return the origin, which carries a scheme, a host, an optional port, and nothing else.
1001
+
1002
+ The origin is a separate field that no part can be composed into, so a composed address
1003
+ can never name a host the recipe did not state. Everything a URL can carry after the
1004
+ authority -- path, query, fragment, user information -- is refused here rather than
1005
+ stripped, because stripping would silently accept an address the author wrote and the
1006
+ adapter did not fetch.
1007
+ """
1008
+
1009
+ if (
1010
+ not isinstance(origin, str)
1011
+ or not 1 <= len(origin) <= URL_ORIGIN_MAX_LENGTH
1012
+ or not origin.isascii()
1013
+ ):
1014
+ raise SourceContractError(
1015
+ "URL_COMPOSITION",
1016
+ "query.url.origin",
1017
+ f"must be ASCII text no longer than {URL_ORIGIN_MAX_LENGTH} characters",
1018
+ )
1019
+ try:
1020
+ parsed = urlsplit(origin)
1021
+ port = parsed.port
1022
+ host = parsed.hostname
1023
+ except ValueError:
1024
+ raise SourceContractError(
1025
+ "URL_COMPOSITION", "query.url.origin", "is not a parseable origin"
1026
+ ) from None
1027
+ if (
1028
+ parsed.scheme != "https"
1029
+ or not host
1030
+ or parsed.username is not None
1031
+ or parsed.password is not None
1032
+ or parsed.path
1033
+ or parsed.query
1034
+ or parsed.fragment
1035
+ # Reconstruction equality is the structural claim: the origin is exactly a scheme
1036
+ # and an authority, spelled the one way that round-trips. It refuses an uppercase
1037
+ # scheme, a stray ``?`` or ``#`` that split into empty components, and any other
1038
+ # spelling whose parse disagrees with its text.
1039
+ or origin != f"https://{parsed.netloc}"
1040
+ ):
1041
+ raise SourceContractError(
1042
+ "URL_COMPOSITION",
1043
+ "query.url.origin",
1044
+ "must be an https scheme with a host and optional port and nothing else",
1045
+ )
1046
+ if port is not None and not 1 <= port <= 65_535:
1047
+ raise SourceContractError("URL_COMPOSITION", "query.url.origin", "states an invalid port")
1048
+ return origin
1049
+
1050
+
1051
+ def _composition_components(path: Any) -> list[str]:
1052
+ """Return the joined path components, each built from bounded literal parts."""
1053
+
1054
+ if not isinstance(path, list) or not 1 <= len(path) <= URL_MAX_PATH_COMPONENTS:
1055
+ raise SourceContractError(
1056
+ "URL_COMPOSITION",
1057
+ "query.url.path",
1058
+ f"must be a list of 1 to {URL_MAX_PATH_COMPONENTS} components",
1059
+ )
1060
+ components: list[str] = []
1061
+ for index, component in enumerate(path):
1062
+ component_path = f"query.url.path[{index}]"
1063
+ if not isinstance(component, list) or not 1 <= len(component) <= URL_MAX_COMPONENT_PARTS:
1064
+ raise SourceContractError(
1065
+ "URL_COMPOSITION",
1066
+ component_path,
1067
+ f"must be a list of 1 to {URL_MAX_COMPONENT_PARTS} parts",
1068
+ )
1069
+ for part_index, part in enumerate(component):
1070
+ if not isinstance(part, str) or not _URL_PART.fullmatch(part):
1071
+ raise SourceContractError(
1072
+ "URL_COMPOSITION",
1073
+ f"{component_path}[{part_index}]",
1074
+ f"must be 1 to {URL_PART_MAX_LENGTH} characters of A-Z a-z 0-9 . _ -",
1075
+ )
1076
+ joined = "".join(component)
1077
+ if not 1 <= len(joined) <= URL_COMPONENT_MAX_LENGTH or joined in {".", ".."}:
1078
+ raise SourceContractError(
1079
+ "URL_COMPOSITION",
1080
+ component_path,
1081
+ f"must join to 1 to {URL_COMPONENT_MAX_LENGTH} characters and not to . or ..",
1082
+ )
1083
+ components.append(joined)
1084
+ return components
1085
+
1086
+
1087
+ def _exact_query(query: Any, expected: set[str]) -> dict[str, Any]:
1088
+ if not isinstance(query, dict) or set(query) != expected:
1089
+ raise SourceContractError(
1090
+ "QUERY_FIELDS",
1091
+ "request.query",
1092
+ f"must contain exactly {sorted(expected)}",
1093
+ )
1094
+ return query
1095
+
1096
+
1097
+ def _format(value: Any) -> str:
1098
+ text = _text(value, "query.data_format", 16)
1099
+ if text not in DATA_FORMATS:
1100
+ raise SourceContractError(
1101
+ "DATA_FORMAT",
1102
+ "query.data_format",
1103
+ "must be one of " + ", ".join(sorted(DATA_FORMATS)),
1104
+ )
1105
+ return text
1106
+
1107
+
1108
+ def _text(value: Any, path: str, maximum: int) -> str:
1109
+ if not isinstance(value, str) or not value or len(value) > maximum:
1110
+ raise SourceContractError("TEXT", path, f"must be a string no longer than {maximum}")
1111
+ if any(0xD800 <= ord(char) <= 0xDFFF for char in value):
1112
+ raise SourceContractError("UNICODE", path, "contains invalid Unicode")
1113
+ return value
1114
+
1115
+
1116
+ def _require_disjoint_roots(input_root: Path, snapshot_root: Path) -> None:
1117
+ try:
1118
+ common = Path(os.path.commonpath((input_root, snapshot_root)))
1119
+ except ValueError:
1120
+ return
1121
+ if common in {input_root, snapshot_root}:
1122
+ raise SourceContractError(
1123
+ "ROOT_OVERLAP",
1124
+ "context.snapshot_root",
1125
+ "input and snapshot roots must be disjoint",
1126
+ )
1127
+
1128
+
1129
+ def _read_confined_file(root: Path, *, relative_path: str, max_bytes: int) -> bytes:
1130
+ path = PurePosixPath(relative_path)
1131
+ if (
1132
+ path.is_absolute()
1133
+ or not path.parts
1134
+ or path.as_posix() != relative_path
1135
+ or any(part in {"", ".", ".."} or part.startswith(".") for part in path.parts)
1136
+ ):
1137
+ raise SourceContractError(
1138
+ "SOURCE_PATH",
1139
+ "query.relative_path",
1140
+ "must be a canonical non-hidden relative POSIX path",
1141
+ )
1142
+ directory_flags = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0)
1143
+ descriptors: list[int] = [os.open(root, directory_flags)]
1144
+ try:
1145
+ for component in path.parts[:-1]:
1146
+ descriptors.append(os.open(component, directory_flags, dir_fd=descriptors[-1]))
1147
+ descriptor = os.open(
1148
+ path.parts[-1],
1149
+ os.O_RDONLY
1150
+ | getattr(os, "O_NOFOLLOW", 0)
1151
+ | getattr(os, "O_NONBLOCK", 0)
1152
+ | getattr(os, "O_CLOEXEC", 0),
1153
+ dir_fd=descriptors[-1],
1154
+ )
1155
+ try:
1156
+ before = os.fstat(descriptor)
1157
+ if (
1158
+ not stat.S_ISREG(before.st_mode)
1159
+ or before.st_nlink != 1
1160
+ or before.st_size < 1
1161
+ or before.st_size > max_bytes
1162
+ ):
1163
+ raise SourceContractError(
1164
+ "SOURCE_FILE",
1165
+ "query.relative_path",
1166
+ "must be one non-hardlinked, non-empty, bounded regular file",
1167
+ )
1168
+ chunks: list[bytes] = []
1169
+ total = 0
1170
+ while total <= max_bytes:
1171
+ chunk = os.read(descriptor, min(64 * 1024, max_bytes + 1 - total))
1172
+ if not chunk:
1173
+ break
1174
+ chunks.append(chunk)
1175
+ total += len(chunk)
1176
+ after = os.fstat(descriptor)
1177
+ current = os.stat(path.parts[-1], dir_fd=descriptors[-1], follow_symlinks=False)
1178
+ if (
1179
+ total > max_bytes
1180
+ or total != before.st_size
1181
+ or _source_stat_signature(before) != _source_stat_signature(after)
1182
+ or _source_stat_signature(before) != _source_stat_signature(current)
1183
+ ):
1184
+ raise SourceContractError(
1185
+ "SOURCE_MUTATED",
1186
+ "query.relative_path",
1187
+ "source changed during its descriptor-bound read",
1188
+ )
1189
+ return b"".join(chunks)
1190
+ finally:
1191
+ os.close(descriptor)
1192
+ except OSError:
1193
+ raise SourceContractError(
1194
+ "SOURCE_PATH",
1195
+ "query.relative_path",
1196
+ "source path could not be opened without following links",
1197
+ ) from None
1198
+ finally:
1199
+ for descriptor in reversed(descriptors):
1200
+ os.close(descriptor)
1201
+
1202
+
1203
+ def _source_stat_signature(info: os.stat_result) -> tuple[int, ...]:
1204
+ """Bind the complete regular-file state used by a confined source observation."""
1205
+
1206
+ return (
1207
+ info.st_dev,
1208
+ info.st_ino,
1209
+ info.st_mode,
1210
+ info.st_nlink,
1211
+ info.st_size,
1212
+ info.st_mtime_ns,
1213
+ info.st_ctime_ns,
1214
+ )