mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,1217 @@
1
+ """Bounded streaming authoring shards behind one atomic manifest.
2
+
3
+ A converged Data.gov v4 generation is immutable staging: raw response blobs, normalized page
4
+ manifests over bounded record segments, segmented evidence indexes, and one terminal externally
5
+ merged pass map. This module turns that generation into immutable authoring
6
+ shards without ever holding it in memory, and replaces the 64 MiB monolithic
7
+ ``mr-data-catalog-fill-authoring.v1`` bundle for the provider-scale workflow. The legacy bounded
8
+ authoring path in :mod:`authoring` is untouched and still readable.
9
+
10
+ The shape of the work is deliberate.
11
+
12
+ *Two bounded passes, never one big map.* The first pass streams the terminal pass's normalized
13
+ pages one segment at a time into a disk-backed scratch index keyed by the exact provider identifier
14
+ bytes. The second pass walks that index in ascending identifier order and authors one record at a
15
+ time into one open output shard. Peak retained records is therefore one input segment plus one
16
+ output shard, not one generation, and the authoring order is a property of the provider identifiers
17
+ rather than of page arrival.
18
+
19
+ *Immutable members, manifest last.* Every shard is content-addressed, installed through a
20
+ same-directory link with fsync and exact readback, and never rewritten. A small journal records
21
+ the committed shards and the exact last committed provider record so an interrupted run resumes
22
+ without duplicating or reordering anything. ``manifest.json`` is written only when the whole
23
+ generation is authored, so a crash leaves either a resumable partial generation or a complete one
24
+ -- never a half-installed head.
25
+
26
+ *Exact coordinates only.* Authoring refuses to start unless the caller names the exact live fill
27
+ checkpoint digest, the fill reports a converged terminal pass whose pass map authenticates, and the
28
+ indexed record population equals that pass map's unique record count. A resume additionally
29
+ requires the exact retained journal digest.
30
+
31
+ *One stated coverage, bounded or not.* A converged sweep can be larger than a release can carry,
32
+ so a caller may ask for a bounded selection of it instead of all of it: an ``AuthoringSelection``
33
+ names one registered rule and how many records to take under it. The rule is a total order over
34
+ the *whole* corpus, evaluated after the index is built and before one record is authored, so the
35
+ records it selects are a property of the corpus rather than of how far this invocation got. That
36
+ is what makes a bounded generation ``complete``: it authored the whole of what it said it would.
37
+
38
+ A bound is emphatically not the record budget. ``limits.max_records`` is a resource guard, it is
39
+ live on every invocation bounded or not, and a run that reaches it still stops at
40
+ ``AUTHOR_RECORD_LIMIT`` with ``status="incomplete"`` and publishes no manifest. An operator who
41
+ sets a budget below their own bound gets an incomplete generation, exactly as they would have
42
+ before this existed. Nothing here relaxes that; ``coverage`` is a claim about which records were
43
+ asked for, ``limits`` is a claim about what this process was allowed to spend, and neither stands
44
+ in for the other.
45
+ """
46
+
47
+ from __future__ import annotations
48
+
49
+ import contextlib
50
+ import os
51
+ import re
52
+ import secrets
53
+ import sqlite3
54
+ import stat
55
+ import tempfile
56
+ import time
57
+ from collections.abc import Callable, Iterator
58
+ from dataclasses import dataclass
59
+ from datetime import UTC, datetime
60
+ from pathlib import Path
61
+ from typing import Any
62
+
63
+ from mostlyright.data_harness.canonical import (
64
+ CanonicalJSONError,
65
+ canonical_json_bytes,
66
+ canonical_sha256,
67
+ parse_canonical_json,
68
+ sha256_bytes,
69
+ )
70
+ from mostlyright.data_harness.sources.catalog.authoring_policy import (
71
+ AuthoringPolicy,
72
+ author_catalog_record,
73
+ resolve_authoring_policy,
74
+ )
75
+ from mostlyright.data_harness.sources.catalog.bounded_io import (
76
+ BoundedReadFailure,
77
+ read_bounded_at,
78
+ )
79
+ from mostlyright.data_harness.sources.catalog.contracts import EMBEDDING_LAYERS
80
+ from mostlyright.data_harness.sources.catalog.coverage import (
81
+ COVERAGE_EXHAUSTIVE_RULE,
82
+ SELECTION_RULES,
83
+ coverage_is_valid,
84
+ )
85
+ from mostlyright.data_harness.sources.catalog.fill_partitions import validate_pass_map
86
+ from mostlyright.data_harness.sources.catalog.fill_staging import FillStaging
87
+ from mostlyright.data_harness.sources.contracts import SourceContractError
88
+
89
+ AUTHORING_SHARD_SCHEMA = "harness-catalog-authoring-shard.v1"
90
+ AUTHORING_MANIFEST_SCHEMA = "harness-catalog-authoring-manifest.v1"
91
+ AUTHORING_JOURNAL_SCHEMA = "harness-catalog-authoring-journal.v1"
92
+ AUTHORING_RECEIPT_SCHEMA = "mr-data-catalog-author.v1"
93
+
94
+ MANIFEST_FILENAME = "manifest.json"
95
+ JOURNAL_FILENAME = "journal.json"
96
+ SHARDS_DIRNAME = "shards"
97
+
98
+ MAX_AUTHORING_SHARD_ENTRIES = 1_000
99
+ MAX_AUTHORING_SHARD_BYTES = 8 * 1024 * 1024
100
+ MAX_AUTHORING_MANIFEST_BYTES = 8 * 1024 * 1024
101
+ MAX_AUTHORING_JOURNAL_BYTES = 8 * 1024 * 1024
102
+
103
+ #: Exactly what one authored shard descriptor states, and nothing else. The journal and the
104
+ #: manifest both carry these descriptors, and the composite generation receipt carries them
105
+ #: onward as authoritative evidence, so every reader closes the key set rather than reading the
106
+ #: members it happens to know and copying whatever else the document brought with it.
107
+ AUTHORING_SHARD_DESCRIPTOR_MEMBERS = frozenset(
108
+ {
109
+ "sha256",
110
+ "bytes",
111
+ "item_count",
112
+ "entry_count",
113
+ "first_provider_record_id",
114
+ "last_provider_record_id",
115
+ }
116
+ )
117
+
118
+
119
+ def shard_descriptor_is_valid(descriptor: Any) -> bool:
120
+ """Whether one authored shard descriptor is exactly its contract, with admissible values.
121
+
122
+ Every reader that admits a descriptor answers this one question, so that adding a member to
123
+ the contract is one edit rather than three that have to agree. Each caller still raises its
124
+ own typed refusal, because the boundary the descriptor failed at is what an operator needs.
125
+ """
126
+
127
+ return (
128
+ isinstance(descriptor, dict)
129
+ and set(descriptor) == AUTHORING_SHARD_DESCRIPTOR_MEMBERS
130
+ and _is_digest(descriptor["sha256"])
131
+ and type(descriptor["bytes"]) is int
132
+ and 1 <= descriptor["bytes"] <= MAX_AUTHORING_SHARD_BYTES
133
+ and type(descriptor["item_count"]) is int
134
+ and 1 <= descriptor["item_count"] <= MAX_AUTHORING_SHARD_ENTRIES
135
+ and type(descriptor["entry_count"]) is int
136
+ and 0 <= descriptor["entry_count"] <= descriptor["item_count"]
137
+ and isinstance(descriptor["first_provider_record_id"], str)
138
+ and isinstance(descriptor["last_provider_record_id"], str)
139
+ )
140
+
141
+
142
+ # Conservative allowance for the shard envelope around the items array: the schema label, the
143
+ # policy digest, the two bounded provider record identifiers, and the JSON punctuation. The real
144
+ # ceiling is still enforced exactly when the shard bytes are composed.
145
+ _SHARD_ENVELOPE_BYTES = 4_096
146
+
147
+ #: The provider observation ``last_harvested_date_desc`` orders by. It is a *volatile* field --
148
+ #: deliberately outside the semantic digest the pass map converges on, because a digest that moved
149
+ #: with a nightly re-harvest would make two adjacent equal pass maps unsatisfiable. That is
150
+ #: exactly right for a selection key: the corpus it orders is immutable staging, so the order is
151
+ #: reproducible from the generation even though the value is not part of any record's identity.
152
+ _RANK_OBSERVATION_PATH = "$.last_harvested_date"
153
+
154
+ #: The longest provider harvest date this rule will order by. A well-formed instant is 20
155
+ #: characters; 64 is room for every spelling of one. Anything longer is not a date this rule can
156
+ #: order by, so it ranks with the undated rather than sorting on its leading bytes.
157
+ MAX_SELECTION_RANK_CHARS = 64
158
+
159
+ #: The shape a provider value must have before this rule will treat it as a harvest date.
160
+ #:
161
+ #: Byte order is only chronological order over values that are actually dates. Comparison here is
162
+ #: bytewise, so *any* string sorts somewhere -- and because ``'N' < '2' < 'u'`` is false in the
163
+ #: middle, a record whose harvest date the provider filled in as ``unknown`` or ``pending`` would
164
+ #: sort above every real ISO-8601 instant and head the catalogue. "The N most recently harvested
165
+ #: datasets" would then be led by the N records with no harvest date at all, which is the exact
166
+ #: opposite of what the rule's name promises.
167
+ #:
168
+ #: So a value is a rank only if it opens with a calendar date, ``YYYY-MM-DD``. Every spelling of
169
+ #: an ISO-8601 instant does, and no English word does. The rule deliberately checks the opening
170
+ #: rather than the whole grammar: what the bytes after the date look like varies by provider and
171
+ #: does not change which day is later, and refusing a real date over its seconds field would drop
172
+ #: a record from the catalogue to satisfy a parser.
173
+ _RANK_DATE_PREFIX = re.compile(r"\d{4}-\d{2}-\d{2}")
174
+
175
+ #: What a record with no orderable harvest date ranks as. The empty string sorts below every
176
+ #: non-empty one, so undated records fall to the bottom of a descending order without a separate
177
+ #: null-handling clause that a query planner could get wrong.
178
+ _UNRANKED = ""
179
+
180
+ _NORMALIZED_PAGE_SCHEMA = "harness-datagov-v4-normalized-page.v2"
181
+ _NORMALIZED_SEGMENT_SCHEMA = "harness-datagov-v4-normalized-record-segment.v1"
182
+ _OPEN_DIRECTORY = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0)
183
+ _OPEN_MEMBER = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0)
184
+
185
+
186
+ class CatalogAuthoringRefused(SourceContractError):
187
+ """A stable refusal of an authoring coordinate, input, or output boundary."""
188
+
189
+
190
+ @dataclass(frozen=True)
191
+ class AuthoringLimits:
192
+ """Explicit caller-supplied authoring bounds; none of them has a hidden default widening."""
193
+
194
+ max_records: int
195
+ max_input_bytes: int
196
+ max_shard_entries: int = MAX_AUTHORING_SHARD_ENTRIES
197
+ max_shard_bytes: int = MAX_AUTHORING_SHARD_BYTES
198
+ max_disk_bytes: int = 30 * 1024 * 1024 * 1024
199
+ max_wall_seconds: int = 12 * 60 * 60
200
+
201
+ def __post_init__(self) -> None:
202
+ for name in (
203
+ "max_records",
204
+ "max_input_bytes",
205
+ "max_shard_entries",
206
+ "max_shard_bytes",
207
+ "max_disk_bytes",
208
+ "max_wall_seconds",
209
+ ):
210
+ value = getattr(self, name)
211
+ if type(value) is not int or value < 1:
212
+ raise CatalogAuthoringRefused(
213
+ "AUTHOR_LIMIT", f"limits.{name}", "must be a positive integer"
214
+ )
215
+ if (
216
+ self.max_shard_entries > MAX_AUTHORING_SHARD_ENTRIES
217
+ or self.max_shard_bytes > MAX_AUTHORING_SHARD_BYTES
218
+ ):
219
+ raise CatalogAuthoringRefused(
220
+ "AUTHOR_LIMIT",
221
+ "limits.max_shard_entries",
222
+ "shard bounds may not exceed the fixed range contract",
223
+ )
224
+
225
+ def to_dict(self) -> dict[str, int]:
226
+ return {
227
+ "max_records": self.max_records,
228
+ "max_input_bytes": self.max_input_bytes,
229
+ "max_shard_entries": self.max_shard_entries,
230
+ "max_shard_bytes": self.max_shard_bytes,
231
+ "max_disk_bytes": self.max_disk_bytes,
232
+ "max_wall_seconds": self.max_wall_seconds,
233
+ }
234
+
235
+
236
+ @dataclass(frozen=True)
237
+ class AuthoringSelection:
238
+ """Which of a converged generation's records this authoring was asked to author.
239
+
240
+ The default is every one of them, which is what authoring did before a selection could be
241
+ stated and is what it still does when none is. Defaulting to exhaustive is the safe
242
+ direction: a caller who forgets the flag publishes a whole catalogue, never a quiet part of
243
+ one.
244
+
245
+ A bounded selection names a *registered* rule rather than describing one, so that "which
246
+ records are in this catalogue" has an answer a reader can look up instead of a sentence
247
+ somebody wrote into a receipt.
248
+ """
249
+
250
+ rule: str = COVERAGE_EXHAUSTIVE_RULE
251
+ bound: int | None = None
252
+
253
+ def __post_init__(self) -> None:
254
+ if self.rule == COVERAGE_EXHAUSTIVE_RULE:
255
+ if self.bound is not None:
256
+ raise CatalogAuthoringRefused(
257
+ "AUTHOR_COVERAGE",
258
+ "selection.bound",
259
+ "an exhaustive authoring states no bound",
260
+ )
261
+ return
262
+ if self.rule not in SELECTION_RULES:
263
+ raise CatalogAuthoringRefused(
264
+ "AUTHOR_COVERAGE",
265
+ "selection.rule",
266
+ f"a bounded authoring names one registered rule {sorted(SELECTION_RULES)}",
267
+ )
268
+ if type(self.bound) is not int or self.bound < 1:
269
+ raise CatalogAuthoringRefused(
270
+ "AUTHOR_COVERAGE",
271
+ "selection.bound",
272
+ "a bounded authoring states a positive record bound",
273
+ )
274
+
275
+ def coverage(self, *, corpus_records: int, selected_records: int) -> dict[str, Any]:
276
+ """What this selection, applied to a corpus this size, says it covers."""
277
+
278
+ coverage = {
279
+ "rule": self.rule,
280
+ "bound": self.bound,
281
+ "corpus_records": corpus_records,
282
+ "selected_records": selected_records,
283
+ }
284
+ if not coverage_is_valid(coverage):
285
+ raise CatalogAuthoringRefused(
286
+ "AUTHOR_COVERAGE",
287
+ "authoring.coverage",
288
+ "the selected population does not reproduce from the stated rule and bound",
289
+ )
290
+ return coverage
291
+
292
+
293
+ @dataclass(frozen=True)
294
+ class AuthoringResult:
295
+ """One authoring invocation's exact outcome."""
296
+
297
+ status: str
298
+ reason_code: str | None
299
+ manifest_sha256: str | None
300
+ journal_sha256: str | None
301
+ coverage: dict[str, Any]
302
+ receipt: dict[str, Any]
303
+
304
+
305
+ def run_catalog_authoring(
306
+ *,
307
+ staging_root: Path,
308
+ expected_fill_checkpoint_sha256: str,
309
+ policy_id: str,
310
+ output_root: Path,
311
+ limits: AuthoringLimits,
312
+ selection: AuthoringSelection | None = None,
313
+ expected_journal_sha256: str | None = None,
314
+ output_descriptor: int | None = None,
315
+ monotonic: Callable[[], float] = time.monotonic,
316
+ ) -> AuthoringResult:
317
+ """Author one converged v4 generation into bounded shards behind one atomic manifest."""
318
+
319
+ selection = AuthoringSelection() if selection is None else selection
320
+ policy = resolve_authoring_policy(policy_id)
321
+ if not _is_digest(expected_fill_checkpoint_sha256):
322
+ raise CatalogAuthoringRefused(
323
+ "AUTHOR_FILL_CHECKPOINT_MISMATCH",
324
+ "expected_fill_checkpoint_sha256",
325
+ "must be a lowercase SHA-256",
326
+ )
327
+ if expected_journal_sha256 is not None and not _is_digest(expected_journal_sha256):
328
+ raise CatalogAuthoringRefused(
329
+ "AUTHOR_JOURNAL_MISMATCH", "expected_journal_sha256", "must be a lowercase SHA-256"
330
+ )
331
+ deadline = monotonic() + limits.max_wall_seconds
332
+ output = _AuthoringOutput(output_root, root_descriptor=output_descriptor)
333
+ with output.opened():
334
+ journal = _resume_point(
335
+ output, expected_journal_sha256, policy, expected_fill_checkpoint_sha256, selection
336
+ )
337
+ _require_bounds_admit_retained(journal, limits)
338
+ staging = FillStaging(staging_root)
339
+ with staging.locked():
340
+ state = _converged_state(staging, expected_fill_checkpoint_sha256)
341
+ pass_map = validate_pass_map(
342
+ staging, state["last_pass_manifest_sha256"], state["final_map_sha256"]
343
+ )
344
+ scratch_root = _scratch_root(output_root, staging_root)
345
+ with tempfile.TemporaryDirectory(
346
+ prefix="mostlyright-authoring-", dir=scratch_root
347
+ ) as directory:
348
+ index = _RecordIndex(Path(directory) / "records.sqlite3")
349
+ try:
350
+ index.build(staging, state, limits=limits)
351
+ if index.count != pass_map.unique_records:
352
+ raise CatalogAuthoringRefused(
353
+ "AUTHOR_RECORD_COUNT",
354
+ "authoring.records",
355
+ "indexed record population differs from the terminal pass map",
356
+ )
357
+ # The rule is applied to the whole corpus before one record is authored, so
358
+ # what it selects is a property of the corpus rather than of how far this
359
+ # invocation gets. `corpus_records` is the pass map's own count, never the
360
+ # caller's, so a coverage statement cannot overstate what it drew from.
361
+ coverage = selection.coverage(
362
+ corpus_records=pass_map.unique_records,
363
+ selected_records=index.select(selection),
364
+ )
365
+ _require_coverage_matches_retained(journal, coverage)
366
+ _refuse_disk(output, index, limits)
367
+ return _author_generation(
368
+ index,
369
+ output=output,
370
+ policy=policy,
371
+ staging_coordinate=_staging_coordinate(
372
+ state, pass_map, expected_fill_checkpoint_sha256
373
+ ),
374
+ coverage=coverage,
375
+ limits=limits,
376
+ journal=journal,
377
+ deadline=deadline,
378
+ monotonic=monotonic,
379
+ )
380
+ finally:
381
+ index.close()
382
+
383
+
384
+ # ------------------------------------------------------------------------------------------
385
+ # Converged staging
386
+ # ------------------------------------------------------------------------------------------
387
+
388
+
389
+ def _converged_state(staging: FillStaging, expected_checkpoint_sha256: str) -> dict[str, Any]:
390
+ state = staging.load()
391
+ if state is None or staging.state_digest() != expected_checkpoint_sha256:
392
+ raise CatalogAuthoringRefused(
393
+ "AUTHOR_FILL_CHECKPOINT_MISMATCH",
394
+ "fill.staging.state",
395
+ "no live checkpoint matches the caller's exact digest",
396
+ )
397
+ if "v4_checkpoint_schema_version" not in state or not isinstance(
398
+ state.get("evidence_indexes"), dict
399
+ ):
400
+ raise CatalogAuthoringRefused(
401
+ "AUTHOR_FILL_PROTOCOL",
402
+ "fill.staging.state",
403
+ "streaming authoring reads only a segmented v4 generation",
404
+ )
405
+ if (
406
+ state.get("completed") is not True
407
+ or not _is_digest(state.get("final_map_sha256"))
408
+ or not _is_digest(state.get("last_pass_manifest_sha256"))
409
+ ):
410
+ raise CatalogAuthoringRefused(
411
+ "AUTHOR_FILL_INCOMPLETE",
412
+ "fill.staging.state",
413
+ "authoring requires a converged terminal fill generation",
414
+ )
415
+ return state
416
+
417
+
418
+ def _selection_rank(record: Any) -> str:
419
+ """The exact provider text ``last_harvested_date_desc`` orders this record by.
420
+
421
+ A record ranks by whatever the provider stated and nothing derived from it: no parsing, no
422
+ normalising, no reinterpretation. Ordering the provider's own bytes is what makes the rule
423
+ reproducible by anyone holding the staging generation, and for the ISO-8601 instants this
424
+ provider publishes, byte order *is* chronological order.
425
+
426
+ A value this rule cannot order by returns ``_UNRANKED``, which sorts below every value it
427
+ can. Five things are unorderable and all of them mean the same thing -- that the provider
428
+ published no usable harvest date for this record: no such observation at all, an observation
429
+ the harvester had to retain in a typed encoding rather than as canonical JSON, a value that
430
+ is not a string, a string too long to be any spelling of an instant, and a string that does
431
+ not open with a calendar date. None of them is treated as recent, because none of them is
432
+ evidence of recency -- and the last of them would otherwise be treated as the *most* recent
433
+ of all, since bytewise ``"unknown" > "2026-08-09T00:00:00Z"``.
434
+ """
435
+
436
+ observations = record.get("observations") if isinstance(record, dict) else None
437
+ if not isinstance(observations, list):
438
+ return _UNRANKED
439
+ for field in observations:
440
+ if not isinstance(field, dict) or field.get("path") != _RANK_OBSERVATION_PATH:
441
+ continue
442
+ if field.get("encoding", "canonical_json") != "canonical_json":
443
+ return _UNRANKED
444
+ value = field.get("value")
445
+ if (
446
+ isinstance(value, str)
447
+ and 0 < len(value) <= MAX_SELECTION_RANK_CHARS
448
+ and _RANK_DATE_PREFIX.match(value) is not None
449
+ ):
450
+ return value
451
+ return _UNRANKED
452
+ return _UNRANKED
453
+
454
+
455
+ class _RecordIndex:
456
+ """A disk-backed identifier-ordered index over one terminal pass's normalized records."""
457
+
458
+ def __init__(self, path: Path) -> None:
459
+ self.path = path
460
+ self.count = 0
461
+ self.selected = 0
462
+ self.input_bytes = 0
463
+ self._bounded = False
464
+ self._connection = sqlite3.connect(path)
465
+ self._connection.execute("PRAGMA journal_mode=OFF")
466
+ self._connection.execute("PRAGMA synchronous=OFF")
467
+ self._connection.execute("PRAGMA temp_store=FILE")
468
+ self._connection.execute(
469
+ "CREATE TABLE record ("
470
+ "identifier BLOB NOT NULL PRIMARY KEY, semantic TEXT NOT NULL, "
471
+ "page TEXT NOT NULL, observed TEXT NOT NULL, rank TEXT NOT NULL, "
472
+ "payload BLOB NOT NULL"
473
+ ") WITHOUT ROWID"
474
+ )
475
+ self._connection.execute(
476
+ "CREATE TABLE response (raw TEXT NOT NULL PRIMARY KEY, observed TEXT NOT NULL) "
477
+ "WITHOUT ROWID"
478
+ )
479
+
480
+ def close(self) -> None:
481
+ self._connection.close()
482
+
483
+ @property
484
+ def disk_bytes(self) -> int:
485
+ try:
486
+ return self.path.stat().st_size
487
+ except OSError:
488
+ return 0
489
+
490
+ def build(
491
+ self, staging: FillStaging, state: dict[str, Any], *, limits: AuthoringLimits
492
+ ) -> None:
493
+ indexes = state["evidence_indexes"]
494
+ for item in staging.iter_index(indexes["response"], stream="response"):
495
+ raw = item.get("raw_response_sha256")
496
+ observed = item.get("observed_epoch_seconds")
497
+ if _is_digest(raw) and type(observed) is int:
498
+ self._connection.execute(
499
+ "INSERT INTO response(raw, observed) VALUES (?, ?) ON CONFLICT(raw) DO NOTHING",
500
+ (raw, _utc_second(observed)),
501
+ )
502
+ self._connection.commit()
503
+ for item in staging.iter_index(indexes["page"], stream="page"):
504
+ if item.get("pass") != state["pass_number"]:
505
+ continue
506
+ self._read_page(staging, item["normalized_sha256"], limits=limits)
507
+ self._connection.commit()
508
+ self.count = int(self._connection.execute("SELECT count(*) FROM record").fetchone()[0])
509
+
510
+ def _read_page(
511
+ self, staging: FillStaging, page_sha256: str, *, limits: AuthoringLimits
512
+ ) -> None:
513
+ manifest = staging.read_shard("normalized", page_sha256)
514
+ if manifest.get("schema_version") != _NORMALIZED_PAGE_SCHEMA or not isinstance(
515
+ manifest.get("record_segments"), list
516
+ ):
517
+ raise CatalogAuthoringRefused(
518
+ "AUTHOR_STAGING_CORRUPT", "fill.normalized", "normalized page manifest is invalid"
519
+ )
520
+ for segment in manifest["record_segments"]:
521
+ payload = staging.read_shard("normalized", segment["sha256"])
522
+ if payload.get("schema_version") != _NORMALIZED_SEGMENT_SCHEMA or not isinstance(
523
+ payload.get("records"), list
524
+ ):
525
+ raise CatalogAuthoringRefused(
526
+ "AUTHOR_STAGING_CORRUPT",
527
+ "fill.normalized",
528
+ "normalized record segment is invalid",
529
+ )
530
+ for record in payload["records"]:
531
+ self._insert(record, page_sha256=page_sha256, limits=limits)
532
+
533
+ def _insert(self, record: Any, *, page_sha256: str, limits: AuthoringLimits) -> None:
534
+ if (
535
+ not isinstance(record, dict)
536
+ or not isinstance(record.get("record_id"), str)
537
+ or not _is_digest(record.get("normalized_sha256"))
538
+ or not _is_digest(record.get("raw_response_sha256"))
539
+ ):
540
+ raise CatalogAuthoringRefused(
541
+ "AUTHOR_STAGING_CORRUPT", "fill.normalized", "normalized record is invalid"
542
+ )
543
+ try:
544
+ raw = canonical_json_bytes(record)
545
+ except CanonicalJSONError as error:
546
+ raise CatalogAuthoringRefused(
547
+ "AUTHOR_STAGING_CORRUPT", "fill.normalized", "normalized record is not canonical"
548
+ ) from error
549
+ self.input_bytes += len(raw)
550
+ if self.input_bytes > limits.max_input_bytes:
551
+ raise CatalogAuthoringRefused(
552
+ "AUTHOR_INPUT_LIMIT",
553
+ "authoring.input",
554
+ "normalized input exceeds the caller's byte bound",
555
+ )
556
+ row = self._connection.execute(
557
+ "SELECT observed FROM response WHERE raw = ?", (record["raw_response_sha256"],)
558
+ ).fetchone()
559
+ if row is None:
560
+ raise CatalogAuthoringRefused(
561
+ "AUTHOR_STAGING_CORRUPT",
562
+ "fill.normalized",
563
+ "normalized record cites no retained response observation",
564
+ )
565
+ identifier = record["record_id"].encode("utf-8")
566
+ existing = self._connection.execute(
567
+ "SELECT semantic FROM record WHERE identifier = ?", (identifier,)
568
+ ).fetchone()
569
+ if existing is not None:
570
+ if existing[0] != record["normalized_sha256"]:
571
+ raise CatalogAuthoringRefused(
572
+ "AUTHOR_RECORD_CONFLICT",
573
+ "authoring.records",
574
+ "one provider identifier carries two different semantic digests",
575
+ )
576
+ return
577
+ self._connection.execute(
578
+ "INSERT INTO record(identifier, semantic, page, observed, rank, payload) "
579
+ "VALUES (?, ?, ?, ?, ?, ?)",
580
+ (
581
+ identifier,
582
+ record["normalized_sha256"],
583
+ page_sha256,
584
+ row[0],
585
+ _selection_rank(record),
586
+ raw,
587
+ ),
588
+ )
589
+
590
+ def select(self, selection: AuthoringSelection) -> int:
591
+ """Apply one selection rule to the whole indexed corpus and answer what it selected.
592
+
593
+ The selection is a set, computed once, before any record is authored. Authoring still
594
+ walks provider identifiers in ascending order afterwards, because that order is what the
595
+ journal resumes on, what a shard's two range identifiers describe, and what every reader
596
+ of an authored generation merges on. Ranking decides *which* records; the identifier
597
+ decides *when* each one is authored. Confusing the two would make the selection depend
598
+ on where an interrupted run stopped, which is the truncation this whole path exists to
599
+ avoid.
600
+ """
601
+
602
+ if selection.bound is None:
603
+ self.selected = self.count
604
+ return self.selected
605
+ # The ordering index is what keeps the selection from re-sorting the whole corpus on
606
+ # every read; its pages live inside the index file, which `_refuse_disk` measures. The
607
+ # sort that *builds* it spills to SQLite's temp directory under `temp_store=FILE`, which
608
+ # is transient and outside that measure -- so the bound's disk cost is accounted where it
609
+ # is durable and merely bounded by the corpus where it is not. A bound is the only thing
610
+ # that needs an order other than the primary key, so an exhaustive authoring pays neither.
611
+ self._connection.execute("CREATE INDEX record_rank ON record(rank DESC, identifier ASC)")
612
+ self._connection.execute(
613
+ "CREATE TABLE selected (identifier BLOB NOT NULL PRIMARY KEY) WITHOUT ROWID"
614
+ )
615
+ self._connection.execute(
616
+ "INSERT INTO selected(identifier) "
617
+ "SELECT identifier FROM record ORDER BY rank DESC, identifier ASC LIMIT ?",
618
+ (selection.bound,),
619
+ )
620
+ self._connection.commit()
621
+ self._bounded = True
622
+ self.selected = int(self._connection.execute("SELECT count(*) FROM selected").fetchone()[0])
623
+ return self.selected
624
+
625
+ def stream(self, after: str | None) -> Iterator[tuple[str, str, str, bytes]]:
626
+ source = (
627
+ "record JOIN selected ON selected.identifier = record.identifier"
628
+ if self._bounded
629
+ else "record"
630
+ )
631
+ columns = "record.identifier, record.page, record.observed, record.payload"
632
+ if after is None:
633
+ cursor = self._connection.execute(
634
+ f"SELECT {columns} FROM {source} ORDER BY record.identifier"
635
+ )
636
+ else:
637
+ cursor = self._connection.execute(
638
+ f"SELECT {columns} FROM {source} "
639
+ "WHERE record.identifier > ? ORDER BY record.identifier",
640
+ (after.encode("utf-8"),),
641
+ )
642
+ while rows := cursor.fetchmany(64):
643
+ for identifier, page, observed, payload in rows:
644
+ yield identifier.decode("utf-8"), page, observed, payload
645
+
646
+
647
+ # ------------------------------------------------------------------------------------------
648
+ # Authoring
649
+ # ------------------------------------------------------------------------------------------
650
+
651
+
652
+ def _author_generation(
653
+ index: _RecordIndex,
654
+ *,
655
+ output: _AuthoringOutput,
656
+ policy: AuthoringPolicy,
657
+ staging_coordinate: dict[str, Any],
658
+ coverage: dict[str, Any],
659
+ limits: AuthoringLimits,
660
+ journal: dict[str, Any],
661
+ deadline: float,
662
+ monotonic: Callable[[], float],
663
+ ) -> AuthoringResult:
664
+ shards: list[dict[str, Any]] = list(journal["shards"])
665
+ counts = dict(journal["counts"])
666
+ digests = dict(journal["layer_text_digests"])
667
+ last_id: str | None = journal["last_provider_record_id"]
668
+ buffered: list[dict[str, Any]] = []
669
+ buffered_bytes = 0
670
+ stop: str | None = None
671
+
672
+ def flush() -> None:
673
+ nonlocal buffered, buffered_bytes
674
+ if not buffered:
675
+ return
676
+ payload = {
677
+ "schema_version": AUTHORING_SHARD_SCHEMA,
678
+ "policy_sha256": policy.digest,
679
+ "first_provider_record_id": buffered[0]["provider_record_id"],
680
+ "last_provider_record_id": buffered[-1]["provider_record_id"],
681
+ "items": buffered,
682
+ }
683
+ digest, size = output.write_shard(payload, maximum=limits.max_shard_bytes)
684
+ shards.append(
685
+ {
686
+ "sha256": digest,
687
+ "bytes": size,
688
+ "item_count": len(buffered),
689
+ "entry_count": sum(1 for item in buffered if item["entry"] is not None),
690
+ "first_provider_record_id": payload["first_provider_record_id"],
691
+ "last_provider_record_id": payload["last_provider_record_id"],
692
+ }
693
+ )
694
+ buffered = []
695
+ buffered_bytes = 0
696
+ _refuse_disk(output, index, limits)
697
+
698
+ for identifier, page_sha256, observed_at, payload in index.stream(last_id):
699
+ if counts["records"] >= limits.max_records:
700
+ stop = "AUTHOR_RECORD_LIMIT"
701
+ break
702
+ if monotonic() >= deadline:
703
+ stop = "AUTHOR_WALL_LIMIT"
704
+ break
705
+ authored = author_catalog_record(
706
+ parse_canonical_json(payload),
707
+ policy=policy,
708
+ observed_at=observed_at,
709
+ page_evidence_sha256=page_sha256,
710
+ )
711
+ item = authored.to_dict()
712
+ item_bytes = len(canonical_json_bytes(item))
713
+ if buffered and (
714
+ len(buffered) >= limits.max_shard_entries
715
+ or buffered_bytes + item_bytes + len(buffered) + _SHARD_ENVELOPE_BYTES
716
+ > limits.max_shard_bytes
717
+ ):
718
+ flush()
719
+ buffered.append(item)
720
+ buffered_bytes += item_bytes
721
+ counts["records"] += 1
722
+ counts[authored.disposition] += 1
723
+ counts["layer_text_truncated"] += len(authored.truncated_layers)
724
+ if authored.entry is not None:
725
+ for layer, text in authored.layer_texts:
726
+ digests[layer] = _layer_chain(digests[layer], authored.entry.entry_id, layer, text)
727
+ last_id = identifier
728
+ flush()
729
+
730
+ # The journal carries the coverage of the generation it continues, so a resume that restated
731
+ # the rule or the bound is refused against retained evidence rather than against an argument
732
+ # this process happens to hold.
733
+ journal_sha256 = output.publish(
734
+ JOURNAL_FILENAME,
735
+ {
736
+ "schema_version": AUTHORING_JOURNAL_SCHEMA,
737
+ "policy_sha256": policy.digest,
738
+ "staging_checkpoint_sha256": staging_coordinate["checkpoint_sha256"],
739
+ "coverage": coverage,
740
+ "shards": shards,
741
+ "counts": counts,
742
+ "layer_text_digests": digests,
743
+ "last_provider_record_id": last_id,
744
+ "reason_code": stop,
745
+ },
746
+ maximum=MAX_AUTHORING_JOURNAL_BYTES,
747
+ )
748
+ _refuse_disk(output, index, limits)
749
+ # The manifest is written only when the stream ran out, which under a selection means the
750
+ # selected population was authored in full: `counts["records"]` and
751
+ # `coverage["selected_records"]` cannot differ here without one of them having been forged
752
+ # after the fact. That is a reader's question rather than a writer's, and every stage that
753
+ # reads an authored manifest asks it -- see `admit_authored_generation` and the publisher's
754
+ # receipt reconciliation, both of which hold a document they did not write.
755
+ manifest_sha256: str | None = None
756
+ if stop is None:
757
+ manifest_sha256 = output.publish(
758
+ MANIFEST_FILENAME,
759
+ _manifest(policy, staging_coordinate, coverage, counts, digests, shards),
760
+ maximum=MAX_AUTHORING_MANIFEST_BYTES,
761
+ )
762
+ return AuthoringResult(
763
+ status="complete" if stop is None else "incomplete",
764
+ reason_code=stop,
765
+ manifest_sha256=manifest_sha256,
766
+ journal_sha256=journal_sha256,
767
+ coverage=coverage,
768
+ receipt={
769
+ "schema_version": AUTHORING_RECEIPT_SCHEMA,
770
+ "status": "complete" if stop is None else "incomplete",
771
+ "reason_code": stop,
772
+ "policy_id": policy.policy_id,
773
+ "policy_sha256": policy.digest,
774
+ "provider_id": policy.provider_id,
775
+ "harvester_coordinate": policy.harvester_coordinate,
776
+ "staging": staging_coordinate,
777
+ "coverage": coverage,
778
+ "counts": counts,
779
+ "layer_text_digests": digests,
780
+ "shard_count": len(shards),
781
+ "shards": shards,
782
+ "limits": limits.to_dict(),
783
+ "manifest_sha256": manifest_sha256,
784
+ "journal_sha256": journal_sha256,
785
+ "output_bytes": output.bytes_written,
786
+ },
787
+ )
788
+
789
+
790
+ def _manifest(
791
+ policy: AuthoringPolicy,
792
+ staging: dict[str, Any],
793
+ coverage: dict[str, Any],
794
+ counts: dict[str, int],
795
+ digests: dict[str, str | None],
796
+ shards: list[dict[str, Any]],
797
+ ) -> dict[str, Any]:
798
+ body = {
799
+ "schema_version": AUTHORING_MANIFEST_SCHEMA,
800
+ "policy_id": policy.policy_id,
801
+ "policy_sha256": policy.digest,
802
+ "staging": staging,
803
+ "coverage": coverage,
804
+ "counts": counts,
805
+ "layer_text_digests": digests,
806
+ "shards": shards,
807
+ }
808
+ return {**body, "root_sha256": canonical_sha256(body)}
809
+
810
+
811
+ def _staging_coordinate(state: dict[str, Any], pass_map, checkpoint_sha256: str) -> dict[str, Any]:
812
+ indexes = state["evidence_indexes"]
813
+ return {
814
+ "checkpoint_sha256": checkpoint_sha256,
815
+ "config_sha256": state["config_sha256"],
816
+ "endpoint": state["endpoint"],
817
+ "harvester_coordinate": state["harvester_coordinate"],
818
+ "pass_number": state["pass_number"],
819
+ "pass_map_sha256": pass_map.manifest_sha256,
820
+ "pass_map_root_sha256": pass_map.root_sha256,
821
+ "unique_records": pass_map.unique_records,
822
+ "raw_response_index_root_sha256": indexes["response"]["root_sha256"],
823
+ "raw_response_index_count": indexes["response"]["count"],
824
+ "normalized_page_index_root_sha256": indexes["page"]["root_sha256"],
825
+ "normalized_page_index_count": indexes["page"]["count"],
826
+ "record_index_root_sha256": indexes["record"]["root_sha256"],
827
+ "record_index_count": indexes["record"]["count"],
828
+ }
829
+
830
+
831
+ def _layer_chain(previous: str | None, entry_id: str, layer: str, text: str) -> str:
832
+ return canonical_sha256(
833
+ {
834
+ "previous_sha256": previous,
835
+ "entry_id": entry_id,
836
+ "layer": layer,
837
+ "text_sha256": sha256_bytes(text.encode("utf-8")),
838
+ }
839
+ )
840
+
841
+
842
+ def _refuse_disk(output: _AuthoringOutput, index: _RecordIndex, limits: AuthoringLimits) -> None:
843
+ if output.bytes_written + index.disk_bytes > limits.max_disk_bytes:
844
+ raise CatalogAuthoringRefused(
845
+ "AUTHOR_DISK_LIMIT", "authoring.output", "authoring exceeds its local disk bound"
846
+ )
847
+
848
+
849
+ def _require_bounds_admit_retained(journal: dict[str, Any], limits: AuthoringLimits) -> None:
850
+ """A resume may not narrow a bound the generation it continues has already passed.
851
+
852
+ The receipt an authoring stage writes states the bounds of the invocation that *finished* the
853
+ generation, and a publication reconciles those bounds against every shard in the manifest --
854
+ including the ones earlier invocations wrote. Narrowing a bound mid-generation would produce
855
+ a complete generation that no publication can accept, and it would say so hours later, at the
856
+ publisher, about work that is already durable. It is refused here instead, where the
857
+ narrowing happened and before any further record is authored.
858
+
859
+ A fresh authoring resumes an empty journal, so nothing here constrains a first invocation.
860
+ """
861
+
862
+ shards = journal["shards"]
863
+ if (
864
+ limits.max_records < journal["counts"]["records"]
865
+ or any(limits.max_shard_entries < descriptor["item_count"] for descriptor in shards)
866
+ or any(limits.max_shard_bytes < descriptor["bytes"] for descriptor in shards)
867
+ ):
868
+ raise CatalogAuthoringRefused(
869
+ "AUTHOR_LIMIT",
870
+ "limits",
871
+ "a resume may not narrow a bound the retained generation has already passed",
872
+ )
873
+
874
+
875
+ def _require_coverage_matches_retained(journal: dict[str, Any], coverage: dict[str, Any]) -> None:
876
+ """A resume continues one coverage claim; it never restates it.
877
+
878
+ ``_require_bounds_admit_retained`` lets a resume widen a *bound* because a bound is a
879
+ permission to spend and a wider one strands nothing. Coverage is not that. It is the claim
880
+ the generation makes about which records it holds, and shards written under one claim plus
881
+ shards written under another are a generation that describes neither. So this is equality,
882
+ in both directions, and it is checked against the retained journal rather than against a
883
+ caller's argument.
884
+
885
+ A fresh authoring resumes an empty journal, so nothing here constrains a first invocation.
886
+ """
887
+
888
+ retained = journal["coverage"]
889
+ if retained is not None and retained != coverage:
890
+ raise CatalogAuthoringRefused(
891
+ "AUTHOR_COVERAGE_MISMATCH",
892
+ "authoring.coverage",
893
+ "the retained partial generation was authored under another stated coverage",
894
+ )
895
+
896
+
897
+ def _empty_journal() -> dict[str, Any]:
898
+ return {
899
+ "coverage": None,
900
+ "shards": [],
901
+ "counts": {
902
+ "records": 0,
903
+ "authored": 0,
904
+ "flagged": 0,
905
+ "skipped": 0,
906
+ "failed": 0,
907
+ "layer_text_truncated": 0,
908
+ },
909
+ "layer_text_digests": dict.fromkeys(EMBEDDING_LAYERS),
910
+ "last_provider_record_id": None,
911
+ }
912
+
913
+
914
+ def _resume_point(
915
+ output: _AuthoringOutput,
916
+ expected_journal_sha256: str | None,
917
+ policy: AuthoringPolicy,
918
+ expected_fill_checkpoint_sha256: str,
919
+ selection: AuthoringSelection,
920
+ ) -> dict[str, Any]:
921
+ manifest = output.read(MANIFEST_FILENAME, maximum=MAX_AUTHORING_MANIFEST_BYTES)
922
+ if manifest is not None:
923
+ raise CatalogAuthoringRefused(
924
+ "AUTHOR_OUTPUT_EXISTS",
925
+ "authoring.output",
926
+ "a published generation is immutable; author into a new output root",
927
+ )
928
+ raw = output.read(JOURNAL_FILENAME, maximum=MAX_AUTHORING_JOURNAL_BYTES)
929
+ if expected_journal_sha256 is None:
930
+ if raw is not None:
931
+ raise CatalogAuthoringRefused(
932
+ "AUTHOR_OUTPUT_EXISTS",
933
+ "authoring.output",
934
+ "a partial generation exists; resume it with its exact journal digest",
935
+ )
936
+ return _empty_journal()
937
+ if raw is None or sha256_bytes(raw) != expected_journal_sha256:
938
+ raise CatalogAuthoringRefused(
939
+ "AUTHOR_JOURNAL_MISMATCH",
940
+ "authoring.journal",
941
+ "no retained journal matches the caller's exact digest",
942
+ )
943
+ try:
944
+ journal = parse_canonical_json(raw)
945
+ except CanonicalJSONError as error:
946
+ raise CatalogAuthoringRefused(
947
+ "AUTHOR_JOURNAL_CORRUPT", "authoring.journal", "journal is not canonical"
948
+ ) from error
949
+ if (
950
+ not isinstance(journal, dict)
951
+ or journal.get("schema_version") != AUTHORING_JOURNAL_SCHEMA
952
+ or not isinstance(journal.get("shards"), list)
953
+ or not isinstance(journal.get("counts"), dict)
954
+ or not isinstance(journal.get("layer_text_digests"), dict)
955
+ or set(journal["layer_text_digests"]) != set(EMBEDDING_LAYERS)
956
+ or set(journal["counts"]) != set(_empty_journal()["counts"])
957
+ or not isinstance(journal.get("last_provider_record_id"), (str, type(None)))
958
+ or not _is_digest(journal.get("policy_sha256"))
959
+ or not _is_digest(journal.get("staging_checkpoint_sha256"))
960
+ or not coverage_is_valid(journal.get("coverage"))
961
+ ):
962
+ raise CatalogAuthoringRefused(
963
+ "AUTHOR_JOURNAL_CORRUPT", "authoring.journal", "journal contract differs"
964
+ )
965
+ if (
966
+ journal["policy_sha256"] != policy.digest
967
+ or journal["staging_checkpoint_sha256"] != expected_fill_checkpoint_sha256
968
+ ):
969
+ raise CatalogAuthoringRefused(
970
+ "AUTHOR_JOURNAL_COORDINATE",
971
+ "authoring.journal",
972
+ "the retained partial generation was authored from another policy or checkpoint",
973
+ )
974
+ # The rule and the bound are the caller's half of the claim and are answerable here, before
975
+ # the staging root is opened or one record is re-read. The corpus and selected populations
976
+ # are the generation's half; they are recomputed from immutable staging and checked against
977
+ # this same journal once the index exists.
978
+ if (journal["coverage"]["rule"], journal["coverage"]["bound"]) != (
979
+ selection.rule,
980
+ selection.bound,
981
+ ):
982
+ raise CatalogAuthoringRefused(
983
+ "AUTHOR_COVERAGE_MISMATCH",
984
+ "authoring.journal.coverage",
985
+ "the retained partial generation was authored under another stated coverage",
986
+ )
987
+ for descriptor in journal["shards"]:
988
+ output.verify_shard(descriptor)
989
+ return journal
990
+
991
+
992
+ def _scratch_root(output_root: Path, staging_root: Path) -> str | None:
993
+ for candidate in (Path(output_root).parent, Path(staging_root).parent):
994
+ if candidate.is_dir():
995
+ return str(candidate)
996
+ return None
997
+
998
+
999
+ def _utc_second(epoch_seconds: int) -> str:
1000
+ try:
1001
+ return datetime.fromtimestamp(epoch_seconds, UTC).strftime("%Y-%m-%dT%H:%M:%SZ")
1002
+ except (OverflowError, OSError, ValueError) as error:
1003
+ raise CatalogAuthoringRefused(
1004
+ "AUTHOR_STAGING_CORRUPT",
1005
+ "fill.staging.response",
1006
+ "retained response observation is outside the representable range",
1007
+ ) from error
1008
+
1009
+
1010
+ def _is_digest(value: Any) -> bool:
1011
+ return (
1012
+ isinstance(value, str) and len(value) == 64 and all(c in "0123456789abcdef" for c in value)
1013
+ )
1014
+
1015
+
1016
+ # ------------------------------------------------------------------------------------------
1017
+ # Durable, descriptor-rooted, no-follow output
1018
+ # ------------------------------------------------------------------------------------------
1019
+
1020
+
1021
+ class _AuthoringOutput:
1022
+ """One locally confined output root whose members install atomically and read back exactly."""
1023
+
1024
+ def __init__(self, root: Path, *, root_descriptor: int | None = None) -> None:
1025
+ self.root = Path(root)
1026
+ self._retained_root_descriptor = root_descriptor
1027
+ self.bytes_written = 0
1028
+ self._root_fd = -1
1029
+ self._shards_fd = -1
1030
+ self._owned: list[int] = []
1031
+
1032
+ @contextlib.contextmanager
1033
+ def opened(self) -> Iterator[None]:
1034
+ try:
1035
+ if self._retained_root_descriptor is None:
1036
+ self.root.mkdir(parents=True, exist_ok=True, mode=0o700)
1037
+ self._root_fd = self._open_directory(self.root)
1038
+ (self.root / SHARDS_DIRNAME).mkdir(exist_ok=True, mode=0o700)
1039
+ self._shards_fd = self._open_directory(self.root / SHARDS_DIRNAME)
1040
+ else:
1041
+ self._root_fd = os.dup(self._retained_root_descriptor)
1042
+ self._owned.append(self._root_fd)
1043
+ self._require_private_directory(self._root_fd)
1044
+ try:
1045
+ os.mkdir(SHARDS_DIRNAME, mode=0o700, dir_fd=self._root_fd)
1046
+ os.fsync(self._root_fd)
1047
+ except FileExistsError:
1048
+ pass
1049
+ self._shards_fd = os.open(SHARDS_DIRNAME, _OPEN_DIRECTORY, dir_fd=self._root_fd)
1050
+ self._owned.append(self._shards_fd)
1051
+ self._require_private_directory(self._shards_fd)
1052
+ except CatalogAuthoringRefused:
1053
+ self._release()
1054
+ raise
1055
+ except OSError as error:
1056
+ self._release()
1057
+ raise CatalogAuthoringRefused(
1058
+ "AUTHOR_OUTPUT_IO", "authoring.output", "authoring output root is unusable"
1059
+ ) from error
1060
+ try:
1061
+ yield
1062
+ finally:
1063
+ self._release()
1064
+
1065
+ def _release(self) -> None:
1066
+ for descriptor in reversed(self._owned):
1067
+ os.close(descriptor)
1068
+ self._owned = []
1069
+ self._root_fd = -1
1070
+ self._shards_fd = -1
1071
+
1072
+ def _open_directory(self, path: Path) -> int:
1073
+ descriptor = os.open(path, _OPEN_DIRECTORY)
1074
+ self._owned.append(descriptor)
1075
+ self._require_private_directory(descriptor)
1076
+ return descriptor
1077
+
1078
+ @staticmethod
1079
+ def _require_private_directory(descriptor: int) -> None:
1080
+ info = os.fstat(descriptor)
1081
+ if not stat.S_ISDIR(info.st_mode) or stat.S_IMODE(info.st_mode) & 0o077:
1082
+ raise CatalogAuthoringRefused(
1083
+ "AUTHOR_OUTPUT_PATH",
1084
+ "authoring.output",
1085
+ "authoring output components must be private directories",
1086
+ )
1087
+
1088
+ def read(self, name: str, *, maximum: int) -> bytes | None:
1089
+ try:
1090
+ return _read_regular_at(self._root_fd, name, maximum=maximum)
1091
+ except FileNotFoundError:
1092
+ return None
1093
+
1094
+ def write_shard(self, payload: dict[str, Any], *, maximum: int) -> tuple[str, int]:
1095
+ raw = canonical_json_bytes(payload)
1096
+ if len(raw) > maximum:
1097
+ raise CatalogAuthoringRefused(
1098
+ "AUTHOR_SHARD_LIMIT", "authoring.shard", "shard exceeds its byte bound"
1099
+ )
1100
+ digest = sha256_bytes(raw)
1101
+ self._install_immutable(f"{digest}.json", raw)
1102
+ return digest, len(raw)
1103
+
1104
+ def verify_shard(self, descriptor: Any) -> None:
1105
+ if not shard_descriptor_is_valid(descriptor):
1106
+ raise CatalogAuthoringRefused(
1107
+ "AUTHOR_JOURNAL_CORRUPT", "authoring.journal", "shard descriptor is invalid"
1108
+ )
1109
+ try:
1110
+ raw = _read_regular_at(
1111
+ self._shards_fd, f"{descriptor['sha256']}.json", maximum=descriptor["bytes"]
1112
+ )
1113
+ except OSError as error:
1114
+ raise CatalogAuthoringRefused(
1115
+ "AUTHOR_OUTPUT_READBACK",
1116
+ "authoring.shard",
1117
+ "a committed shard is absent or unreadable",
1118
+ ) from error
1119
+ if len(raw) != descriptor["bytes"] or sha256_bytes(raw) != descriptor["sha256"]:
1120
+ raise CatalogAuthoringRefused(
1121
+ "AUTHOR_OUTPUT_READBACK", "authoring.shard", "a committed shard differs"
1122
+ )
1123
+
1124
+ def _install_immutable(self, name: str, raw: bytes) -> None:
1125
+ try:
1126
+ existing = _read_regular_at(self._shards_fd, name, maximum=len(raw))
1127
+ except FileNotFoundError:
1128
+ existing = None
1129
+ except OSError as error:
1130
+ raise CatalogAuthoringRefused(
1131
+ "AUTHOR_OUTPUT_IO", "authoring.shard", "immutable shard lookup failed"
1132
+ ) from error
1133
+ if existing is not None:
1134
+ if existing != raw:
1135
+ raise CatalogAuthoringRefused(
1136
+ "AUTHOR_OUTPUT_READBACK", "authoring.shard", "digest path bytes differ"
1137
+ )
1138
+ return
1139
+ temporary = f".publish.{os.getpid()}.{secrets.token_hex(16)}.tmp"
1140
+ try:
1141
+ _write_regular_at(self._shards_fd, temporary, raw)
1142
+ with contextlib.suppress(FileExistsError):
1143
+ os.link(
1144
+ temporary,
1145
+ name,
1146
+ src_dir_fd=self._shards_fd,
1147
+ dst_dir_fd=self._shards_fd,
1148
+ follow_symlinks=False,
1149
+ )
1150
+ except OSError as error:
1151
+ raise CatalogAuthoringRefused(
1152
+ "AUTHOR_OUTPUT_IO", "authoring.shard", "immutable shard publication failed"
1153
+ ) from error
1154
+ finally:
1155
+ with contextlib.suppress(FileNotFoundError):
1156
+ os.unlink(temporary, dir_fd=self._shards_fd)
1157
+ os.fsync(self._shards_fd)
1158
+ if _read_regular_at(self._shards_fd, name, maximum=len(raw)) != raw:
1159
+ raise CatalogAuthoringRefused(
1160
+ "AUTHOR_OUTPUT_READBACK", "authoring.shard", "immutable shard readback differs"
1161
+ )
1162
+ self.bytes_written += len(raw)
1163
+
1164
+ def publish(self, name: str, payload: dict[str, Any], *, maximum: int) -> str:
1165
+ raw = canonical_json_bytes(payload)
1166
+ if len(raw) > maximum:
1167
+ raise CatalogAuthoringRefused(
1168
+ "AUTHOR_SHARD_LIMIT", f"authoring.{name}", "document exceeds its byte bound"
1169
+ )
1170
+ temporary = f".{name}.{os.getpid()}.{secrets.token_hex(8)}.tmp"
1171
+ try:
1172
+ _write_regular_at(self._root_fd, temporary, raw)
1173
+ os.replace(temporary, name, src_dir_fd=self._root_fd, dst_dir_fd=self._root_fd)
1174
+ os.fsync(self._root_fd)
1175
+ except OSError as error:
1176
+ raise CatalogAuthoringRefused(
1177
+ "AUTHOR_OUTPUT_IO", f"authoring.{name}", "document publication failed"
1178
+ ) from error
1179
+ finally:
1180
+ with contextlib.suppress(FileNotFoundError):
1181
+ os.unlink(temporary, dir_fd=self._root_fd)
1182
+ if _read_regular_at(self._root_fd, name, maximum=maximum) != raw:
1183
+ raise CatalogAuthoringRefused(
1184
+ "AUTHOR_OUTPUT_READBACK", f"authoring.{name}", "document readback differs"
1185
+ )
1186
+ self.bytes_written += len(raw)
1187
+ return sha256_bytes(raw)
1188
+
1189
+
1190
+ def _read_regular_at(parent_fd: int, name: str, *, maximum: int) -> bytes:
1191
+ try:
1192
+ raw = read_bounded_at(parent_fd, name, maximum=maximum)
1193
+ except BoundedReadFailure as error:
1194
+ raise CatalogAuthoringRefused(
1195
+ "AUTHOR_OUTPUT_PATH",
1196
+ "authoring.output",
1197
+ "output members must be stable bounded single-link regular files",
1198
+ ) from error
1199
+ assert raw is not None
1200
+ return raw
1201
+
1202
+
1203
+ def _write_regular_at(parent_fd: int, name: str, raw: bytes) -> None:
1204
+ flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0)
1205
+ descriptor = os.open(name, flags, 0o600, dir_fd=parent_fd)
1206
+ try:
1207
+ written = 0
1208
+ while written < len(raw):
1209
+ count = os.write(descriptor, raw[written:])
1210
+ if count < 1:
1211
+ raise CatalogAuthoringRefused(
1212
+ "AUTHOR_OUTPUT_IO", "authoring.output", "write made no progress"
1213
+ )
1214
+ written += count
1215
+ os.fsync(descriptor)
1216
+ finally:
1217
+ os.close(descriptor)