mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,1097 @@
1
+ """Exact streaming delta classification over sorted authoring shards.
2
+
3
+ The v1 classifier in :mod:`delta` reads one whole generation into dictionaries, and it explicitly
4
+ requires the caller to hand it a complete provider-record-to-entry map. At 548k provider records
5
+ neither is possible: the map alone exceeds what ``canonical`` will serialize, and the entries alone
6
+ exceed the plan's memory budget. This module replaces that path for provider-scale sweeps while
7
+ :mod:`delta` remains exactly as it is for bounded v1 fills.
8
+
9
+ The shape of the work is a merge, not a join.
10
+
11
+ *Two sorted streams, one record of lookahead.* Authoring shards and identity packs use ascending
12
+ UTF-8 byte order of the exact provider identifier. Classifying
13
+ one sweep is therefore a linear two-stream merge with one record of lookahead per stream: a current
14
+ record with no predecessor is new, a predecessor with no current record is absent, and a matched
15
+ pair is unchanged or changed depending on one digest comparison. Nothing is indexed, sorted, or
16
+ held beyond one input shard, one history pack, and bounded output buffers.
17
+
18
+ *One digest decides.* The comparison is the semantic classification digest: the entry's facts,
19
+ their provider provenance, and its identity -- deliberately without the managed version link, the
20
+ append-only observations, or the sealed vector instances. The semantic digest excludes
21
+ ``last_harvested_date`` and ``popularity`` from the provider record's semantic digest. A nightly
22
+ re-harvest that rewrites only those values reaches this comparison as *equal*: the sweep appends one
23
+ observation, reuses the entry and its member references, and schedules no encode at all.
24
+
25
+ *Every provider record lands somewhere.* A record is new, changed, unchanged, flagged, skipped, or
26
+ failed -- and the six counts sum to exactly the provider record measure, which is checked rather
27
+ than asserted in prose. ``requests``, ``responses``, ``pages``, and provider records stay four
28
+ separate numbers, because conflating them is how a sweep starts reporting page counts as coverage.
29
+ A skipped record keeps its authoring reason code in the emitted shard and a durable disposition in
30
+ its identity history; it never quietly leaves the denominator.
31
+
32
+ *A delta is complete or it is nothing.* Output shards are content-addressed and immutable, the
33
+ successor identity generation is published before the delta manifest, and the delta manifest is
34
+ published last. A bound reached mid-classification refuses rather than publishing a partial
35
+ manifest, so no consumer can ever read a truncated classification as a whole one.
36
+ """
37
+
38
+ from __future__ import annotations
39
+
40
+ import contextlib
41
+ import os
42
+ import time
43
+ from collections.abc import Callable, Iterator
44
+ from dataclasses import dataclass
45
+ from pathlib import Path
46
+ from typing import Any
47
+
48
+ from mostlyright.data_harness.canonical import (
49
+ CanonicalJSONError,
50
+ canonical_json_bytes,
51
+ canonical_sha256,
52
+ parse_canonical_json,
53
+ sha256_bytes,
54
+ )
55
+ from mostlyright.data_harness.sources.catalog.authoring_policy import (
56
+ AUTHORING_DISPOSITIONS,
57
+ AuthoringPolicy,
58
+ resolve_authoring_policy,
59
+ )
60
+ from mostlyright.data_harness.sources.catalog.authoring_shards import (
61
+ AUTHORING_MANIFEST_SCHEMA,
62
+ AUTHORING_SHARD_SCHEMA,
63
+ MANIFEST_FILENAME,
64
+ MAX_AUTHORING_MANIFEST_BYTES,
65
+ MAX_AUTHORING_SHARD_ENTRIES,
66
+ SHARDS_DIRNAME,
67
+ )
68
+ from mostlyright.data_harness.sources.catalog.bounded_io import (
69
+ BoundedReadFailure,
70
+ read_bounded_path,
71
+ )
72
+ from mostlyright.data_harness.sources.catalog.coverage import coverage_is_valid
73
+ from mostlyright.data_harness.sources.catalog.entry_v2 import (
74
+ CatalogEntryV2,
75
+ catalog_entry_v2_from_dict,
76
+ )
77
+ from mostlyright.data_harness.sources.catalog.generation_receipt import retained_child_directory
78
+ from mostlyright.data_harness.sources.catalog.identity_history import (
79
+ MAX_IDENTITY_PACK_RECORDS,
80
+ DurableDirectory,
81
+ IdentityHistory,
82
+ IdentityHistoryStore,
83
+ IdentityHistoryWriter,
84
+ ObservationEvidence,
85
+ ProviderIdentity,
86
+ advance_identity_history,
87
+ derive_entry_id,
88
+ observe_identity_absence,
89
+ record_identity_disposition,
90
+ )
91
+ from mostlyright.data_harness.sources.contracts import SourceContractError
92
+
93
+ DELTA_SHARD_SCHEMA = "harness-catalog-delta-shard.v1"
94
+ DELTA_MANIFEST_SCHEMA = "harness-catalog-delta-manifest.v1"
95
+ DELTA_RECEIPT_SCHEMA = "mr-data-catalog-delta.v2"
96
+
97
+ DELTA_MANIFEST_FILENAME = "delta-manifest.json"
98
+ DELTA_SHARDS_DIRNAME = "shards"
99
+ DELTA_IDENTITY_DIRNAME = "identity"
100
+
101
+ # The complete closed partition of one sweep. The first six describe current provider records and
102
+ # sum to the provider record measure; ``absent`` describes predecessor records instead.
103
+ DELTA_CLASSIFICATIONS = (
104
+ "new",
105
+ "changed",
106
+ "unchanged",
107
+ "absent",
108
+ "flagged",
109
+ "skipped",
110
+ "failed",
111
+ )
112
+ _CURRENT_CLASSIFICATIONS = ("new", "changed", "unchanged", "flagged", "skipped", "failed")
113
+
114
+ MAX_DELTA_SHARD_ENTRIES = 1_000
115
+ MAX_DELTA_SHARD_BYTES = 8 * 1024 * 1024
116
+ MAX_DELTA_MANIFEST_BYTES = 8 * 1024 * 1024
117
+ MAX_DELTA_INPUT_SHARD_BYTES = 8 * 1024 * 1024
118
+
119
+ DELTA_ABSENCE_REASON = "DELTA_ABSENT"
120
+
121
+ _OPEN_MEMBER = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0)
122
+
123
+
124
+ class CatalogDeltaStreamRefused(SourceContractError):
125
+ """A stable refusal of a delta input, coordinate, bound, or durable output."""
126
+
127
+
128
+ @dataclass(frozen=True)
129
+ class SweepMeasures:
130
+ """The four distinct measures of one sweep; none of them stands in for another."""
131
+
132
+ requests: int
133
+ responses: int
134
+ pages: int
135
+ provider_records: int
136
+
137
+ def __post_init__(self) -> None:
138
+ for name in ("requests", "responses", "pages", "provider_records"):
139
+ value = getattr(self, name)
140
+ if type(value) is not int or value < 0:
141
+ raise CatalogDeltaStreamRefused(
142
+ "DELTA_MEASURE_CONTRACT",
143
+ f"measures.{name}",
144
+ "must be a non-negative integer",
145
+ )
146
+ if self.responses > self.requests:
147
+ raise CatalogDeltaStreamRefused(
148
+ "DELTA_MEASURE_CONTRACT",
149
+ "measures.responses",
150
+ "a sweep cannot retain more responses than it made requests",
151
+ )
152
+ if self.pages > self.responses:
153
+ raise CatalogDeltaStreamRefused(
154
+ "DELTA_MEASURE_CONTRACT",
155
+ "measures.pages",
156
+ "a sweep cannot normalize more pages than it retained responses",
157
+ )
158
+
159
+ def to_dict(self) -> dict[str, int]:
160
+ return {
161
+ "requests": self.requests,
162
+ "responses": self.responses,
163
+ "pages": self.pages,
164
+ "provider_records": self.provider_records,
165
+ }
166
+
167
+
168
+ @dataclass(frozen=True)
169
+ class StreamingDeltaLimits:
170
+ """Explicit caller-supplied classification bounds; a bound reached is a refusal."""
171
+
172
+ max_records: int = 2_000_000
173
+ max_shard_entries: int = MAX_DELTA_SHARD_ENTRIES
174
+ max_shard_bytes: int = MAX_DELTA_SHARD_BYTES
175
+ max_pack_records: int = MAX_IDENTITY_PACK_RECORDS
176
+ max_disk_bytes: int = 20 * 1024 * 1024 * 1024
177
+ max_wall_seconds: int = 12 * 60 * 60
178
+
179
+ def __post_init__(self) -> None:
180
+ for name in (
181
+ "max_records",
182
+ "max_shard_entries",
183
+ "max_shard_bytes",
184
+ "max_pack_records",
185
+ "max_disk_bytes",
186
+ "max_wall_seconds",
187
+ ):
188
+ value = getattr(self, name)
189
+ if type(value) is not int or value < 1:
190
+ raise CatalogDeltaStreamRefused(
191
+ "DELTA_LIMIT", f"limits.{name}", "must be a positive integer"
192
+ )
193
+ if (
194
+ self.max_shard_entries > MAX_DELTA_SHARD_ENTRIES
195
+ or self.max_shard_bytes > MAX_DELTA_SHARD_BYTES
196
+ ):
197
+ raise CatalogDeltaStreamRefused(
198
+ "DELTA_LIMIT",
199
+ "limits.max_shard_entries",
200
+ "shard bounds may not exceed the fixed residency contract",
201
+ )
202
+
203
+ def to_dict(self) -> dict[str, int]:
204
+ return {
205
+ "max_records": self.max_records,
206
+ "max_shard_entries": self.max_shard_entries,
207
+ "max_shard_bytes": self.max_shard_bytes,
208
+ "max_pack_records": self.max_pack_records,
209
+ "max_disk_bytes": self.max_disk_bytes,
210
+ "max_wall_seconds": self.max_wall_seconds,
211
+ }
212
+
213
+
214
+ @dataclass(frozen=True)
215
+ class StreamingDeltaResult:
216
+ """One complete classification's exact durable outcome."""
217
+
218
+ manifest_sha256: str
219
+ identity_manifest_sha256: str
220
+ counts: dict[str, int]
221
+ receipt: dict[str, Any]
222
+
223
+
224
+ def semantic_classification_digest(entry: CatalogEntryV2) -> str:
225
+ """Digest exactly what decides sameness: identity, facts, and their provider provenance.
226
+
227
+ The managed version link, the append-only observations, and the sealed vector instances are
228
+ deliberately excluded. An author always spells a fresh entry as version one, so including the
229
+ version would make every re-observation look changed; including observations would make every
230
+ sweep look changed, which is precisely why observations are excluded from the provider digest.
231
+ """
232
+
233
+ if not isinstance(entry, CatalogEntryV2):
234
+ raise CatalogDeltaStreamRefused("DELTA_INPUT_SHARD", "entry", "must be a CatalogEntryV2")
235
+ payload = entry.to_dict()
236
+ for managed in ("entry_version", "previous_semantic_digest", "observations", "vectors"):
237
+ del payload[managed]
238
+ return canonical_sha256(payload)
239
+
240
+
241
+ def run_streaming_delta(
242
+ *,
243
+ authoring_root: Path,
244
+ expected_authoring_manifest_sha256: str,
245
+ output_root: Path,
246
+ output_descriptor: int | None = None,
247
+ measures: SweepMeasures,
248
+ observed_at: str,
249
+ limits: StreamingDeltaLimits | None = None,
250
+ predecessor_root: Path | None = None,
251
+ expected_predecessor_manifest_sha256: str | None = None,
252
+ predecessor_root_descriptor: int | None = None,
253
+ monotonic: Callable[[], float] = time.monotonic,
254
+ ) -> StreamingDeltaResult:
255
+ """Classify one authored generation against its predecessor identity/history exactly."""
256
+
257
+ bounds = StreamingDeltaLimits() if limits is None else limits
258
+ if not isinstance(bounds, StreamingDeltaLimits):
259
+ raise CatalogDeltaStreamRefused("DELTA_LIMIT", "limits", "must be StreamingDeltaLimits")
260
+ if not isinstance(measures, SweepMeasures):
261
+ raise CatalogDeltaStreamRefused(
262
+ "DELTA_MEASURE_CONTRACT", "measures", "must be SweepMeasures"
263
+ )
264
+ if not _is_digest(expected_authoring_manifest_sha256):
265
+ raise CatalogDeltaStreamRefused(
266
+ "DELTA_INPUT_DIGEST",
267
+ "expected_authoring_manifest_sha256",
268
+ "must be a lowercase SHA-256",
269
+ )
270
+ if (predecessor_root is None) != (expected_predecessor_manifest_sha256 is None):
271
+ raise CatalogDeltaStreamRefused(
272
+ "DELTA_PREDECESSOR_REQUIRED",
273
+ "predecessor",
274
+ "a predecessor is named by both its root and its exact manifest digest, or not at all",
275
+ )
276
+ deadline = monotonic() + bounds.max_wall_seconds
277
+
278
+ manifest, manifest_sha256 = _read_authoring_manifest(
279
+ Path(authoring_root), expected_authoring_manifest_sha256
280
+ )
281
+ policy = _resolve_policy(manifest)
282
+ _verify_measures(measures, manifest)
283
+ current = _AuthoredStream(Path(authoring_root), manifest, bounds)
284
+
285
+ predecessor: IdentityHistoryStore | None = None
286
+ if predecessor_root is not None:
287
+ predecessor = IdentityHistoryStore.open(
288
+ Path(predecessor_root),
289
+ expected_manifest_sha256=expected_predecessor_manifest_sha256,
290
+ root_descriptor=predecessor_root_descriptor,
291
+ )
292
+ if predecessor.provider_id != policy.provider_id:
293
+ raise CatalogDeltaStreamRefused(
294
+ "DELTA_PREDECESSOR_PROVIDER",
295
+ "predecessor.provider_id",
296
+ "the predecessor generation describes another provider",
297
+ )
298
+
299
+ identity_context = (
300
+ contextlib.nullcontext(None)
301
+ if output_descriptor is None
302
+ else retained_child_directory(
303
+ output_descriptor,
304
+ DELTA_IDENTITY_DIRNAME,
305
+ create=True,
306
+ label="delta.identity",
307
+ code_prefix="DELTA",
308
+ refusal=CatalogDeltaStreamRefused,
309
+ )
310
+ )
311
+ with identity_context as identity_descriptor:
312
+ identity_writer = IdentityHistoryWriter(
313
+ Path(output_root) / DELTA_IDENTITY_DIRNAME,
314
+ provider_id=policy.provider_id,
315
+ sweep_sha256=manifest_sha256,
316
+ predecessor_manifest_sha256=expected_predecessor_manifest_sha256,
317
+ max_pack_records=bounds.max_pack_records,
318
+ output_descriptor=identity_descriptor,
319
+ )
320
+ shards = _DeltaShardWriter(
321
+ Path(output_root),
322
+ bounds,
323
+ sweep_sha256=manifest_sha256,
324
+ output_descriptor=output_descriptor,
325
+ )
326
+ with shards.opened(), identity_writer.opened():
327
+ state = _merge(
328
+ current=current,
329
+ predecessor=predecessor,
330
+ policy=policy,
331
+ sweep_sha256=manifest_sha256,
332
+ observed_at=observed_at,
333
+ shards=shards,
334
+ identity_writer=identity_writer,
335
+ limits=bounds,
336
+ deadline=deadline,
337
+ monotonic=monotonic,
338
+ )
339
+ shards.flush_all()
340
+ _verify_denominator(state, manifest)
341
+ identity_manifest_sha256 = identity_writer.publish()
342
+ receipt = _receipt(
343
+ manifest=manifest,
344
+ manifest_sha256=manifest_sha256,
345
+ policy=policy,
346
+ measures=measures,
347
+ observed_at=observed_at,
348
+ state=state,
349
+ shards=shards,
350
+ limits=bounds,
351
+ identity_manifest_sha256=identity_manifest_sha256,
352
+ predecessor_manifest_sha256=expected_predecessor_manifest_sha256,
353
+ )
354
+ body = {key: value for key, value in receipt.items() if key != "manifest_sha256"}
355
+ delta_manifest_sha256 = shards.publish_manifest(
356
+ {**body, "schema_version": DELTA_MANIFEST_SCHEMA}
357
+ )
358
+ return StreamingDeltaResult(
359
+ manifest_sha256=delta_manifest_sha256,
360
+ identity_manifest_sha256=identity_manifest_sha256,
361
+ counts={name: state.counts[name] for name in DELTA_CLASSIFICATIONS},
362
+ receipt={**receipt, "manifest_sha256": delta_manifest_sha256},
363
+ )
364
+
365
+
366
+ # ------------------------------------------------------------------------------------------
367
+ # Inputs
368
+ # ------------------------------------------------------------------------------------------
369
+
370
+
371
+ def _read_authoring_manifest(root: Path, expected_sha256: str) -> tuple[dict[str, Any], str]:
372
+ raw = _read_bounded(root / MANIFEST_FILENAME, maximum=MAX_AUTHORING_MANIFEST_BYTES)
373
+ if raw is None:
374
+ raise CatalogDeltaStreamRefused(
375
+ "DELTA_INPUT_MANIFEST",
376
+ "authoring.manifest",
377
+ "no readable authored generation manifest at this root",
378
+ )
379
+ digest = sha256_bytes(raw)
380
+ if digest != expected_sha256:
381
+ raise CatalogDeltaStreamRefused(
382
+ "DELTA_INPUT_DIGEST",
383
+ "authoring.manifest",
384
+ "the retained authored generation is not the caller's exact manifest",
385
+ )
386
+ try:
387
+ manifest = parse_canonical_json(raw)
388
+ except CanonicalJSONError as error:
389
+ raise CatalogDeltaStreamRefused(
390
+ "DELTA_INPUT_MANIFEST", "authoring.manifest", "manifest is not canonical"
391
+ ) from error
392
+ expected_members = {
393
+ "schema_version",
394
+ "policy_id",
395
+ "policy_sha256",
396
+ "staging",
397
+ "coverage",
398
+ "counts",
399
+ "layer_text_digests",
400
+ "shards",
401
+ "root_sha256",
402
+ }
403
+ if (
404
+ not isinstance(manifest, dict)
405
+ or set(manifest) != expected_members
406
+ or manifest["schema_version"] != AUTHORING_MANIFEST_SCHEMA
407
+ or not isinstance(manifest["staging"], dict)
408
+ or not isinstance(manifest["counts"], dict)
409
+ or not isinstance(manifest["shards"], list)
410
+ or not coverage_is_valid(manifest["coverage"])
411
+ ):
412
+ raise CatalogDeltaStreamRefused(
413
+ "DELTA_INPUT_MANIFEST", "authoring.manifest", "manifest contract differs"
414
+ )
415
+ body = {key: value for key, value in manifest.items() if key != "root_sha256"}
416
+ if canonical_sha256(body) != manifest["root_sha256"]:
417
+ raise CatalogDeltaStreamRefused(
418
+ "DELTA_INPUT_MANIFEST", "authoring.manifest", "manifest root digest differs"
419
+ )
420
+ for name in ("raw_response_index_count", "normalized_page_index_count", "unique_records"):
421
+ if type(manifest["staging"].get(name)) is not int:
422
+ raise CatalogDeltaStreamRefused(
423
+ "DELTA_INPUT_MANIFEST",
424
+ f"authoring.manifest.staging.{name}",
425
+ "the authored staging coordinate is incomplete",
426
+ )
427
+ for name in ("records", *AUTHORING_DISPOSITIONS):
428
+ if type(manifest["counts"].get(name)) is not int:
429
+ raise CatalogDeltaStreamRefused(
430
+ "DELTA_INPUT_MANIFEST",
431
+ f"authoring.manifest.counts.{name}",
432
+ "the authored disposition counts are incomplete",
433
+ )
434
+ return manifest, digest
435
+
436
+
437
+ def _resolve_policy(manifest: dict[str, Any]) -> AuthoringPolicy:
438
+ policy = resolve_authoring_policy(manifest["policy_id"])
439
+ if manifest["policy_sha256"] != policy.digest:
440
+ raise CatalogDeltaStreamRefused(
441
+ "DELTA_INPUT_POLICY",
442
+ "authoring.manifest.policy_sha256",
443
+ "the authored generation names another spelling of this policy coordinate",
444
+ )
445
+ return policy
446
+
447
+
448
+ def _verify_measures(measures: SweepMeasures, manifest: dict[str, Any]) -> None:
449
+ """Bind the sweep that was harvested to the corpus the authored generation drew from.
450
+
451
+ This used to read ``measures.provider_records != manifest["counts"]["records"]`` -- the sweep
452
+ found exactly as many records as authoring wrote -- which is the same statement as "authoring
453
+ is exhaustive", asserted rather than recorded. It was true of every generation that could be
454
+ published, so it looked like an invariant; it was really the *only* coverage the pipeline
455
+ could express.
456
+
457
+ It is now two statements, because it was always two facts. The sweep's unique record count is
458
+ the corpus the selection chose from, and it must equal what the coverage says it drew from --
459
+ a coverage cannot claim a corpus its own harvest never found. Separately, what the selection
460
+ says it took must equal what authoring actually wrote. Between them they say what the single
461
+ comparison used to say whenever the coverage is exhaustive, and they say something true
462
+ rather than nothing at all when it is not.
463
+ """
464
+
465
+ staging = manifest["staging"]
466
+ coverage = manifest["coverage"]
467
+ if (
468
+ measures.responses != staging["raw_response_index_count"]
469
+ or measures.pages != staging["normalized_page_index_count"]
470
+ or measures.provider_records != coverage["corpus_records"]
471
+ ):
472
+ raise CatalogDeltaStreamRefused(
473
+ "DELTA_MEASURE_MISMATCH",
474
+ "measures",
475
+ "responses, pages, and provider records must equal the authored generation's own "
476
+ "distinct counts",
477
+ )
478
+ if (
479
+ coverage["corpus_records"] != staging["unique_records"]
480
+ or coverage["selected_records"] != manifest["counts"]["records"]
481
+ ):
482
+ raise CatalogDeltaStreamRefused(
483
+ "DELTA_COVERAGE_MISMATCH",
484
+ "authoring.manifest.coverage",
485
+ "the stated coverage must draw from the staging coordinate's corpus and account for "
486
+ "every authored record",
487
+ )
488
+
489
+
490
+ class _AuthoredStream:
491
+ """One authored generation as one ascending record stream, one shard resident."""
492
+
493
+ def __init__(self, root: Path, manifest: dict[str, Any], limits: StreamingDeltaLimits) -> None:
494
+ self.root = Path(root)
495
+ self.manifest = manifest
496
+ self.limits = limits
497
+ for descriptor in self.manifest["shards"]:
498
+ self._validate_descriptor(descriptor)
499
+ self._iterator = self._records()
500
+ self._peeked: tuple[bytes, dict[str, Any], str] | None = None
501
+ self._exhausted = False
502
+
503
+ def peek(self) -> tuple[bytes, dict[str, Any], str] | None:
504
+ if self._peeked is None and not self._exhausted:
505
+ self._peeked = next(self._iterator, None)
506
+ if self._peeked is None:
507
+ self._exhausted = True
508
+ return self._peeked
509
+
510
+ def take(self) -> tuple[bytes, dict[str, Any], str]:
511
+ item = self.peek()
512
+ if item is None:
513
+ raise CatalogDeltaStreamRefused(
514
+ "DELTA_INPUT_SHARD", "authoring.shards", "stream exhausted"
515
+ )
516
+ self._peeked = None
517
+ return item
518
+
519
+ def _records(self) -> Iterator[tuple[bytes, dict[str, Any], str]]:
520
+ last: bytes | None = None
521
+ for descriptor in self.manifest["shards"]:
522
+ for item in self._shard(descriptor):
523
+ if (
524
+ not isinstance(item, dict)
525
+ or not isinstance(item.get("provider_record_id"), str)
526
+ or not item["provider_record_id"]
527
+ or item.get("disposition") not in AUTHORING_DISPOSITIONS
528
+ or not isinstance(item.get("reason_code"), str)
529
+ or not isinstance(item.get("entry"), (dict, type(None)))
530
+ ):
531
+ raise CatalogDeltaStreamRefused(
532
+ "DELTA_INPUT_SHARD", "authoring.shard.items[]", "authored item differs"
533
+ )
534
+ key = item["provider_record_id"].encode("utf-8")
535
+ if last is not None and key <= last:
536
+ raise CatalogDeltaStreamRefused(
537
+ "DELTA_INPUT_ORDER",
538
+ "authoring.shards",
539
+ "an authored generation must ascend by provider record",
540
+ )
541
+ last = key
542
+ yield key, item, descriptor["sha256"]
543
+
544
+ def _shard(self, descriptor: Any) -> list[Any]:
545
+ self._validate_descriptor(descriptor)
546
+ raw = _read_bounded(
547
+ self.root / SHARDS_DIRNAME / f"{descriptor['sha256']}.json",
548
+ maximum=MAX_DELTA_INPUT_SHARD_BYTES,
549
+ )
550
+ if (
551
+ raw is None
552
+ or len(raw) != descriptor["bytes"]
553
+ or sha256_bytes(raw) != descriptor["sha256"]
554
+ ):
555
+ raise CatalogDeltaStreamRefused(
556
+ "DELTA_INPUT_SHARD", "authoring.shard", "a committed authored shard differs"
557
+ )
558
+ try:
559
+ payload = parse_canonical_json(raw)
560
+ except CanonicalJSONError as error:
561
+ raise CatalogDeltaStreamRefused(
562
+ "DELTA_INPUT_SHARD", "authoring.shard", "shard is not canonical"
563
+ ) from error
564
+ if (
565
+ not isinstance(payload, dict)
566
+ or payload.get("schema_version") != AUTHORING_SHARD_SCHEMA
567
+ or not isinstance(payload.get("items"), list)
568
+ or len(payload["items"]) != descriptor["item_count"]
569
+ ):
570
+ raise CatalogDeltaStreamRefused(
571
+ "DELTA_INPUT_SHARD", "authoring.shard", "shard contract differs"
572
+ )
573
+ return payload["items"]
574
+
575
+ @staticmethod
576
+ def _validate_descriptor(descriptor: Any) -> None:
577
+ if (
578
+ not isinstance(descriptor, dict)
579
+ or not _is_digest(descriptor.get("sha256"))
580
+ or type(descriptor.get("bytes")) is not int
581
+ or not 1 <= descriptor["bytes"] <= MAX_DELTA_INPUT_SHARD_BYTES
582
+ or type(descriptor.get("item_count")) is not int
583
+ or not 1 <= descriptor["item_count"] <= MAX_AUTHORING_SHARD_ENTRIES
584
+ ):
585
+ raise CatalogDeltaStreamRefused(
586
+ "DELTA_INPUT_MANIFEST", "authoring.manifest.shards[]", "shard descriptor is invalid"
587
+ )
588
+
589
+
590
+ class _PredecessorStream:
591
+ """One predecessor identity generation as one ascending record stream, one pack resident."""
592
+
593
+ def __init__(self, store: IdentityHistoryStore | None) -> None:
594
+ self._iterator: Iterator[IdentityHistory] = iter(()) if store is None else store.stream()
595
+ self._peeked: IdentityHistory | None = None
596
+ self._exhausted = False
597
+
598
+ def peek(self) -> IdentityHistory | None:
599
+ if self._peeked is None and not self._exhausted:
600
+ self._peeked = next(self._iterator, None)
601
+ if self._peeked is None:
602
+ self._exhausted = True
603
+ return self._peeked
604
+
605
+ def take(self) -> IdentityHistory:
606
+ item = self.peek()
607
+ if item is None:
608
+ raise CatalogDeltaStreamRefused(
609
+ "DELTA_PREDECESSOR_STREAM", "predecessor", "stream exhausted"
610
+ )
611
+ self._peeked = None
612
+ return item
613
+
614
+
615
+ # ------------------------------------------------------------------------------------------
616
+ # The merge
617
+ # ------------------------------------------------------------------------------------------
618
+
619
+
620
+ @dataclass
621
+ class _MergeState:
622
+ counts: dict[str, int]
623
+ predecessor_records: int = 0
624
+ reused_entries: int = 0
625
+ reused_vectors: int = 0
626
+ scheduled_encodes: int = 0
627
+
628
+
629
+ def _merge(
630
+ *,
631
+ current: _AuthoredStream,
632
+ predecessor: IdentityHistoryStore | None,
633
+ policy: AuthoringPolicy,
634
+ sweep_sha256: str,
635
+ observed_at: str,
636
+ shards: _DeltaShardWriter,
637
+ identity_writer: IdentityHistoryWriter,
638
+ limits: StreamingDeltaLimits,
639
+ deadline: float,
640
+ monotonic: Callable[[], float],
641
+ ) -> _MergeState:
642
+ previous = _PredecessorStream(predecessor)
643
+ state = _MergeState(counts=dict.fromkeys(DELTA_CLASSIFICATIONS, 0))
644
+ classified = 0
645
+ while True:
646
+ head = current.peek()
647
+ retained = previous.peek()
648
+ if head is None and retained is None:
649
+ return state
650
+ if classified >= limits.max_records:
651
+ raise CatalogDeltaStreamRefused(
652
+ "DELTA_RECORD_LIMIT",
653
+ "delta.records",
654
+ "classification exceeds the caller's record bound; no manifest was published",
655
+ )
656
+ if monotonic() >= deadline:
657
+ raise CatalogDeltaStreamRefused(
658
+ "DELTA_WALL_LIMIT",
659
+ "delta.records",
660
+ "classification exceeds the caller's wall bound; no manifest was published",
661
+ )
662
+ classified += 1
663
+ if retained is None or (head is not None and head[0] < retained.sort_key):
664
+ _key, item, shard_sha256 = current.take()
665
+ _classify_current(
666
+ item,
667
+ shard_sha256=shard_sha256,
668
+ retained=None,
669
+ policy=policy,
670
+ sweep_sha256=sweep_sha256,
671
+ observed_at=observed_at,
672
+ shards=shards,
673
+ identity_writer=identity_writer,
674
+ state=state,
675
+ )
676
+ continue
677
+ if head is None or retained.sort_key < head[0]:
678
+ _classify_absent(
679
+ previous.take(),
680
+ sweep_sha256=sweep_sha256,
681
+ observed_at=observed_at,
682
+ shards=shards,
683
+ identity_writer=identity_writer,
684
+ state=state,
685
+ )
686
+ continue
687
+ _key, item, shard_sha256 = current.take()
688
+ _classify_current(
689
+ item,
690
+ shard_sha256=shard_sha256,
691
+ retained=previous.take(),
692
+ policy=policy,
693
+ sweep_sha256=sweep_sha256,
694
+ observed_at=observed_at,
695
+ shards=shards,
696
+ identity_writer=identity_writer,
697
+ state=state,
698
+ )
699
+ state.predecessor_records += 1
700
+
701
+
702
+ def _classify_current(
703
+ item: dict[str, Any],
704
+ *,
705
+ shard_sha256: str,
706
+ retained: IdentityHistory | None,
707
+ policy: AuthoringPolicy,
708
+ sweep_sha256: str,
709
+ observed_at: str,
710
+ shards: _DeltaShardWriter,
711
+ identity_writer: IdentityHistoryWriter,
712
+ state: _MergeState,
713
+ ) -> None:
714
+ record_id = item["provider_record_id"]
715
+ entry = _entry(item)
716
+ identity = _identity(record_id, entry, retained, policy)
717
+ if item["disposition"] != "authored":
718
+ history = record_identity_disposition(
719
+ retained,
720
+ identity=identity,
721
+ policy_sha256=policy.digest,
722
+ harvester_coordinate=policy.harvester_coordinate,
723
+ disposition=item["disposition"],
724
+ reason_code=item["reason_code"],
725
+ observed_at=observed_at,
726
+ sweep_sha256=sweep_sha256,
727
+ provider_record_sha256=(
728
+ None if entry is None else entry.provider_record.provider_record_sha256
729
+ ),
730
+ )
731
+ identity_writer.append(history)
732
+ state.counts[item["disposition"]] += 1
733
+ shards.emit(
734
+ {
735
+ "provider_record_id": record_id,
736
+ "classification": item["disposition"],
737
+ "disposition": item["disposition"],
738
+ "entry_id": identity.entry_id,
739
+ "entry_version": None,
740
+ "semantic_facts_digest": (
741
+ None if entry is None else semantic_classification_digest(entry)
742
+ ),
743
+ "previous_semantic_digest": None,
744
+ "provider_record_sha256": (
745
+ None if entry is None else entry.provider_record.provider_record_sha256
746
+ ),
747
+ "reason_code": item["reason_code"],
748
+ "reused_vector_sha256s": [],
749
+ "encode_scheduled": False,
750
+ "authoring_shard_sha256": shard_sha256,
751
+ }
752
+ )
753
+ return
754
+
755
+ if entry is None:
756
+ raise CatalogDeltaStreamRefused(
757
+ "DELTA_INPUT_SHARD",
758
+ "authoring.shard.items[].entry",
759
+ "an authored record must carry its entry",
760
+ )
761
+ digest = semantic_classification_digest(entry)
762
+ retained_head = None if retained is None else retained.head
763
+ history = advance_identity_history(
764
+ retained,
765
+ identity=identity,
766
+ policy_sha256=policy.digest,
767
+ harvester_coordinate=policy.harvester_coordinate,
768
+ semantic_facts_digest=digest,
769
+ provider_record_sha256=entry.provider_record.provider_record_sha256,
770
+ observation=_observation(entry, digest, sweep_sha256),
771
+ )
772
+ identity_writer.append(history)
773
+ head = history.head
774
+ assert head is not None
775
+ if retained_head is None:
776
+ classification = "new"
777
+ elif retained_head.semantic_facts_digest == digest:
778
+ classification = "unchanged"
779
+ else:
780
+ classification = "changed"
781
+ state.counts[classification] += 1
782
+ if classification == "unchanged":
783
+ state.reused_entries += 1
784
+ state.reused_vectors += len(head.vector_sha256s)
785
+ else:
786
+ state.scheduled_encodes += 1
787
+ shards.emit(
788
+ {
789
+ "provider_record_id": record_id,
790
+ "classification": classification,
791
+ "disposition": "authored",
792
+ "entry_id": identity.entry_id,
793
+ "entry_version": head.entry_version,
794
+ "semantic_facts_digest": head.semantic_facts_digest,
795
+ "previous_semantic_digest": head.previous_semantic_digest,
796
+ "provider_record_sha256": head.provider_record_sha256,
797
+ "reason_code": item["reason_code"],
798
+ "reused_vector_sha256s": list(head.vector_sha256s)
799
+ if classification == "unchanged"
800
+ else [],
801
+ "encode_scheduled": classification != "unchanged",
802
+ "authoring_shard_sha256": shard_sha256,
803
+ }
804
+ )
805
+
806
+
807
+ def _classify_absent(
808
+ retained: IdentityHistory,
809
+ *,
810
+ sweep_sha256: str,
811
+ observed_at: str,
812
+ shards: _DeltaShardWriter,
813
+ identity_writer: IdentityHistoryWriter,
814
+ state: _MergeState,
815
+ ) -> None:
816
+ head = retained.head
817
+ history = (
818
+ retained
819
+ if head is None
820
+ else observe_identity_absence(retained, observed_at=observed_at, sweep_sha256=sweep_sha256)
821
+ )
822
+ identity_writer.append(history)
823
+ state.counts["absent"] += 1
824
+ state.predecessor_records += 1
825
+ shards.emit(
826
+ {
827
+ "provider_record_id": retained.identity.provider_record_id,
828
+ "classification": "absent",
829
+ "disposition": None,
830
+ "entry_id": retained.identity.entry_id,
831
+ "entry_version": None if head is None else head.entry_version,
832
+ "semantic_facts_digest": None if head is None else head.semantic_facts_digest,
833
+ "previous_semantic_digest": None,
834
+ "provider_record_sha256": None if head is None else head.provider_record_sha256,
835
+ "reason_code": DELTA_ABSENCE_REASON,
836
+ "reused_vector_sha256s": [],
837
+ "encode_scheduled": False,
838
+ "authoring_shard_sha256": None,
839
+ }
840
+ )
841
+
842
+
843
+ def _entry(item: dict[str, Any]) -> CatalogEntryV2 | None:
844
+ if item["entry"] is None:
845
+ return None
846
+ try:
847
+ return catalog_entry_v2_from_dict(item["entry"])
848
+ except SourceContractError as error:
849
+ raise CatalogDeltaStreamRefused(
850
+ "DELTA_INPUT_SHARD",
851
+ "authoring.shard.items[].entry",
852
+ "an authored entry does not rebuild under the v2 contract",
853
+ ) from error
854
+
855
+
856
+ def _identity(
857
+ record_id: str,
858
+ entry: CatalogEntryV2 | None,
859
+ retained: IdentityHistory | None,
860
+ policy: AuthoringPolicy,
861
+ ) -> ProviderIdentity:
862
+ if retained is not None:
863
+ identity = retained.identity
864
+ else:
865
+ identity = ProviderIdentity(
866
+ provider_id=policy.provider_id,
867
+ provider_record_id=record_id,
868
+ entry_id=derive_entry_id(policy.provider_id, record_id, prefix=policy.entry_id_prefix),
869
+ )
870
+ if entry is not None and entry.entry_id != identity.entry_id:
871
+ raise CatalogDeltaStreamRefused(
872
+ "DELTA_IDENTITY_DERIVATION",
873
+ "authoring.shard.items[].entry.entry_id",
874
+ "an authored entry id must equal this provider record's persisted identity",
875
+ )
876
+ if entry is not None and entry.provider_record.provider_record_id != record_id:
877
+ raise CatalogDeltaStreamRefused(
878
+ "DELTA_IDENTITY_DERIVATION",
879
+ "authoring.shard.items[].entry.provider_record",
880
+ "an authored entry must name the provider record it was authored from",
881
+ )
882
+ return identity
883
+
884
+
885
+ def _observation(entry: CatalogEntryV2, digest: str, sweep_sha256: str) -> ObservationEvidence:
886
+ latest = entry.observations[-1]
887
+ return ObservationEvidence(
888
+ observed_at=latest.observed_at,
889
+ semantic_facts_digest=digest,
890
+ sweep_sha256=sweep_sha256,
891
+ response_evidence_sha256=latest.response_evidence_sha256,
892
+ page_evidence_sha256=latest.page_evidence_sha256,
893
+ )
894
+
895
+
896
+ def _verify_denominator(state: _MergeState, manifest: dict[str, Any]) -> None:
897
+ """Every record this generation *holds* lands in exactly one current classification.
898
+
899
+ The denominator is the selected population rather than the swept one. Under an exhaustive
900
+ coverage the two are the same number and this is the check it has always been. Under a bound
901
+ they differ by exactly the records the rule did not select, and holding the classification to
902
+ the swept population would refuse every bounded generation -- not because a record went
903
+ missing, but because the check was measuring coverage while claiming to measure completeness.
904
+
905
+ Completeness of the *classification* is what is load-bearing here: no authored record may be
906
+ dropped, double-counted, or left unclassified. Coverage is stated in the coverage block, and
907
+ ``_verify_measures`` has already bound both of its populations to evidence.
908
+ """
909
+
910
+ selected = manifest["coverage"]["selected_records"]
911
+ classified = sum(state.counts[name] for name in _CURRENT_CLASSIFICATIONS)
912
+ if classified != selected:
913
+ raise CatalogDeltaStreamRefused(
914
+ "DELTA_DENOMINATOR",
915
+ "delta.counts",
916
+ "every authored record must land in exactly one current classification; "
917
+ f"classified={classified}, selected_records={selected}",
918
+ )
919
+
920
+
921
+ # ------------------------------------------------------------------------------------------
922
+ # Outputs
923
+ # ------------------------------------------------------------------------------------------
924
+
925
+
926
+ class _DeltaShardWriter:
927
+ """Immutable content-addressed output shards, one bounded buffer per classification."""
928
+
929
+ def __init__(
930
+ self,
931
+ root: Path,
932
+ limits: StreamingDeltaLimits,
933
+ *,
934
+ sweep_sha256: str,
935
+ output_descriptor: int | None = None,
936
+ ) -> None:
937
+ self.limits = limits
938
+ self.sweep_sha256 = sweep_sha256
939
+ self.shards: list[dict[str, Any]] = []
940
+ self.chains: dict[str, str | None] = dict.fromkeys(DELTA_CLASSIFICATIONS)
941
+ self._directory = DurableDirectory(
942
+ root,
943
+ member_dirname=DELTA_SHARDS_DIRNAME,
944
+ code_prefix="DELTA",
945
+ refusal=CatalogDeltaStreamRefused,
946
+ root_descriptor=output_descriptor,
947
+ )
948
+ self._buffers: dict[str, list[dict[str, Any]]] = {
949
+ name: [] for name in DELTA_CLASSIFICATIONS
950
+ }
951
+ self._buffered_bytes: dict[str, int] = dict.fromkeys(DELTA_CLASSIFICATIONS, 0)
952
+
953
+ @contextlib.contextmanager
954
+ def opened(self) -> Iterator[_DeltaShardWriter]:
955
+ with self._directory.opened():
956
+ if (
957
+ self._directory.read(DELTA_MANIFEST_FILENAME, maximum=MAX_DELTA_MANIFEST_BYTES)
958
+ is not None
959
+ ):
960
+ raise CatalogDeltaStreamRefused(
961
+ "DELTA_OUTPUT_EXISTS",
962
+ "delta.output",
963
+ "a published classification is immutable; classify into a new output root",
964
+ )
965
+ yield self
966
+
967
+ def emit(self, item: dict[str, Any]) -> None:
968
+ classification = item["classification"]
969
+ buffer = self._buffers[classification]
970
+ size = len(canonical_json_bytes(item))
971
+ if buffer and (
972
+ len(buffer) >= self.limits.max_shard_entries
973
+ or self._buffered_bytes[classification] + size + len(buffer) + 4_096
974
+ > self.limits.max_shard_bytes
975
+ ):
976
+ self.flush(classification)
977
+ buffer = self._buffers[classification]
978
+ buffer.append(item)
979
+ self._buffered_bytes[classification] += size
980
+ self.chains[classification] = canonical_sha256(
981
+ {"previous_sha256": self.chains[classification], "item": item}
982
+ )
983
+
984
+ def flush(self, classification: str) -> None:
985
+ buffer = self._buffers[classification]
986
+ if not buffer:
987
+ return
988
+ payload = {
989
+ "schema_version": DELTA_SHARD_SCHEMA,
990
+ "classification": classification,
991
+ "sweep_sha256": self.sweep_sha256,
992
+ "first_provider_record_id": buffer[0]["provider_record_id"],
993
+ "last_provider_record_id": buffer[-1]["provider_record_id"],
994
+ "items": buffer,
995
+ }
996
+ digest, size = self._directory.write_member(payload, maximum=self.limits.max_shard_bytes)
997
+ self.shards.append(
998
+ {
999
+ "sha256": digest,
1000
+ "bytes": size,
1001
+ "classification": classification,
1002
+ "item_count": len(buffer),
1003
+ "first_provider_record_id": payload["first_provider_record_id"],
1004
+ "last_provider_record_id": payload["last_provider_record_id"],
1005
+ }
1006
+ )
1007
+ self._buffers[classification] = []
1008
+ self._buffered_bytes[classification] = 0
1009
+ if self._directory.bytes_written > self.limits.max_disk_bytes:
1010
+ raise CatalogDeltaStreamRefused(
1011
+ "DELTA_DISK_LIMIT", "delta.output", "classification exceeds its local disk bound"
1012
+ )
1013
+
1014
+ def flush_all(self) -> None:
1015
+ for classification in DELTA_CLASSIFICATIONS:
1016
+ self.flush(classification)
1017
+
1018
+ def publish_manifest(self, body: dict[str, Any]) -> str:
1019
+ manifest = {**body, "root_sha256": canonical_sha256(body)}
1020
+ return self._directory.publish(
1021
+ DELTA_MANIFEST_FILENAME, manifest, maximum=MAX_DELTA_MANIFEST_BYTES
1022
+ )
1023
+
1024
+
1025
+ def _receipt(
1026
+ *,
1027
+ manifest: dict[str, Any],
1028
+ manifest_sha256: str,
1029
+ policy: AuthoringPolicy,
1030
+ measures: SweepMeasures,
1031
+ observed_at: str,
1032
+ state: _MergeState,
1033
+ shards: _DeltaShardWriter,
1034
+ limits: StreamingDeltaLimits,
1035
+ identity_manifest_sha256: str,
1036
+ predecessor_manifest_sha256: str | None,
1037
+ ) -> dict[str, Any]:
1038
+ counts = {name: state.counts[name] for name in DELTA_CLASSIFICATIONS}
1039
+ return {
1040
+ "schema_version": DELTA_RECEIPT_SCHEMA,
1041
+ "observed_at": observed_at,
1042
+ "inputs": {
1043
+ "authoring_manifest_sha256": manifest_sha256,
1044
+ "authoring_root_sha256": manifest["root_sha256"],
1045
+ "policy_id": policy.policy_id,
1046
+ "policy_sha256": policy.digest,
1047
+ "provider_id": policy.provider_id,
1048
+ "harvester_coordinate": policy.harvester_coordinate,
1049
+ "staging": manifest["staging"],
1050
+ "coverage": manifest["coverage"],
1051
+ "authored_counts": manifest["counts"],
1052
+ "predecessor_identity_manifest_sha256": predecessor_manifest_sha256,
1053
+ },
1054
+ "measures": measures.to_dict(),
1055
+ "counts": {
1056
+ **counts,
1057
+ "provider_records": measures.provider_records,
1058
+ "predecessor_records": state.predecessor_records,
1059
+ },
1060
+ # Three numbers, because a reader of this receipt alone must be able to tell a
1061
+ # classification that dropped records from a coverage that never selected them.
1062
+ "denominator": {
1063
+ "provider_records": measures.provider_records,
1064
+ "selected_records": manifest["coverage"]["selected_records"],
1065
+ "classified": sum(state.counts[name] for name in _CURRENT_CLASSIFICATIONS),
1066
+ },
1067
+ "reuse": {
1068
+ "reused_entries": state.reused_entries,
1069
+ "reused_vectors": state.reused_vectors,
1070
+ "scheduled_encodes": state.scheduled_encodes,
1071
+ },
1072
+ "outputs": {
1073
+ "identity_manifest_sha256": identity_manifest_sha256,
1074
+ "shard_count": len(shards.shards),
1075
+ "shards": shards.shards,
1076
+ "classification_digests": dict(shards.chains),
1077
+ },
1078
+ "limits": limits.to_dict(),
1079
+ "manifest_sha256": None,
1080
+ }
1081
+
1082
+
1083
+ def _read_bounded(path: Path, *, maximum: int) -> bytes | None:
1084
+ try:
1085
+ return read_bounded_path(path, maximum=maximum, missing_ok=True)
1086
+ except BoundedReadFailure as error:
1087
+ raise CatalogDeltaStreamRefused(
1088
+ "DELTA_INPUT_PATH",
1089
+ "delta.input",
1090
+ "input members must be stable bounded single-link regular files",
1091
+ ) from error
1092
+
1093
+
1094
+ def _is_digest(value: Any) -> bool:
1095
+ return (
1096
+ isinstance(value, str) and len(value) == 64 and all(c in "0123456789abcdef" for c in value)
1097
+ )