mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,2802 @@
1
+ """The streaming packed publisher, and the only place a packed generation's work is counted.
2
+
3
+ This module turns one classified sweep into one packed generation. Its inputs are exactly the
4
+ delta's classification shards, the successor identity generation
5
+ behind them, and the authored shards the classification was computed from -- and its output is the
6
+ range-local schema in :mod:`packed_catalog`.
7
+
8
+ Four properties are load-bearing.
9
+
10
+ *One pack in, one pack out.* The three publishable classification streams are already ascending by
11
+ provider record, so publication is a merge rather than a join: at most one shard per classification,
12
+ one authored shard, one identity pack, one predecessor range and one range under construction are
13
+ ever resident. Nothing is indexed and nothing is sorted.
14
+
15
+ *Encoding happens exactly once per nonempty fresh range, layer and backend.* A range that contains
16
+ at least one new or changed entry issues one bounded ``encode_many`` invocation per layer per
17
+ backend, carrying exactly that range's fresh texts; an unchanged entry's four layer vectors are read
18
+ back out of the predecessor's packs instead. So the totals are equations rather than observations
19
+ after the fact: ``batch_invocations = nonempty_fresh_range_count * 4 * backend_count`` and
20
+ ``encoded_layer_items = fresh_entries * 4 * backend_count``, and this module refuses to install a
21
+ head whose counters do not satisfy them.
22
+
23
+ *The counters are produced, never accepted.* There is no parameter through which a caller can hand
24
+ this function a work total. Every number comes from a seam-minted :class:`BatchEncodeResult` this
25
+ module summed in order, and when a publication spans several invocations the head carries each
26
+ invocation's record beside the ordered sum of them.
27
+
28
+ *A publication is complete, resumable, or nothing.* Members are immutable and content-addressed;
29
+ after each range is durable, a typed-incomplete work journal binds the output root, the ordered
30
+ completed-range member digests and this invocation's counter record. An explicit resume validates
31
+ the journal's own digest, re-reads every already-emitted member and refuses one whose bytes moved,
32
+ and continues from the first missing one without re-encoding a pre-head complete range. Before the
33
+ head is installed, a terminal record captures the exact final journal and packed-receipt inputs; the
34
+ head binds that record's digest. Recovering that mutable terminal state deterministically replays
35
+ the generation before completing its evidence transaction. A crash therefore leaves either
36
+ ordinary range work to resume or a head-bound state that can be verified and completed exactly.
37
+ """
38
+
39
+ from __future__ import annotations
40
+
41
+ import contextlib
42
+ import os
43
+ import stat
44
+ import time
45
+ from collections.abc import Callable, Iterator, Mapping, Sequence
46
+ from dataclasses import dataclass, field
47
+ from pathlib import Path
48
+ from typing import Any
49
+
50
+ from mostlyright.data_harness.canonical import (
51
+ CanonicalJSONError,
52
+ canonical_json_bytes,
53
+ canonical_sha256,
54
+ parse_canonical_json,
55
+ sha256_bytes,
56
+ )
57
+ from mostlyright.data_harness.sources.catalog.admission import admit_public_fact_bytes
58
+ from mostlyright.data_harness.sources.catalog.authoring_policy import (
59
+ AUTHORING_DISPOSITIONS,
60
+ AUTHORING_REASON_CODES,
61
+ MAX_LAYER_TEXT_BYTES,
62
+ )
63
+ from mostlyright.data_harness.sources.catalog.authoring_shards import (
64
+ AUTHORING_MANIFEST_SCHEMA,
65
+ AUTHORING_SHARD_SCHEMA,
66
+ MANIFEST_FILENAME,
67
+ MAX_AUTHORING_MANIFEST_BYTES,
68
+ MAX_AUTHORING_SHARD_BYTES,
69
+ SHARDS_DIRNAME,
70
+ shard_descriptor_is_valid,
71
+ )
72
+ from mostlyright.data_harness.sources.catalog.bounded_io import (
73
+ BoundedReadFailure,
74
+ read_bounded_at,
75
+ read_bounded_path,
76
+ )
77
+ from mostlyright.data_harness.sources.catalog.contracts import EMBEDDING_LAYERS
78
+ from mostlyright.data_harness.sources.catalog.coverage import coverage_is_valid
79
+ from mostlyright.data_harness.sources.catalog.embedding import (
80
+ BOUNDED_SCALAR_ADAPTER,
81
+ BatchEncodeResult,
82
+ EmbeddingBackend,
83
+ )
84
+ from mostlyright.data_harness.sources.catalog.entry_v2 import catalog_entry_v2_from_dict
85
+ from mostlyright.data_harness.sources.catalog.identity_history import (
86
+ IdentityHistory,
87
+ IdentityHistoryStore,
88
+ derive_entry_id,
89
+ )
90
+ from mostlyright.data_harness.sources.catalog.packed_catalog import (
91
+ COUNTER_MEMBERS,
92
+ MAX_BACKEND_DIMENSION,
93
+ MAX_FACTS_MEMBER_BYTES,
94
+ MAX_HISTORY_MEMBER_BYTES,
95
+ MAX_PACKED_BACKENDS,
96
+ MAX_PACKED_HEAD_BYTES,
97
+ MAX_POSTING_SEGMENT_BYTES,
98
+ MAX_RANGE_DESCRIPTORS,
99
+ MAX_RANGE_MANIFEST_BYTES,
100
+ MAX_VECTOR_MEMBER_BYTES,
101
+ MEMBER_HEADER_BYTES,
102
+ PACKED_HEAD_FILENAME,
103
+ RANGE_ENTRIES,
104
+ BuiltPackedRange,
105
+ CatalogPackedRefused,
106
+ PackedBackend,
107
+ PackedDirectory,
108
+ PackedRangeDescriptor,
109
+ PackedRangeInput,
110
+ VerifiedPackedGeneration,
111
+ bound_segment_bytes,
112
+ build_packed_head,
113
+ build_packed_range,
114
+ encode_layer_batch,
115
+ member_descriptor_from_dict,
116
+ parse_vector_member,
117
+ range_binding_sha256,
118
+ range_descriptor_from_dict,
119
+ ranges_chain_sha256,
120
+ verify_packed_generation,
121
+ verify_packed_range,
122
+ )
123
+ from mostlyright.data_harness.sources.catalog.streaming_delta import (
124
+ DELTA_IDENTITY_DIRNAME,
125
+ DELTA_MANIFEST_FILENAME,
126
+ DELTA_MANIFEST_SCHEMA,
127
+ DELTA_SHARDS_DIRNAME,
128
+ MAX_DELTA_MANIFEST_BYTES,
129
+ MAX_DELTA_SHARD_BYTES,
130
+ )
131
+ from mostlyright.data_harness.sources.contracts import SourceContractError
132
+
133
+ PACKED_JOURNAL_SCHEMA = "harness-catalog-packed-journal.v1"
134
+ PACKED_JOURNAL_FILENAME = "packed-journal.json"
135
+ PACKED_RECEIPT_SCHEMA = "mr-data-catalog-packed.v1"
136
+ PACKED_TERMINAL_SCHEMA = "harness-catalog-packed-terminal.v1"
137
+ PACKED_TERMINAL_FILENAME = "packed-terminal.json"
138
+ MAX_PACKED_JOURNAL_BYTES = 8 * 1024 * 1024
139
+ MAX_PACKED_TERMINAL_BYTES = 16 * 1024 * 1024
140
+
141
+ #: The three classifications that put an entry in a packed generation. ``absent`` describes a
142
+ #: predecessor record that is gone, and the three non-admitted dispositions never had an entry.
143
+ PUBLISHABLE_CLASSIFICATIONS = ("changed", "new", "unchanged")
144
+ FRESH_CLASSIFICATIONS = ("changed", "new")
145
+
146
+ #: Each member kind's own byte ceiling, so a journal reverification allocates a kind's exact bound
147
+ #: rather than the largest bound any kind has.
148
+ MEMBER_MAXIMUM_BYTES = {
149
+ "facts": MAX_FACTS_MEMBER_BYTES,
150
+ "history": MAX_HISTORY_MEMBER_BYTES,
151
+ "vector": MAX_VECTOR_MEMBER_BYTES,
152
+ "posting": MAX_POSTING_SEGMENT_BYTES,
153
+ "bound": bound_segment_bytes(dimensions=MAX_BACKEND_DIMENSION),
154
+ "range": MAX_RANGE_MANIFEST_BYTES,
155
+ }
156
+
157
+ _OPEN_MEMBER = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0)
158
+ _OPEN_DIRECTORY = _OPEN_MEMBER | getattr(os, "O_DIRECTORY", 0)
159
+
160
+
161
+ def _refuse(code: str, path: str, detail: str) -> CatalogPackedRefused:
162
+ return CatalogPackedRefused(code, path, detail)
163
+
164
+
165
+ def _is_digest(value: Any) -> bool:
166
+ return (
167
+ isinstance(value, str) and len(value) == 64 and all(c in "0123456789abcdef" for c in value)
168
+ )
169
+
170
+
171
+ @dataclass(frozen=True)
172
+ class PackedLimits:
173
+ """Every bound one publication may consume, stated before it consumes any of them."""
174
+
175
+ max_ranges: int = MAX_RANGE_DESCRIPTORS
176
+ max_entries: int = MAX_RANGE_DESCRIPTORS * RANGE_ENTRIES
177
+ max_terms_per_segment: int | None = None
178
+ max_disk_bytes: int = 64 * 1024 * 1024 * 1024
179
+ max_wall_seconds: int = 24 * 60 * 60
180
+
181
+ def __post_init__(self) -> None:
182
+ if type(self.max_ranges) is not int or not 1 <= self.max_ranges <= MAX_RANGE_DESCRIPTORS:
183
+ raise _refuse(
184
+ "PACKED_LIMIT",
185
+ "limits.max_ranges",
186
+ f"a generation holds 1 to {MAX_RANGE_DESCRIPTORS} ranges",
187
+ )
188
+ if type(self.max_entries) is not int or self.max_entries < 1:
189
+ raise _refuse("PACKED_LIMIT", "limits.max_entries", "must be a positive integer")
190
+ if type(self.max_disk_bytes) is not int or self.max_disk_bytes < 1:
191
+ raise _refuse("PACKED_LIMIT", "limits.max_disk_bytes", "must be a positive integer")
192
+ if type(self.max_wall_seconds) is not int or self.max_wall_seconds < 1:
193
+ raise _refuse("PACKED_LIMIT", "limits.max_wall_seconds", "must be a positive integer")
194
+
195
+ def to_dict(self) -> dict[str, Any]:
196
+ return {
197
+ "max_ranges": self.max_ranges,
198
+ "max_entries": self.max_entries,
199
+ "max_terms_per_segment": self.max_terms_per_segment,
200
+ "max_disk_bytes": self.max_disk_bytes,
201
+ "max_wall_seconds": self.max_wall_seconds,
202
+ }
203
+
204
+
205
+ @dataclass(frozen=True)
206
+ class PackedRangeProgress:
207
+ """One range that is durable on disk, handed to the caller's observer after it is installed."""
208
+
209
+ range_index: int
210
+ first_key: str
211
+ last_key: str
212
+ entry_count: int
213
+ fresh_entries: int
214
+ reused_entries: int
215
+ range_binding_sha256: str
216
+ posting_backend_coordinate: str
217
+ manifest_sha256: str
218
+ member_sha256s: tuple[tuple[str, str], ...]
219
+
220
+
221
+ @dataclass(frozen=True)
222
+ class PackedPublicationResult:
223
+ """One publication's exact durable outcome and its non-forgeable work evidence."""
224
+
225
+ status: str
226
+ head_sha256: str | None
227
+ journal_sha256: str
228
+ counters: dict[str, int]
229
+ counts: dict[str, int]
230
+ receipt: dict[str, Any]
231
+
232
+
233
+ # ------------------------------------------------------------------------------------------
234
+ # The publisher's only door to encoding
235
+ # ------------------------------------------------------------------------------------------
236
+
237
+
238
+ @dataclass
239
+ class _Counters:
240
+ """Ordered sums of seam-minted records. Nothing here is ever set from an argument."""
241
+
242
+ batch_invocations: int = 0
243
+ encoded_layer_items: int = 0
244
+ backend_scalar_invocations: int = 0
245
+ encoded_utf8_bytes: int = 0
246
+ reused_layer_items: int = 0
247
+
248
+ def absorb(self, record: BatchEncodeResult) -> None:
249
+ self.batch_invocations += record.batch_invocations
250
+ self.encoded_layer_items += len(record.vectors)
251
+ self.backend_scalar_invocations += record.scalar_invocations
252
+ self.encoded_utf8_bytes += record.encoded_utf8_bytes
253
+
254
+ def absorb_record(self, record: Mapping[str, int]) -> None:
255
+ for name in COUNTER_MEMBERS:
256
+ setattr(self, name, getattr(self, name) + record[name])
257
+
258
+ def to_dict(self) -> dict[str, int]:
259
+ return {name: getattr(self, name) for name in COUNTER_MEMBERS}
260
+
261
+
262
+ class _CountingSeam:
263
+ """One backend behind one counted door.
264
+
265
+ Every vector this publisher writes for a fresh entry comes through here, and every invocation
266
+ leaves its own seam-minted record behind. There is deliberately no other method: a second
267
+ encoding path would be a second accounting path.
268
+ """
269
+
270
+ def __init__(self, backend: EmbeddingBackend, counters: _Counters) -> None:
271
+ self._backend = backend
272
+ self._counters = counters
273
+ self.coordinate = backend.descriptor.coordinate
274
+ self.dimensions = backend.descriptor.dimensions
275
+
276
+ def encode_many(self, texts: Sequence[str]) -> tuple[tuple[int, ...], ...]:
277
+ vectors, record = encode_layer_batch(
278
+ self._backend,
279
+ texts,
280
+ entry_count=len(texts),
281
+ dimensions=self.dimensions,
282
+ )
283
+ if record.backend_coordinate != self.coordinate:
284
+ raise _refuse(
285
+ "PACKED_COUNTER_MISMATCH",
286
+ "packed.encode.record",
287
+ "a counter record names another backend than the one that was asked",
288
+ )
289
+ if record.adapter_mode != BOUNDED_SCALAR_ADAPTER:
290
+ raise _refuse(
291
+ "PACKED_COUNTER_ADAPTER",
292
+ "packed.encode.record.adapter_mode",
293
+ f"the only encoding adapter is {BOUNDED_SCALAR_ADAPTER!r}",
294
+ )
295
+ self._counters.absorb(record)
296
+ return vectors
297
+
298
+
299
+ # ------------------------------------------------------------------------------------------
300
+ # Bounded inputs
301
+ # ------------------------------------------------------------------------------------------
302
+
303
+
304
+ def _read_bounded(path: Path, *, maximum: int) -> bytes | None:
305
+ try:
306
+ return read_bounded_path(path, maximum=maximum, missing_ok=True)
307
+ except BoundedReadFailure as error:
308
+ raise _refuse(
309
+ "PACKED_INPUT_PATH",
310
+ "packed.input",
311
+ "input members must be stable bounded single-link regular files",
312
+ ) from error
313
+
314
+
315
+ def _read_bounded_at(parent_descriptor: int, name: str, *, maximum: int) -> bytes | None:
316
+ try:
317
+ return read_bounded_at(parent_descriptor, name, maximum=maximum, missing_ok=True)
318
+ except BoundedReadFailure as error:
319
+ raise _refuse(
320
+ "PACKED_INPUT_PATH",
321
+ "packed.input",
322
+ "input members must be stable bounded single-link regular files",
323
+ ) from error
324
+
325
+
326
+ @contextlib.contextmanager
327
+ def _opened_directory_at(parent_descriptor: int, name: str) -> Iterator[int]:
328
+ descriptor = -1
329
+ try:
330
+ descriptor = os.open(name, _OPEN_DIRECTORY, dir_fd=parent_descriptor)
331
+ info = os.fstat(descriptor)
332
+ if not stat.S_ISDIR(info.st_mode):
333
+ raise OSError
334
+ except OSError as error:
335
+ if descriptor >= 0:
336
+ os.close(descriptor)
337
+ raise _refuse(
338
+ "PACKED_INPUT_PATH", "packed.input", "input directory is unreadable"
339
+ ) from error
340
+ try:
341
+ yield descriptor
342
+ finally:
343
+ os.close(descriptor)
344
+
345
+
346
+ def _read_manifest(
347
+ path: Path,
348
+ *,
349
+ expected_sha256: str,
350
+ maximum: int,
351
+ schema: str,
352
+ label: str,
353
+ parent_descriptor: int | None = None,
354
+ name: str | None = None,
355
+ ) -> dict[str, Any]:
356
+ raw = (
357
+ _read_bounded(path, maximum=maximum)
358
+ if parent_descriptor is None
359
+ else _read_bounded_at(parent_descriptor, name or path.name, maximum=maximum)
360
+ )
361
+ if raw is None:
362
+ raise _refuse("PACKED_INPUT_MANIFEST", label, "no readable manifest at this root")
363
+ if sha256_bytes(raw) != expected_sha256:
364
+ raise _refuse(
365
+ "PACKED_INPUT_DIGEST", label, "the retained generation is not the caller's exact one"
366
+ )
367
+ try:
368
+ manifest = parse_canonical_json(raw)
369
+ except CanonicalJSONError as error:
370
+ raise _refuse("PACKED_INPUT_MANIFEST", label, "manifest is not canonical") from error
371
+ if (
372
+ not isinstance(manifest, dict)
373
+ or manifest.get("schema_version") != schema
374
+ or not _is_digest(manifest.get("root_sha256"))
375
+ ):
376
+ raise _refuse("PACKED_INPUT_MANIFEST", label, "manifest contract differs")
377
+ body = {key: value for key, value in manifest.items() if key != "root_sha256"}
378
+ if canonical_sha256(body) != manifest["root_sha256"]:
379
+ raise _refuse("PACKED_INPUT_MANIFEST", label, "manifest root digest differs")
380
+ return manifest
381
+
382
+
383
+ class _ClassificationStream:
384
+ """The three publishable classification streams, merged into one ascending record stream."""
385
+
386
+ def __init__(
387
+ self,
388
+ root: Path,
389
+ manifest: Mapping[str, Any],
390
+ *,
391
+ shards_descriptor: int | None = None,
392
+ ) -> None:
393
+ self._root = Path(root) / DELTA_SHARDS_DIRNAME
394
+ self._shards_descriptor = shards_descriptor
395
+ shards = manifest["outputs"]["shards"]
396
+ if not isinstance(shards, list):
397
+ raise _refuse(
398
+ "PACKED_INPUT_MANIFEST", "delta.manifest.outputs.shards", "shard list differs"
399
+ )
400
+ self._descriptors: dict[str, list[Mapping[str, Any]]] = {
401
+ name: [] for name in PUBLISHABLE_CLASSIFICATIONS
402
+ }
403
+ for descriptor in shards:
404
+ if (
405
+ not isinstance(descriptor, dict)
406
+ or not _is_digest(descriptor.get("sha256"))
407
+ or type(descriptor.get("bytes")) is not int
408
+ or not 1 <= descriptor["bytes"] <= MAX_DELTA_SHARD_BYTES
409
+ or type(descriptor.get("item_count")) is not int
410
+ or descriptor["item_count"] < 1
411
+ or not isinstance(descriptor.get("classification"), str)
412
+ ):
413
+ raise _refuse(
414
+ "PACKED_INPUT_MANIFEST",
415
+ "delta.manifest.outputs.shards[]",
416
+ "shard descriptor is invalid",
417
+ )
418
+ if descriptor["classification"] in self._descriptors:
419
+ self._descriptors[descriptor["classification"]].append(descriptor)
420
+ self._iterators = {name: self._items(name) for name in PUBLISHABLE_CLASSIFICATIONS}
421
+ self._peeked: dict[str, dict[str, Any] | None] = dict.fromkeys(PUBLISHABLE_CLASSIFICATIONS)
422
+ self._exhausted: dict[str, bool] = dict.fromkeys(PUBLISHABLE_CLASSIFICATIONS, False)
423
+ self._loaded: dict[str, bool] = dict.fromkeys(PUBLISHABLE_CLASSIFICATIONS, False)
424
+ self.max_resident_packs = 0
425
+
426
+ def _observe_residency(self) -> None:
427
+ resident = sum(1 for value in self._loaded.values() if value)
428
+ self.max_resident_packs = max(self.max_resident_packs, resident)
429
+
430
+ def _items(self, classification: str) -> Iterator[dict[str, Any]]:
431
+ last: bytes | None = None
432
+ for descriptor in self._descriptors[classification]:
433
+ name = f"{descriptor['sha256']}.json"
434
+ raw = (
435
+ _read_bounded(
436
+ self._root / name,
437
+ maximum=MAX_DELTA_SHARD_BYTES,
438
+ )
439
+ if self._shards_descriptor is None
440
+ else _read_bounded_at(
441
+ self._shards_descriptor,
442
+ name,
443
+ maximum=MAX_DELTA_SHARD_BYTES,
444
+ )
445
+ )
446
+ if (
447
+ raw is None
448
+ or len(raw) != descriptor["bytes"]
449
+ or sha256_bytes(raw) != descriptor["sha256"]
450
+ ):
451
+ raise _refuse(
452
+ "PACKED_INPUT_SHARD", "delta.shard", "a committed delta shard differs"
453
+ )
454
+ try:
455
+ payload = parse_canonical_json(raw)
456
+ except CanonicalJSONError as error:
457
+ raise _refuse(
458
+ "PACKED_INPUT_SHARD", "delta.shard", "shard is not canonical"
459
+ ) from error
460
+ if (
461
+ not isinstance(payload, dict)
462
+ or not isinstance(payload.get("items"), list)
463
+ or len(payload["items"]) != descriptor["item_count"]
464
+ or payload.get("classification") != classification
465
+ ):
466
+ raise _refuse("PACKED_INPUT_SHARD", "delta.shard", "shard contract differs")
467
+ self._loaded[classification] = True
468
+ self._observe_residency()
469
+ for item in payload["items"]:
470
+ if (
471
+ not isinstance(item, dict)
472
+ or not isinstance(item.get("provider_record_id"), str)
473
+ or not isinstance(item.get("entry_id"), str)
474
+ or item.get("classification") != classification
475
+ ):
476
+ raise _refuse(
477
+ "PACKED_INPUT_SHARD", "delta.shard.items[]", "a delta item contract differs"
478
+ )
479
+ key = item["provider_record_id"].encode("utf-8")
480
+ if last is not None and key <= last:
481
+ raise _refuse(
482
+ "PACKED_INPUT_ORDER",
483
+ "delta.shards",
484
+ "a classification stream must ascend by provider record",
485
+ )
486
+ last = key
487
+ yield item
488
+ self._loaded[classification] = False
489
+ self._observe_residency()
490
+
491
+ def _peek(self, classification: str) -> dict[str, Any] | None:
492
+ if self._peeked[classification] is None and not self._exhausted[classification]:
493
+ self._peeked[classification] = next(self._iterators[classification], None)
494
+ if self._peeked[classification] is None:
495
+ self._exhausted[classification] = True
496
+ return self._peeked[classification]
497
+
498
+ def __iter__(self) -> Iterator[dict[str, Any]]:
499
+ last: bytes | None = None
500
+ while True:
501
+ live = {
502
+ name: item
503
+ for name in PUBLISHABLE_CLASSIFICATIONS
504
+ if (item := self._peek(name)) is not None
505
+ }
506
+ if not live:
507
+ return
508
+ key = min(item["provider_record_id"] for item in live.values())
509
+ owners = [name for name, item in live.items() if item["provider_record_id"] == key]
510
+ if len(owners) != 1:
511
+ raise _refuse(
512
+ "PACKED_INPUT_ORDER",
513
+ "delta.shards",
514
+ "one provider record may hold only one publishable classification",
515
+ )
516
+ encoded = key.encode("utf-8")
517
+ if last is not None and encoded <= last:
518
+ raise _refuse(
519
+ "PACKED_INPUT_ORDER", "delta.shards", "the merged stream must ascend by key"
520
+ )
521
+ last = encoded
522
+ item = self._peeked[owners[0]]
523
+ self._peeked[owners[0]] = None
524
+ assert item is not None
525
+ yield item
526
+
527
+
528
+ class _AuthoredStream:
529
+ """One authored generation as one ascending record stream, one shard resident."""
530
+
531
+ def __init__(self, root: Path, manifest: Mapping[str, Any]) -> None:
532
+ self._root = Path(root)
533
+ self._manifest = manifest
534
+ if not isinstance(manifest.get("shards"), list):
535
+ raise _refuse(
536
+ "PACKED_INPUT_MANIFEST", "authoring.manifest.shards", "shard list differs"
537
+ )
538
+ self._iterator = self._records()
539
+ self._peeked: dict[str, Any] | None = None
540
+ self._exhausted = False
541
+ self.max_resident_shards = 0
542
+
543
+ def _records(self) -> Iterator[dict[str, Any]]:
544
+ last: bytes | None = None
545
+ for descriptor in self._manifest["shards"]:
546
+ if not shard_descriptor_is_valid(descriptor):
547
+ raise _refuse(
548
+ "PACKED_INPUT_MANIFEST",
549
+ "authoring.manifest.shards[]",
550
+ "shard descriptor is invalid",
551
+ )
552
+ raw = _read_bounded(
553
+ self._root / SHARDS_DIRNAME / f"{descriptor['sha256']}.json",
554
+ maximum=MAX_AUTHORING_SHARD_BYTES,
555
+ )
556
+ if (
557
+ raw is None
558
+ or len(raw) != descriptor["bytes"]
559
+ or sha256_bytes(raw) != descriptor["sha256"]
560
+ ):
561
+ raise _refuse(
562
+ "PACKED_INPUT_SHARD", "authoring.shard", "a committed authored shard differs"
563
+ )
564
+ try:
565
+ payload = parse_canonical_json(raw)
566
+ except CanonicalJSONError as error:
567
+ raise _refuse(
568
+ "PACKED_INPUT_SHARD", "authoring.shard", "shard is not canonical"
569
+ ) from error
570
+ if (
571
+ not isinstance(payload, dict)
572
+ or payload.get("schema_version") != AUTHORING_SHARD_SCHEMA
573
+ or payload.get("policy_sha256") != self._manifest.get("policy_sha256")
574
+ or not isinstance(payload.get("items"), list)
575
+ or len(payload["items"]) != descriptor["item_count"]
576
+ or payload.get("first_provider_record_id") != descriptor["first_provider_record_id"]
577
+ or payload.get("last_provider_record_id") != descriptor["last_provider_record_id"]
578
+ or not payload["items"]
579
+ or not isinstance(payload["items"][0], dict)
580
+ or not isinstance(payload["items"][-1], dict)
581
+ or payload["items"][0].get("provider_record_id")
582
+ != descriptor["first_provider_record_id"]
583
+ or payload["items"][-1].get("provider_record_id")
584
+ != descriptor["last_provider_record_id"]
585
+ or sum(
586
+ 1
587
+ for item in payload["items"]
588
+ if isinstance(item, dict) and item.get("entry") is not None
589
+ )
590
+ != descriptor["entry_count"]
591
+ ):
592
+ raise _refuse("PACKED_INPUT_SHARD", "authoring.shard", "shard contract differs")
593
+ self.max_resident_shards = max(self.max_resident_shards, 1)
594
+ for item in payload["items"]:
595
+ if not isinstance(item, dict) or not isinstance(
596
+ item.get("provider_record_id"), str
597
+ ):
598
+ raise _refuse(
599
+ "PACKED_INPUT_SHARD",
600
+ "authoring.shard.items[]",
601
+ "an authored item contract differs",
602
+ )
603
+ key = item["provider_record_id"].encode("utf-8")
604
+ if last is not None and key <= last:
605
+ raise _refuse(
606
+ "PACKED_INPUT_ORDER",
607
+ "authoring.shards",
608
+ "an authored generation must ascend by provider record",
609
+ )
610
+ last = key
611
+ yield item
612
+
613
+ def advance_to(self, record_id: str) -> dict[str, Any]:
614
+ while True:
615
+ if self._peeked is None and not self._exhausted:
616
+ self._peeked = next(self._iterator, None)
617
+ if self._peeked is None:
618
+ self._exhausted = True
619
+ if self._peeked is None:
620
+ raise _refuse(
621
+ "PACKED_INPUT_AUTHORING",
622
+ "authoring.shards",
623
+ "a classified record has no authored entry in this generation",
624
+ )
625
+ found = self._peeked["provider_record_id"]
626
+ if found == record_id:
627
+ item = self._peeked
628
+ self._peeked = None
629
+ return item
630
+ if found.encode("utf-8") > record_id.encode("utf-8"):
631
+ raise _refuse(
632
+ "PACKED_INPUT_AUTHORING",
633
+ "authoring.shards",
634
+ "a classified record has no authored entry in this generation",
635
+ )
636
+ self._peeked = None
637
+
638
+
639
+ class _HistoryStream:
640
+ """The successor identity generation as one ascending stream, one pack resident."""
641
+
642
+ def __init__(self, store: IdentityHistoryStore) -> None:
643
+ self._store = store
644
+ self._iterator = store.stream()
645
+ self._peeked: IdentityHistory | None = None
646
+ self._exhausted = False
647
+ self.max_resident_packs = 0
648
+
649
+ def advance_to(self, record_id: str) -> IdentityHistory:
650
+ while True:
651
+ if self._peeked is None and not self._exhausted:
652
+ self._peeked = next(self._iterator, None)
653
+ self.max_resident_packs = max(self.max_resident_packs, self._store.resident_packs)
654
+ if self._peeked is None:
655
+ self._exhausted = True
656
+ if self._peeked is None:
657
+ raise _refuse(
658
+ "PACKED_INPUT_IDENTITY",
659
+ "identity.records",
660
+ "a classified record has no identity history in this generation",
661
+ )
662
+ found = self._peeked.identity.provider_record_id
663
+ if found == record_id:
664
+ history = self._peeked
665
+ self._peeked = None
666
+ return history
667
+ if found.encode("utf-8") > record_id.encode("utf-8"):
668
+ raise _refuse(
669
+ "PACKED_INPUT_IDENTITY",
670
+ "identity.records",
671
+ "a classified record has no identity history in this generation",
672
+ )
673
+ self._peeked = None
674
+
675
+
676
+ class _PredecessorPacks:
677
+ """One predecessor packed generation, read one range at a time in ascending key order."""
678
+
679
+ def __init__(self, directory: PackedDirectory, head: Mapping[str, Any]) -> None:
680
+ self._directory = directory
681
+ self._descriptors = [
682
+ range_descriptor_from_dict(value, path=f"predecessor.head.ranges[{index}]")
683
+ for index, value in enumerate(head["ranges"])
684
+ ]
685
+ self._backends = {item["coordinate"]: item["dimensions"] for item in head["backends"]}
686
+ self._position = 0
687
+ self._resident: int | None = None
688
+ self._ordinals: dict[str, int] = {}
689
+ self._vectors: dict[str, dict[str, tuple[tuple[int, ...], ...]]] = {}
690
+ self.max_resident_ranges = 0
691
+
692
+ def require_backends(self, coordinates: Sequence[str]) -> None:
693
+ missing = [name for name in coordinates if name not in self._backends]
694
+ if missing:
695
+ raise _refuse(
696
+ "PACKED_REUSE_BACKEND",
697
+ "predecessor.backends",
698
+ "the predecessor generation has no packs for a backend this publication reuses",
699
+ )
700
+
701
+ def lookup(self, key: str) -> dict[str, dict[str, tuple[int, ...]]]:
702
+ encoded = key.encode("utf-8")
703
+ while self._position < len(self._descriptors):
704
+ if encoded <= self._descriptors[self._position].last_key.encode("utf-8"):
705
+ break
706
+ self._position += 1
707
+ if self._position >= len(self._descriptors):
708
+ raise _refuse(
709
+ "PACKED_REUSE_MISSING",
710
+ "predecessor.ranges",
711
+ "an unchanged entry has no predecessor pack to reuse",
712
+ )
713
+ if self._resident != self._position:
714
+ self._load(self._position)
715
+ ordinal = self._ordinals.get(key)
716
+ if ordinal is None:
717
+ raise _refuse(
718
+ "PACKED_REUSE_MISSING",
719
+ "predecessor.ranges",
720
+ "an unchanged entry has no predecessor pack to reuse",
721
+ )
722
+ return {
723
+ coordinate: {layer: vectors[ordinal] for layer, vectors in layers.items()}
724
+ for coordinate, layers in self._vectors.items()
725
+ }
726
+
727
+ def _load(self, position: int) -> None:
728
+ descriptor = self._descriptors[position]
729
+ raw = self._directory.read_member(
730
+ descriptor.manifest_sha256, kind="range", maximum=MAX_RANGE_MANIFEST_BYTES
731
+ )
732
+ if raw is None or sha256_bytes(raw) != descriptor.manifest_sha256:
733
+ raise _refuse(
734
+ "PACKED_REUSE_MEMBER",
735
+ "predecessor.range.manifest",
736
+ "a predecessor range manifest is not the manifest its head binds",
737
+ )
738
+ manifest = parse_canonical_json(raw)
739
+ members = [
740
+ member_descriptor_from_dict(value, path="predecessor.range.members[]")
741
+ for value in manifest["members"]
742
+ ]
743
+ facts = self._member(_only(members, "facts"), maximum=MAX_FACTS_MEMBER_BYTES)
744
+ payload = parse_canonical_json(facts)
745
+ self._ordinals = {item["key"]: ordinal for ordinal, item in enumerate(payload["entries"])}
746
+ vectors: dict[str, dict[str, tuple[tuple[int, ...], ...]]] = {}
747
+ for member in members:
748
+ if member.kind != "vector":
749
+ continue
750
+ raw_member = self._member(member, maximum=MAX_VECTOR_MEMBER_BYTES)
751
+ parsed = parse_vector_member(raw_member, descriptor=member)
752
+ vectors.setdefault(member.backend_coordinate or "", {})[parsed.layer] = parsed.vectors
753
+ self._vectors = vectors
754
+ self._resident = position
755
+ self.max_resident_ranges = max(self.max_resident_ranges, 1)
756
+
757
+ def _member(self, descriptor: Any, *, maximum: int) -> bytes:
758
+ raw = self._directory.read_member(descriptor.sha256, kind=descriptor.kind, maximum=maximum)
759
+ if raw is None or len(raw) != descriptor.bytes or sha256_bytes(raw) != descriptor.sha256:
760
+ raise _refuse(
761
+ "PACKED_REUSE_MEMBER",
762
+ "predecessor.range.members[]",
763
+ "a predecessor member is not the member its descriptor binds",
764
+ )
765
+ return raw
766
+
767
+
768
+ def _only(descriptors: Sequence[Any], kind: str) -> Any:
769
+ found = [descriptor for descriptor in descriptors if descriptor.kind == kind]
770
+ if len(found) != 1:
771
+ raise _refuse(
772
+ "PACKED_REUSE_MEMBER",
773
+ "predecessor.range.members[]",
774
+ f"a predecessor range carries exactly one {kind} member",
775
+ )
776
+ return found[0]
777
+
778
+
779
+ # ------------------------------------------------------------------------------------------
780
+ # The work journal
781
+ # ------------------------------------------------------------------------------------------
782
+
783
+
784
+ #: Exactly what one completed-range record carries. A resume reads nothing else from it, and a
785
+ #: record carrying anything else is a record this publisher did not write.
786
+ _COMPLETED_RANGE_MEMBERS = frozenset(
787
+ {
788
+ "range_index",
789
+ "first_key",
790
+ "last_key",
791
+ "entry_count",
792
+ "facts_sha256",
793
+ "range_binding_sha256",
794
+ "range_manifest_sha256",
795
+ "range_manifest_bytes",
796
+ "member_sha256s",
797
+ "vector_payload_sha256s",
798
+ }
799
+ )
800
+
801
+
802
+ @dataclass
803
+ class _Journal:
804
+ """The typed work journal an interruption leaves behind and a resume validates."""
805
+
806
+ output_root: str
807
+ generation: dict[str, Any]
808
+ completed_ranges: list[dict[str, Any]] = field(default_factory=list)
809
+ invocations: list[dict[str, Any]] = field(default_factory=list)
810
+
811
+ def body(self, *, status: str, head_sha256: str | None) -> dict[str, Any]:
812
+ return {
813
+ "schema_version": PACKED_JOURNAL_SCHEMA,
814
+ "status": status,
815
+ "output_root": self.output_root,
816
+ "generation": self.generation,
817
+ "completed_ranges": list(self.completed_ranges),
818
+ "invocations": list(self.invocations),
819
+ "head_sha256": head_sha256,
820
+ }
821
+
822
+ def document(self, *, status: str, head_sha256: str | None) -> dict[str, Any]:
823
+ body = self.body(status=status, head_sha256=head_sha256)
824
+ return {**body, "root_sha256": canonical_sha256(body)}
825
+
826
+
827
+ def _read_journal(directory: PackedDirectory) -> dict[str, Any] | None:
828
+ raw = directory.read(PACKED_JOURNAL_FILENAME, maximum=MAX_PACKED_JOURNAL_BYTES)
829
+ if raw is None:
830
+ return None
831
+ try:
832
+ journal = parse_canonical_json(raw)
833
+ except CanonicalJSONError as error:
834
+ raise _refuse(
835
+ "PACKED_JOURNAL_DIGEST", "packed.journal", "the work journal is not canonical"
836
+ ) from error
837
+ if (
838
+ not isinstance(journal, dict)
839
+ or journal.get("schema_version") != PACKED_JOURNAL_SCHEMA
840
+ or not _is_digest(journal.get("root_sha256"))
841
+ or not isinstance(journal.get("completed_ranges"), list)
842
+ or not isinstance(journal.get("invocations"), list)
843
+ or not isinstance(journal.get("generation"), dict)
844
+ ):
845
+ raise _refuse(
846
+ "PACKED_JOURNAL_DIGEST", "packed.journal", "the work journal contract differs"
847
+ )
848
+ body = {key: value for key, value in journal.items() if key != "root_sha256"}
849
+ if canonical_sha256(body) != journal["root_sha256"]:
850
+ raise _refuse(
851
+ "PACKED_JOURNAL_DIGEST",
852
+ "packed.journal.root_sha256",
853
+ "the work journal does not reproduce its own digest",
854
+ )
855
+ # The journal's digest is its own, not a secret's, so a rewriter can mint a consistent one.
856
+ # Every record it carries is therefore re-derived rather than believed: the counter records
857
+ # below only survive because the equations at the end of a publication must still hold.
858
+ for position, record in enumerate(journal["invocations"]):
859
+ if not isinstance(record, dict) or set(record) != set(COUNTER_MEMBERS):
860
+ raise _refuse(
861
+ "PACKED_JOURNAL_COUNTER",
862
+ f"packed.journal.invocations[{position}]",
863
+ f"one invocation record carries exactly {list(COUNTER_MEMBERS)}",
864
+ )
865
+ for name in COUNTER_MEMBERS:
866
+ if type(record[name]) is not int or record[name] < 0:
867
+ raise _refuse(
868
+ "PACKED_JOURNAL_COUNTER",
869
+ f"packed.journal.invocations[{position}].{name}",
870
+ "a work counter is a non-negative integer",
871
+ )
872
+ return journal
873
+
874
+
875
+ def _resume_from(
876
+ directory: PackedDirectory,
877
+ journal: dict[str, Any],
878
+ *,
879
+ output_root: Path,
880
+ generation: Mapping[str, Any],
881
+ ) -> _Journal:
882
+ if journal["status"] != "incomplete" or journal["head_sha256"] is not None:
883
+ raise _refuse(
884
+ "PACKED_JOURNAL_STATUS",
885
+ "packed.journal.status",
886
+ "only a typed-incomplete journal may be resumed",
887
+ )
888
+ if journal["output_root"] != os.path.abspath(os.fspath(output_root)):
889
+ raise _refuse(
890
+ "PACKED_JOURNAL_ROOT",
891
+ "packed.journal.output_root",
892
+ "the journal was written for another output root",
893
+ )
894
+ if journal["generation"] != dict(generation):
895
+ raise _refuse(
896
+ "PACKED_JOURNAL_GENERATION",
897
+ "packed.journal.generation",
898
+ "the journal was written for another generation coordinate",
899
+ )
900
+ for position, completed in enumerate(journal["completed_ranges"]):
901
+ if (
902
+ not isinstance(completed, dict)
903
+ or set(completed) != _COMPLETED_RANGE_MEMBERS
904
+ or completed["range_index"] != position
905
+ or not _is_digest(completed["range_manifest_sha256"])
906
+ or not _is_digest(completed["facts_sha256"])
907
+ or not _is_digest(completed["range_binding_sha256"])
908
+ or not isinstance(completed["member_sha256s"], list)
909
+ or not isinstance(completed["vector_payload_sha256s"], list)
910
+ or type(completed["entry_count"]) is not int
911
+ ):
912
+ raise _refuse(
913
+ "PACKED_JOURNAL_RANGE",
914
+ f"packed.journal.completed_ranges[{position}]",
915
+ "a completed range record contract differs",
916
+ )
917
+ payloads: list[str] = []
918
+ for entry in [
919
+ ["range", completed["range_manifest_sha256"]],
920
+ *completed["member_sha256s"],
921
+ ]:
922
+ if not isinstance(entry, list) or len(entry) != 2:
923
+ raise _refuse(
924
+ "PACKED_JOURNAL_RANGE",
925
+ f"packed.journal.completed_ranges[{position}].member_sha256s[]",
926
+ "a completed member record is exactly one kind and one digest",
927
+ )
928
+ kind, sha256 = entry
929
+ if kind not in MEMBER_MAXIMUM_BYTES or not _is_digest(sha256):
930
+ raise _refuse(
931
+ "PACKED_JOURNAL_RANGE",
932
+ f"packed.journal.completed_ranges[{position}].member_sha256s[]",
933
+ "a completed member record names a known kind and an exact digest",
934
+ )
935
+ raw = directory.read_member(sha256, kind=kind, maximum=MEMBER_MAXIMUM_BYTES[kind])
936
+ if raw is None or sha256_bytes(raw) != sha256:
937
+ raise _refuse(
938
+ "PACKED_JOURNAL_MEMBER",
939
+ f"packed.journal.completed_ranges[{position}]",
940
+ "an already-emitted immutable member does not read back exactly",
941
+ )
942
+ if kind == "vector":
943
+ payloads.append(sha256_bytes(raw[MEMBER_HEADER_BYTES:]))
944
+ # The receipt states these, so they are recomputed from the member bytes this loop just
945
+ # read rather than carried over from a journal that a rewriter could have restated.
946
+ if payloads != list(completed["vector_payload_sha256s"]):
947
+ raise _refuse(
948
+ "PACKED_JOURNAL_MEMBER",
949
+ f"packed.journal.completed_ranges[{position}].vector_payload_sha256s",
950
+ "a completed range's vector payload digests are not the digests of its own packs",
951
+ )
952
+ return _Journal(
953
+ output_root=journal["output_root"],
954
+ generation=dict(journal["generation"]),
955
+ completed_ranges=list(journal["completed_ranges"]),
956
+ invocations=list(journal["invocations"]),
957
+ )
958
+
959
+
960
+ def _verify_recovery_ranges(
961
+ directory: PackedDirectory,
962
+ installed: VerifiedPackedGeneration,
963
+ *,
964
+ completed_ranges: Sequence[Mapping[str, Any]],
965
+ delta_root: Path,
966
+ delta_manifest: Mapping[str, Any],
967
+ delta_shards_descriptor: int | None,
968
+ identity_descriptor: int | None,
969
+ authoring_root: Path,
970
+ authoring_manifest: Mapping[str, Any],
971
+ identity_manifest_sha256: str,
972
+ packed_backends: Sequence[PackedBackend],
973
+ seam_backends: Sequence[EmbeddingBackend],
974
+ posting_backend: PackedBackend,
975
+ predecessor: _PredecessorPacks | None,
976
+ limits: PackedLimits,
977
+ deadline: float,
978
+ ) -> tuple[dict[str, int], dict[str, int]]:
979
+ """Replay the exact retained generation and bind every installed range to those inputs."""
980
+
981
+ classification = _ClassificationStream(
982
+ delta_root, delta_manifest, shards_descriptor=delta_shards_descriptor
983
+ )
984
+ authored = _AuthoredStream(authoring_root, authoring_manifest)
985
+ identity = _HistoryStream(
986
+ IdentityHistoryStore.open(
987
+ delta_root / DELTA_IDENTITY_DIRNAME,
988
+ expected_manifest_sha256=identity_manifest_sha256,
989
+ root_descriptor=identity_descriptor,
990
+ )
991
+ )
992
+ counters = _Counters()
993
+ seams = {
994
+ backend.descriptor.coordinate: _CountingSeam(backend, counters) for backend in seam_backends
995
+ }
996
+ counts = {
997
+ "ranges": 0,
998
+ "entries": 0,
999
+ "fresh_ranges": 0,
1000
+ "new": 0,
1001
+ "changed": 0,
1002
+ "unchanged": 0,
1003
+ "members": 0,
1004
+ }
1005
+ for range_index, entries in enumerate(_ranges(classification, authored, identity)):
1006
+ if range_index >= limits.max_ranges:
1007
+ raise _refuse(
1008
+ "PACKED_RANGE_LIMIT",
1009
+ "packed.terminal",
1010
+ "terminal recovery exceeds the declared range bound",
1011
+ )
1012
+ if time.monotonic() >= deadline:
1013
+ raise _refuse(
1014
+ "PACKED_WALL_LIMIT",
1015
+ "packed.terminal",
1016
+ "terminal recovery exceeded its declared wall bound",
1017
+ )
1018
+ if range_index >= len(completed_ranges) or range_index >= len(installed.head["ranges"]):
1019
+ raise _refuse(
1020
+ "PACKED_TERMINAL_RANGE",
1021
+ "packed.terminal.journal.completed_ranges",
1022
+ "the terminal record omits a range produced by the retained inputs",
1023
+ )
1024
+ range_input = PackedRangeInput(
1025
+ range_index=range_index, entries=tuple(item.to_entry() for item in entries)
1026
+ )
1027
+ completed = completed_ranges[range_index]
1028
+ _verify_completed_range(directory, completed, range_input)
1029
+
1030
+ fresh = tuple(
1031
+ position
1032
+ for position, item in enumerate(entries)
1033
+ if item.classification in FRESH_CLASSIFICATIONS
1034
+ )
1035
+ fresh_positions = set(fresh)
1036
+ reused = tuple(
1037
+ position for position in range(len(entries)) if position not in fresh_positions
1038
+ )
1039
+ if reused and predecessor is None:
1040
+ raise _refuse(
1041
+ "PACKED_PREDECESSOR_REQUIRED",
1042
+ "predecessor",
1043
+ "an unchanged entry can only be verified against the packs it reuses",
1044
+ )
1045
+ recovered: dict[int, dict[str, dict[str, tuple[int, ...]]]] = {}
1046
+ for position in reused:
1047
+ assert predecessor is not None
1048
+ recovered[position] = predecessor.lookup(entries[position].key)
1049
+
1050
+ # All terminal files are locally rewriteable. For a fresh entry no retained artifact other
1051
+ # than a deterministic backend replay can prove that a vector came from its exact layer
1052
+ # text, so recovery performs that replay before it mutates the journal.
1053
+ vectors: dict[str, dict[str, tuple[tuple[int, ...], ...]]] = {}
1054
+ for backend in packed_backends:
1055
+ seam = seams[backend.coordinate]
1056
+ layers: dict[str, tuple[tuple[int, ...], ...]] = {}
1057
+ for layer in EMBEDDING_LAYERS:
1058
+ column: list[tuple[int, ...] | None] = [None] * len(entries)
1059
+ if fresh:
1060
+ encoded = seam.encode_many(
1061
+ tuple(entries[position].layer_texts[layer] for position in fresh)
1062
+ )
1063
+ for position, vector in zip(fresh, encoded, strict=True):
1064
+ column[position] = vector
1065
+ for position in reused:
1066
+ vector = recovered[position].get(backend.coordinate, {}).get(layer)
1067
+ if vector is None or len(vector) != backend.dimensions:
1068
+ raise _refuse(
1069
+ "PACKED_REUSE_MISSING",
1070
+ "predecessor.ranges",
1071
+ "an unchanged entry has no reusable layer vector for this backend",
1072
+ )
1073
+ column[position] = vector
1074
+ layers[layer] = tuple(value for value in column if value is not None)
1075
+ if len(layers[layer]) != len(entries): # pragma: no cover - filled above
1076
+ raise _refuse(
1077
+ "PACKED_RANGE_ENTRY",
1078
+ "packed.range.vectors",
1079
+ "a layer member does not cover every entry in its range",
1080
+ )
1081
+ vectors[backend.coordinate] = layers
1082
+
1083
+ built = build_packed_range(
1084
+ range_input,
1085
+ vectors,
1086
+ backends=packed_backends,
1087
+ posting_backend_coordinate=posting_backend.coordinate,
1088
+ max_terms_per_segment=limits.max_terms_per_segment,
1089
+ )
1090
+ expected_completed = _completed_range_record(built)
1091
+ if dict(completed) != expected_completed or installed.head["ranges"][range_index] != (
1092
+ built.descriptor.to_dict()
1093
+ ):
1094
+ raise _refuse(
1095
+ "PACKED_TERMINAL_RANGE",
1096
+ f"packed.terminal.journal.completed_ranges[{range_index}]",
1097
+ "an installed range is not the exact range reproduced from the retained inputs",
1098
+ )
1099
+ counts["ranges"] += 1
1100
+ counts["entries"] += len(entries)
1101
+ counts["fresh_ranges"] += 1 if fresh else 0
1102
+ counts["members"] += len(built.members)
1103
+ counters.reused_layer_items += len(reused) * len(EMBEDDING_LAYERS) * len(packed_backends)
1104
+ for item in entries:
1105
+ counts[item.classification] += 1
1106
+
1107
+ if len(completed_ranges) != counts["ranges"] or counts["entries"] > limits.max_entries:
1108
+ raise _refuse(
1109
+ "PACKED_TERMINAL_RANGE",
1110
+ "packed.terminal.journal.completed_ranges",
1111
+ "the terminal range list differs from the retained generation",
1112
+ )
1113
+ if (
1114
+ counts["ranges"] != installed.range_count
1115
+ or counts["entries"] != installed.entry_count
1116
+ or counts["members"] != installed.member_count
1117
+ or counts["fresh_ranges"] != installed.fresh_range_count
1118
+ or {name: counts[name] for name in PUBLISHABLE_CLASSIFICATIONS}
1119
+ != dict(installed.classification_counts)
1120
+ ):
1121
+ raise _refuse(
1122
+ "PACKED_TERMINAL_COUNTER",
1123
+ "packed.terminal.receipt.counts",
1124
+ "terminal recovery inputs do not reproduce the installed generation",
1125
+ )
1126
+ _verify_counter_equations(counters, counts, backend_count=len(packed_backends))
1127
+ return counts, counters.to_dict()
1128
+
1129
+
1130
+ def _canonical_residency(
1131
+ counts: Mapping[str, int], generation: Mapping[str, Any]
1132
+ ) -> dict[str, int]:
1133
+ """State deterministic streaming bounds, not unauthenticatable process observations."""
1134
+
1135
+ return {
1136
+ "max_resident_classification_packs": sum(
1137
+ 1 for name in PUBLISHABLE_CLASSIFICATIONS if counts[name] > 0
1138
+ ),
1139
+ "max_resident_authoring_shards": 1,
1140
+ "max_resident_identity_packs": 1,
1141
+ "max_resident_output_ranges": 1,
1142
+ "max_resident_predecessor_ranges": (
1143
+ 0 if generation["predecessor_head_sha256"] is None else 1
1144
+ ),
1145
+ }
1146
+
1147
+
1148
+ def _verified_receipt_outputs(installed: VerifiedPackedGeneration) -> dict[str, Any]:
1149
+ """Derive every receipt output from the generation's independently verified bytes."""
1150
+
1151
+ return {
1152
+ "range_count": installed.range_count,
1153
+ "entry_count": installed.entry_count,
1154
+ "member_count": installed.member_count,
1155
+ "posting_term_count": installed.posting_term_count,
1156
+ "ranges": list(installed.head["ranges"]),
1157
+ "ranges_chain_sha256": installed.head["ranges_chain_sha256"],
1158
+ "vector_member_sha256s": list(installed.vector_payload_sha256s),
1159
+ "posting_member_sha256s": list(installed.posting_member_sha256s),
1160
+ "bound_member_sha256s": list(installed.bound_member_sha256s),
1161
+ }
1162
+
1163
+
1164
+ def _verify_recovery_evidence(
1165
+ installed: VerifiedPackedGeneration,
1166
+ *,
1167
+ terminal: Mapping[str, Any],
1168
+ journal_template: Mapping[str, Any],
1169
+ journal: Mapping[str, Any],
1170
+ expected_counts: Mapping[str, int],
1171
+ expected_counters: Mapping[str, int],
1172
+ delta_manifest: Mapping[str, Any],
1173
+ packed_backends: Sequence[PackedBackend],
1174
+ posting_backend: PackedBackend,
1175
+ limits: PackedLimits,
1176
+ ) -> None:
1177
+ """Rebuild every terminal receipt claim from caller inputs and verified member bytes."""
1178
+
1179
+ receipt = terminal.get("receipt")
1180
+ if not isinstance(receipt, Mapping):
1181
+ raise _refuse("PACKED_TERMINAL_SCHEMA", "packed.terminal.receipt", "receipt differs")
1182
+ invocations = receipt.get("invocations")
1183
+ canonical_invocations = [dict(expected_counters)]
1184
+ expected_outputs = _verified_receipt_outputs(installed)
1185
+ expected_receipt = {
1186
+ "schema_version": PACKED_RECEIPT_SCHEMA,
1187
+ "status": "complete",
1188
+ "provider_id": terminal["generation"]["provider_id"],
1189
+ "adapter_mode": BOUNDED_SCALAR_ADAPTER,
1190
+ # Rebuilt from the *installed head*, not carried over from the retained receipt, so a
1191
+ # recovery reconciles the two documents against each other rather than believing either.
1192
+ # `_parse_head` has already required the head's statement to be a coverage at all.
1193
+ "coverage": dict(installed.head["coverage"]),
1194
+ "inputs": {
1195
+ "delta_manifest_sha256": terminal["generation"]["delta_manifest_sha256"],
1196
+ "delta_root_sha256": delta_manifest["root_sha256"],
1197
+ "authoring_manifest_sha256": terminal["generation"]["authoring_manifest_sha256"],
1198
+ "identity_manifest_sha256": terminal["generation"]["identity_manifest_sha256"],
1199
+ "predecessor_head_sha256": terminal["generation"]["predecessor_head_sha256"],
1200
+ "publication_binding_sha256": terminal["generation"].get("publication_binding_sha256"),
1201
+ },
1202
+ "backends": [backend.to_dict() for backend in packed_backends],
1203
+ "posting_backend_coordinate": posting_backend.coordinate,
1204
+ "counts": dict(expected_counts),
1205
+ "counters": dict(expected_counters),
1206
+ "invocations": canonical_invocations,
1207
+ "outputs": expected_outputs,
1208
+ "residency": _canonical_residency(expected_counts, terminal["generation"]),
1209
+ "limits": limits.to_dict(),
1210
+ }
1211
+ if (
1212
+ dict(receipt) != expected_receipt
1213
+ or installed.head.get("provider_id") != terminal["generation"]["provider_id"]
1214
+ or installed.head.get("generation_sha256")
1215
+ != terminal["generation"]["delta_manifest_sha256"]
1216
+ or installed.head.get("predecessor_head_sha256")
1217
+ != terminal["generation"]["predecessor_head_sha256"]
1218
+ or installed.head.get("backends") != expected_receipt["backends"]
1219
+ or installed.head.get("posting_backend_coordinate") != posting_backend.coordinate
1220
+ or installed.head.get("counters") != dict(expected_counters)
1221
+ or invocations != canonical_invocations
1222
+ or installed.head.get("invocations") != canonical_invocations
1223
+ or journal_template.get("invocations") != canonical_invocations
1224
+ or journal.get("invocations") != canonical_invocations
1225
+ ):
1226
+ raise _refuse(
1227
+ "PACKED_TERMINAL_COUNTER",
1228
+ "packed.terminal.receipt",
1229
+ "terminal work evidence does not reproduce from the installed generation",
1230
+ )
1231
+
1232
+
1233
+ def _recover_terminal_publication(
1234
+ directory: PackedDirectory,
1235
+ *,
1236
+ output_root: Path,
1237
+ generation: Mapping[str, Any],
1238
+ head_raw: bytes,
1239
+ delta_root: Path,
1240
+ delta_manifest: Mapping[str, Any],
1241
+ delta_shards_descriptor: int | None,
1242
+ identity_descriptor: int | None,
1243
+ authoring_root: Path,
1244
+ authoring_manifest: Mapping[str, Any],
1245
+ packed_backends: Sequence[PackedBackend],
1246
+ seam_backends: Sequence[EmbeddingBackend],
1247
+ posting_backend: PackedBackend,
1248
+ predecessor: _PredecessorPacks | None,
1249
+ limits: PackedLimits,
1250
+ deadline: float,
1251
+ ) -> PackedPublicationResult:
1252
+ """Finish the evidence transaction a hard kill left after installing the packed head."""
1253
+
1254
+ head_sha256 = sha256_bytes(head_raw)
1255
+ installed = verify_packed_generation(
1256
+ output_root,
1257
+ expected_head_sha256=head_sha256,
1258
+ root_descriptor=directory.root_descriptor,
1259
+ )
1260
+ terminal_raw = directory.read(PACKED_TERMINAL_FILENAME, maximum=MAX_PACKED_TERMINAL_BYTES)
1261
+ if terminal_raw is None or sha256_bytes(terminal_raw) != installed.head["terminal_sha256"]:
1262
+ raise _refuse(
1263
+ "PACKED_TERMINAL_DIGEST",
1264
+ "packed.terminal",
1265
+ "the published head does not have its exact terminal record",
1266
+ )
1267
+ terminal = _parse_terminal(terminal_raw)
1268
+ if terminal["generation"] != dict(generation):
1269
+ raise _refuse(
1270
+ "PACKED_TERMINAL_GENERATION",
1271
+ "packed.terminal.generation",
1272
+ "the published terminal record belongs to another generation",
1273
+ )
1274
+ template = terminal["journal"]
1275
+ if (
1276
+ set(template) != {"output_root", "generation", "completed_ranges", "invocations"}
1277
+ or template["output_root"] != os.path.abspath(os.fspath(output_root))
1278
+ or template["generation"] != dict(generation)
1279
+ or not isinstance(template["completed_ranges"], list)
1280
+ or not isinstance(template["invocations"], list)
1281
+ ):
1282
+ raise _refuse(
1283
+ "PACKED_TERMINAL_SCHEMA",
1284
+ "packed.terminal.journal",
1285
+ "the terminal journal template differs from this generation",
1286
+ )
1287
+ existing = _read_journal(directory)
1288
+ if existing is None:
1289
+ raise _refuse(
1290
+ "PACKED_JOURNAL_MISSING",
1291
+ "packed.journal",
1292
+ "terminal recovery requires the publication journal",
1293
+ )
1294
+ if existing["completed_ranges"] != template["completed_ranges"]:
1295
+ raise _refuse(
1296
+ "PACKED_JOURNAL_DIGEST",
1297
+ "packed.journal.completed_ranges",
1298
+ "the publication journal differs from the head-bound terminal ranges",
1299
+ )
1300
+ expected_counts, expected_counters = _verify_recovery_ranges(
1301
+ directory,
1302
+ installed,
1303
+ completed_ranges=existing["completed_ranges"],
1304
+ delta_root=delta_root,
1305
+ delta_manifest=delta_manifest,
1306
+ delta_shards_descriptor=delta_shards_descriptor,
1307
+ identity_descriptor=identity_descriptor,
1308
+ authoring_root=authoring_root,
1309
+ authoring_manifest=authoring_manifest,
1310
+ identity_manifest_sha256=generation["identity_manifest_sha256"],
1311
+ packed_backends=packed_backends,
1312
+ seam_backends=seam_backends,
1313
+ posting_backend=posting_backend,
1314
+ predecessor=predecessor,
1315
+ limits=limits,
1316
+ deadline=deadline,
1317
+ )
1318
+ _verify_recovery_evidence(
1319
+ installed,
1320
+ terminal=terminal,
1321
+ journal_template=template,
1322
+ journal=existing,
1323
+ expected_counts=expected_counts,
1324
+ expected_counters=expected_counters,
1325
+ delta_manifest=delta_manifest,
1326
+ packed_backends=packed_backends,
1327
+ posting_backend=posting_backend,
1328
+ limits=limits,
1329
+ )
1330
+ final_journal = _Journal(
1331
+ output_root=template["output_root"],
1332
+ generation=dict(template["generation"]),
1333
+ completed_ranges=[dict(item) for item in template["completed_ranges"]],
1334
+ invocations=[dict(item) for item in template["invocations"]],
1335
+ ).document(status="complete", head_sha256=head_sha256)
1336
+ journal_sha256 = sha256_bytes(canonical_json_bytes(final_journal))
1337
+ if existing.get("status") == "incomplete" and existing.get("head_sha256") is None:
1338
+ incomplete = _Journal(
1339
+ output_root=template["output_root"],
1340
+ generation=dict(template["generation"]),
1341
+ completed_ranges=[dict(item) for item in template["completed_ranges"]],
1342
+ invocations=[dict(item) for item in template["invocations"]],
1343
+ ).document(status="incomplete", head_sha256=None)
1344
+ if existing != incomplete:
1345
+ raise _refuse(
1346
+ "PACKED_JOURNAL_DIGEST",
1347
+ "packed.journal",
1348
+ "the incomplete journal differs from the head-bound terminal record",
1349
+ )
1350
+ if (
1351
+ directory.publish(
1352
+ PACKED_JOURNAL_FILENAME, final_journal, maximum=MAX_PACKED_JOURNAL_BYTES
1353
+ )
1354
+ != journal_sha256
1355
+ ):
1356
+ raise _refuse(
1357
+ "PACKED_JOURNAL_DIGEST",
1358
+ "packed.journal",
1359
+ "the completed journal differs from the head-bound terminal record",
1360
+ )
1361
+ elif existing != final_journal:
1362
+ raise _refuse(
1363
+ "PACKED_JOURNAL_DIGEST",
1364
+ "packed.journal",
1365
+ "the complete journal differs from the head-bound terminal record",
1366
+ )
1367
+ receipt = _receipt_from_terminal(
1368
+ terminal, head_sha256=head_sha256, journal_sha256=journal_sha256
1369
+ )
1370
+ verify_packed_receipt(receipt, root=output_root, root_descriptor=directory.root_descriptor)
1371
+ counts = receipt.get("counts")
1372
+ counters = receipt.get("counters")
1373
+ if not isinstance(counts, dict) or not isinstance(counters, dict):
1374
+ raise _refuse("PACKED_TERMINAL_SCHEMA", "packed.terminal.receipt", "work totals differ")
1375
+ return PackedPublicationResult(
1376
+ status="complete",
1377
+ head_sha256=head_sha256,
1378
+ journal_sha256=journal_sha256,
1379
+ counters=dict(counters),
1380
+ counts=dict(counts),
1381
+ receipt=receipt,
1382
+ )
1383
+
1384
+
1385
+ # ------------------------------------------------------------------------------------------
1386
+ # Publication
1387
+ # ------------------------------------------------------------------------------------------
1388
+
1389
+
1390
+ def publish_packed_generation(
1391
+ *,
1392
+ delta_root: Path,
1393
+ expected_delta_manifest_sha256: str,
1394
+ authoring_root: Path,
1395
+ expected_authoring_manifest_sha256: str,
1396
+ output_root: Path,
1397
+ output_descriptor: int | None = None,
1398
+ delta_descriptor: int | None = None,
1399
+ backends: Sequence[EmbeddingBackend],
1400
+ predecessor_root: Path | None = None,
1401
+ expected_predecessor_head_sha256: str | None = None,
1402
+ expected_predecessor_generation_sha256: str | None = None,
1403
+ limits: PackedLimits | None = None,
1404
+ resume: bool = False,
1405
+ publication_binding_sha256: str | None = None,
1406
+ after_range: Callable[[PackedRangeProgress], None] | None = None,
1407
+ ) -> PackedPublicationResult:
1408
+ """Publish one classified sweep as one packed generation, counting its own work exactly."""
1409
+
1410
+ bounds = PackedLimits() if limits is None else limits
1411
+ if not isinstance(bounds, PackedLimits):
1412
+ raise _refuse("PACKED_LIMIT", "limits", "must be PackedLimits")
1413
+ if type(resume) is not bool:
1414
+ raise _refuse("PACKED_CONTRACT", "resume", "must be a boolean")
1415
+ if publication_binding_sha256 is not None and not _is_digest(publication_binding_sha256):
1416
+ raise _refuse(
1417
+ "PACKED_CONTRACT",
1418
+ "publication_binding_sha256",
1419
+ "must be a lowercase SHA-256 when the caller binds external generation evidence",
1420
+ )
1421
+ packed_backends, seam_backends = _resolve_backends(backends)
1422
+ posting_backend = _posting_backend(packed_backends)
1423
+ predecessor_coordinates = (
1424
+ predecessor_root,
1425
+ expected_predecessor_head_sha256,
1426
+ expected_predecessor_generation_sha256,
1427
+ )
1428
+ if any(value is None for value in predecessor_coordinates) and not all(
1429
+ value is None for value in predecessor_coordinates
1430
+ ):
1431
+ raise _refuse(
1432
+ "PACKED_PREDECESSOR_REQUIRED",
1433
+ "predecessor",
1434
+ "a predecessor is named by its root, exact head digest, and exact generation digest, "
1435
+ "or not at all",
1436
+ )
1437
+ if expected_predecessor_generation_sha256 is not None and not _is_digest(
1438
+ expected_predecessor_generation_sha256
1439
+ ):
1440
+ raise _refuse(
1441
+ "PACKED_PREDECESSOR_GENERATION",
1442
+ "predecessor.generation_sha256",
1443
+ "must be a lowercase SHA-256 digest",
1444
+ )
1445
+ deadline = time.monotonic() + bounds.max_wall_seconds
1446
+
1447
+ delta_manifest = _read_manifest(
1448
+ Path(delta_root) / DELTA_MANIFEST_FILENAME,
1449
+ expected_sha256=expected_delta_manifest_sha256,
1450
+ maximum=MAX_DELTA_MANIFEST_BYTES,
1451
+ schema=DELTA_MANIFEST_SCHEMA,
1452
+ label="delta.manifest",
1453
+ parent_descriptor=delta_descriptor,
1454
+ name=DELTA_MANIFEST_FILENAME,
1455
+ )
1456
+ authoring_manifest = _read_manifest(
1457
+ Path(authoring_root) / MANIFEST_FILENAME,
1458
+ expected_sha256=expected_authoring_manifest_sha256,
1459
+ maximum=MAX_AUTHORING_MANIFEST_BYTES,
1460
+ schema=AUTHORING_MANIFEST_SCHEMA,
1461
+ label="authoring.manifest",
1462
+ )
1463
+ if delta_manifest["inputs"]["authoring_manifest_sha256"] != expected_authoring_manifest_sha256:
1464
+ raise _refuse(
1465
+ "PACKED_INPUT_GENERATION",
1466
+ "authoring.manifest",
1467
+ "the classification was computed from another authored generation",
1468
+ )
1469
+ admit_authored_generation(Path(authoring_root), authoring_manifest)
1470
+ # Validate every classification descriptor while output is still absent. Shard bytes stay
1471
+ # lazy and bounded, but no caller-controlled size can defer a fixed-ceiling refusal until
1472
+ # publication has opened or mutated its output root.
1473
+ _ClassificationStream(Path(delta_root), delta_manifest)
1474
+ identity_manifest_sha256 = delta_manifest["outputs"]["identity_manifest_sha256"]
1475
+ provider_id = delta_manifest["inputs"]["provider_id"]
1476
+
1477
+ generation = {
1478
+ "authoring_manifest_sha256": expected_authoring_manifest_sha256,
1479
+ "backends": [backend.coordinate for backend in packed_backends],
1480
+ "delta_manifest_sha256": expected_delta_manifest_sha256,
1481
+ "identity_manifest_sha256": identity_manifest_sha256,
1482
+ "posting_backend_coordinate": posting_backend.coordinate,
1483
+ "predecessor_head_sha256": expected_predecessor_head_sha256,
1484
+ "provider_id": provider_id,
1485
+ }
1486
+ if publication_binding_sha256 is not None:
1487
+ generation["publication_binding_sha256"] = publication_binding_sha256
1488
+
1489
+ # Authenticate the complete predecessor coordinate before the output root is opened. The
1490
+ # predecessor is opened again under the publication stack below so its descriptor remains
1491
+ # retained while vectors are read; both reads require the same exact head bytes and generation.
1492
+ if predecessor_root is not None:
1493
+ predecessor_preflight = PackedDirectory(Path(predecessor_root))
1494
+ with predecessor_preflight.opened():
1495
+ _read_predecessor_head(
1496
+ predecessor_preflight,
1497
+ expected_predecessor_head_sha256,
1498
+ expected_predecessor_generation_sha256,
1499
+ )
1500
+
1501
+ directory = PackedDirectory(Path(output_root), root_descriptor=output_descriptor)
1502
+ with directory.opened(), contextlib.ExitStack() as input_stack:
1503
+ delta_shards_descriptor = None
1504
+ identity_descriptor = None
1505
+ if delta_descriptor is not None:
1506
+ delta_shards_descriptor = input_stack.enter_context(
1507
+ _opened_directory_at(delta_descriptor, DELTA_SHARDS_DIRNAME)
1508
+ )
1509
+ identity_descriptor = input_stack.enter_context(
1510
+ _opened_directory_at(delta_descriptor, DELTA_IDENTITY_DIRNAME)
1511
+ )
1512
+ head_raw = directory.read(PACKED_HEAD_FILENAME, maximum=MAX_PACKED_HEAD_BYTES)
1513
+ if head_raw is not None and resume:
1514
+ with contextlib.ExitStack() as stack:
1515
+ predecessor: _PredecessorPacks | None = None
1516
+ if predecessor_root is not None:
1517
+ predecessor_directory = PackedDirectory(Path(predecessor_root))
1518
+ stack.enter_context(predecessor_directory.opened())
1519
+ predecessor_head = _read_predecessor_head(
1520
+ predecessor_directory,
1521
+ expected_predecessor_head_sha256,
1522
+ expected_predecessor_generation_sha256,
1523
+ )
1524
+ predecessor = _PredecessorPacks(predecessor_directory, predecessor_head)
1525
+ predecessor.require_backends(
1526
+ [backend.coordinate for backend in packed_backends]
1527
+ )
1528
+ return _recover_terminal_publication(
1529
+ directory,
1530
+ output_root=Path(output_root),
1531
+ generation=generation,
1532
+ head_raw=head_raw,
1533
+ delta_root=Path(delta_root),
1534
+ delta_manifest=delta_manifest,
1535
+ delta_shards_descriptor=delta_shards_descriptor,
1536
+ identity_descriptor=identity_descriptor,
1537
+ authoring_root=Path(authoring_root),
1538
+ authoring_manifest=authoring_manifest,
1539
+ packed_backends=packed_backends,
1540
+ seam_backends=seam_backends,
1541
+ posting_backend=posting_backend,
1542
+ predecessor=predecessor,
1543
+ limits=bounds,
1544
+ deadline=deadline,
1545
+ )
1546
+ if head_raw is not None:
1547
+ raise _refuse(
1548
+ "PACKED_OUTPUT_EXISTS",
1549
+ "packed.output",
1550
+ "a published generation is immutable; publish into a new output root",
1551
+ )
1552
+ existing = _read_journal(directory)
1553
+ if resume:
1554
+ if existing is None:
1555
+ raise _refuse(
1556
+ "PACKED_JOURNAL_MISSING",
1557
+ "packed.journal",
1558
+ "an explicit resume needs the work journal its interruption left behind",
1559
+ )
1560
+ journal = _resume_from(
1561
+ directory,
1562
+ existing,
1563
+ output_root=Path(output_root),
1564
+ generation=generation,
1565
+ )
1566
+ else:
1567
+ if existing is not None:
1568
+ raise _refuse(
1569
+ "PACKED_OUTPUT_EXISTS",
1570
+ "packed.output",
1571
+ "this root holds an in-flight publication; resume it explicitly",
1572
+ )
1573
+ journal = _Journal(
1574
+ output_root=os.path.abspath(os.fspath(output_root)), generation=dict(generation)
1575
+ )
1576
+
1577
+ with contextlib.ExitStack() as stack:
1578
+ predecessor: _PredecessorPacks | None = None
1579
+ if predecessor_root is not None:
1580
+ predecessor_directory = PackedDirectory(Path(predecessor_root))
1581
+ stack.enter_context(predecessor_directory.opened())
1582
+ head = _read_predecessor_head(
1583
+ predecessor_directory,
1584
+ expected_predecessor_head_sha256,
1585
+ expected_predecessor_generation_sha256,
1586
+ )
1587
+ predecessor = _PredecessorPacks(predecessor_directory, head)
1588
+ predecessor.require_backends([backend.coordinate for backend in packed_backends])
1589
+ return _publish(
1590
+ directory=directory,
1591
+ journal=journal,
1592
+ delta_root=Path(delta_root),
1593
+ delta_manifest=delta_manifest,
1594
+ delta_shards_descriptor=delta_shards_descriptor,
1595
+ identity_descriptor=identity_descriptor,
1596
+ authoring_root=Path(authoring_root),
1597
+ authoring_manifest=authoring_manifest,
1598
+ identity_manifest_sha256=identity_manifest_sha256,
1599
+ packed_backends=packed_backends,
1600
+ seam_backends=seam_backends,
1601
+ posting_backend=posting_backend,
1602
+ predecessor=predecessor,
1603
+ provider_id=provider_id,
1604
+ generation=generation,
1605
+ limits=bounds,
1606
+ deadline=deadline,
1607
+ after_range=after_range,
1608
+ )
1609
+
1610
+
1611
+ def _resolve_backends(
1612
+ backends: Sequence[EmbeddingBackend],
1613
+ ) -> tuple[tuple[PackedBackend, ...], tuple[EmbeddingBackend, ...]]:
1614
+ if not isinstance(backends, Sequence) or not 1 <= len(backends) <= MAX_PACKED_BACKENDS:
1615
+ raise _refuse(
1616
+ "PACKED_BACKEND",
1617
+ "backends",
1618
+ f"a packed generation carries 1 to {MAX_PACKED_BACKENDS} backends",
1619
+ )
1620
+ resolved = []
1621
+ for backend in backends:
1622
+ descriptor = backend.descriptor
1623
+ resolved.append(
1624
+ (
1625
+ PackedBackend(
1626
+ coordinate=descriptor.coordinate,
1627
+ dimensions=descriptor.dimensions,
1628
+ quantization=descriptor.quantization,
1629
+ query_safe=descriptor.query_safe,
1630
+ ),
1631
+ backend,
1632
+ )
1633
+ )
1634
+ resolved.sort(key=lambda pair: pair[0].coordinate)
1635
+ coordinates = [packed.coordinate for packed, _backend in resolved]
1636
+ if len(set(coordinates)) != len(coordinates):
1637
+ raise _refuse("PACKED_BACKEND", "backends", "backends must be distinct coordinates")
1638
+ return tuple(packed for packed, _ in resolved), tuple(backend for _, backend in resolved)
1639
+
1640
+
1641
+ def admit_authored_generation(root: Path, manifest: Mapping[str, Any]) -> None:
1642
+ """Reconcile and scan every authored record before opening the output directory."""
1643
+
1644
+ expected_counts = {
1645
+ "records": 0,
1646
+ "authored": 0,
1647
+ "flagged": 0,
1648
+ "skipped": 0,
1649
+ "failed": 0,
1650
+ "layer_text_truncated": 0,
1651
+ }
1652
+ expected_digests: dict[str, str | None] = dict.fromkeys(EMBEDDING_LAYERS)
1653
+ for item in _AuthoredStream(root, manifest)._records():
1654
+ _admit_authored_item(item)
1655
+ disposition = item["disposition"]
1656
+ expected_counts["records"] += 1
1657
+ expected_counts[disposition] += 1
1658
+ expected_counts["layer_text_truncated"] += len(item["layer_text_truncated"])
1659
+ entry = item["entry"]
1660
+ if entry is not None:
1661
+ for layer, text in _layer_texts(item).items():
1662
+ expected_digests[layer] = _authored_layer_chain(
1663
+ expected_digests[layer], entry["entry_id"], layer, text
1664
+ )
1665
+ staging = manifest.get("staging")
1666
+ coverage = manifest.get("coverage")
1667
+ if (
1668
+ manifest.get("counts") != expected_counts
1669
+ or manifest.get("layer_text_digests") != expected_digests
1670
+ or not isinstance(staging, Mapping)
1671
+ or type(staging.get("record_index_count")) is not int
1672
+ or staging["record_index_count"] < expected_counts["records"]
1673
+ ):
1674
+ raise _refuse(
1675
+ "PACKED_INPUT_MANIFEST",
1676
+ "authoring.manifest",
1677
+ "authored manifest totals do not reproduce from its exact shard records",
1678
+ )
1679
+ # This used to be `staging["unique_records"] != expected_counts["records"]`: the sweep found
1680
+ # exactly as many records as the shards hold. That is the exhaustive case of the two
1681
+ # statements below, and it was the only case the pipeline could publish. Stated as two, a
1682
+ # bounded generation is admissible and still has to account for itself exactly -- the corpus
1683
+ # it claims to have drawn from is the one its own staging coordinate names, and the selection
1684
+ # it claims to have taken is the population its own shards actually contain.
1685
+ if (
1686
+ not coverage_is_valid(coverage)
1687
+ or coverage["corpus_records"] != staging.get("unique_records")
1688
+ or coverage["selected_records"] != expected_counts["records"]
1689
+ ):
1690
+ raise _refuse(
1691
+ "PACKED_INPUT_COVERAGE",
1692
+ "authoring.manifest.coverage",
1693
+ "the authored generation's stated coverage does not reproduce from its own staging "
1694
+ "coordinate and shard records",
1695
+ )
1696
+
1697
+
1698
+ def _authored_layer_chain(previous: str | None, entry_id: str, layer: str, text: str) -> str:
1699
+ return canonical_sha256(
1700
+ {
1701
+ "previous_sha256": previous,
1702
+ "entry_id": entry_id,
1703
+ "layer": layer,
1704
+ "text_sha256": sha256_bytes(text.encode("utf-8")),
1705
+ }
1706
+ )
1707
+
1708
+
1709
+ _AUTHORED_ITEM_MEMBERS = frozenset(
1710
+ {
1711
+ "provider_record_id",
1712
+ "disposition",
1713
+ "reason_code",
1714
+ "entry",
1715
+ "layer_texts",
1716
+ "layer_text_truncated",
1717
+ }
1718
+ )
1719
+
1720
+ _REASON_CODES_BY_DISPOSITION = {
1721
+ "authored": frozenset({"AUTHOR_OK"}),
1722
+ "flagged": frozenset(
1723
+ {
1724
+ "AUTHOR_RIGHTS_ABSENT",
1725
+ "AUTHOR_RIGHTS_CONDITIONAL",
1726
+ "AUTHOR_RIGHTS_PROHIBITED",
1727
+ "AUTHOR_RIGHTS_PROSE",
1728
+ "AUTHOR_RIGHTS_UNMAPPED",
1729
+ }
1730
+ ),
1731
+ "skipped": frozenset(
1732
+ {
1733
+ "AUTHOR_IDENTIFIER_UNSAFE",
1734
+ "AUTHOR_TITLE_ABSENT",
1735
+ "AUTHOR_TITLE_LIMIT",
1736
+ "AUTHOR_TITLE_UNSAFE",
1737
+ }
1738
+ ),
1739
+ "failed": frozenset(
1740
+ {
1741
+ "AUTHOR_ENTRY_CONTRACT",
1742
+ "AUTHOR_RECORD_DIGEST",
1743
+ "AUTHOR_RECORD_FIELDS",
1744
+ "AUTHOR_RECORD_SCHEMA",
1745
+ }
1746
+ ),
1747
+ }
1748
+
1749
+ assert frozenset().union(*_REASON_CODES_BY_DISPOSITION.values()) == AUTHORING_REASON_CODES
1750
+
1751
+
1752
+ def _admit_authored_item(item: Mapping[str, Any]) -> None:
1753
+ """Validate one closed authored-record shape and scan every unconstrained public string."""
1754
+
1755
+ if not isinstance(item, dict) or set(item) != _AUTHORED_ITEM_MEMBERS:
1756
+ raise _refuse(
1757
+ "PACKED_INPUT_ENTRY",
1758
+ "authoring.shard.items[]",
1759
+ "an authored item must carry exactly the authored-record members",
1760
+ )
1761
+ disposition = item["disposition"]
1762
+ if disposition not in AUTHORING_DISPOSITIONS:
1763
+ raise _refuse(
1764
+ "PACKED_INPUT_ENTRY",
1765
+ "authoring.shard.items[].disposition",
1766
+ "an authored item disposition differs",
1767
+ )
1768
+ provider_record_id = item["provider_record_id"]
1769
+ reason_code = item["reason_code"]
1770
+ if not isinstance(provider_record_id, str) or not isinstance(reason_code, str):
1771
+ raise _refuse(
1772
+ "PACKED_INPUT_ENTRY",
1773
+ "authoring.shard.items[]",
1774
+ "an authored item has invalid public strings",
1775
+ )
1776
+ if reason_code not in _REASON_CODES_BY_DISPOSITION[disposition]:
1777
+ raise _refuse(
1778
+ "PACKED_INPUT_ENTRY",
1779
+ "authoring.shard.items[].reason_code",
1780
+ "an authored item reason does not match its disposition",
1781
+ )
1782
+ for field_name, value in (
1783
+ ("provider_record_id", provider_record_id),
1784
+ ("reason_code", reason_code),
1785
+ ):
1786
+ admit_public_fact_bytes(canonical_json_bytes({field_name: value}))
1787
+
1788
+ truncated = item["layer_text_truncated"]
1789
+ if (
1790
+ not isinstance(truncated, list)
1791
+ or any(not isinstance(layer, str) or layer not in EMBEDDING_LAYERS for layer in truncated)
1792
+ or len(set(truncated)) != len(truncated)
1793
+ ):
1794
+ raise _refuse(
1795
+ "PACKED_LAYER_TEXTS",
1796
+ "authoring.shard.items[].layer_text_truncated",
1797
+ "truncated layers must be a unique subset of the embedding layers",
1798
+ )
1799
+
1800
+ if disposition in {"skipped", "failed"}:
1801
+ if item["entry"] is not None or item["layer_texts"] != [] or truncated:
1802
+ raise _refuse(
1803
+ "PACKED_INPUT_ENTRY",
1804
+ "authoring.shard.items[]",
1805
+ "a skipped or failed item cannot carry public facts",
1806
+ )
1807
+ return
1808
+
1809
+ entry = item["entry"]
1810
+ if not isinstance(entry, dict):
1811
+ raise _refuse(
1812
+ "PACKED_INPUT_ENTRY",
1813
+ "authoring.shard.items[].entry",
1814
+ "an authored or flagged item must carry one public entry",
1815
+ )
1816
+ _admit_entry_prose(entry, provider_record_id=provider_record_id)
1817
+ for text in _layer_texts(item).values():
1818
+ admit_public_fact_bytes(text.encode("utf-8"))
1819
+
1820
+
1821
+ _PUBLIC_PROSE_FACTS = (
1822
+ "title",
1823
+ "publisher",
1824
+ "description",
1825
+ "spatial_scope",
1826
+ "data_formats",
1827
+ "access_kind",
1828
+ "authentication_required",
1829
+ "rights",
1830
+ "declared_columns",
1831
+ "declared_vocabulary",
1832
+ "declared_row_count",
1833
+ "profiles",
1834
+ )
1835
+
1836
+
1837
+ def _admit_entry_prose(entry: Mapping[str, Any], *, provider_record_id: str) -> None:
1838
+ """Classify publishable semantic values without treating verified digests as prose."""
1839
+
1840
+ try:
1841
+ captured = catalog_entry_v2_from_dict(entry)
1842
+ except SourceContractError as error:
1843
+ raise _refuse(error.code, error.path, error.detail) from error
1844
+ canonical_entry = captured.to_dict()
1845
+ if canonical_entry != entry:
1846
+ raise _refuse(
1847
+ "PACKED_INPUT_ENTRY",
1848
+ "authoring.shard.items[].entry",
1849
+ "the authored entry is not its strict canonical contract spelling",
1850
+ )
1851
+ identity = captured.provider_record
1852
+ if identity.provider_record_id != provider_record_id:
1853
+ raise _refuse(
1854
+ "PACKED_INPUT_IDENTITY",
1855
+ "authoring.shard.items[].provider_record_id",
1856
+ "the authored item key differs from its provider identity",
1857
+ )
1858
+ if captured.entry_id != derive_entry_id(identity.provider_id, provider_record_id):
1859
+ raise _refuse(
1860
+ "PACKED_INPUT_IDENTITY",
1861
+ "authoring.shard.items[].entry.entry_id",
1862
+ "the entry id is not derived from the authored provider identity",
1863
+ )
1864
+ admit_public_fact_bytes(canonical_json_bytes({"provider_record_id": provider_record_id}))
1865
+ for fact_name in _PUBLIC_PROSE_FACTS:
1866
+ fact = canonical_entry[fact_name]
1867
+ # Keep the field name next to its exact value so assignment-shaped content is still
1868
+ # classified, while authenticated coordinates, digests and evidence identifiers cannot
1869
+ # become accidental prose findings.
1870
+ admit_public_fact_bytes(canonical_json_bytes({fact_name: fact["value"]}))
1871
+ for path, value in _publishable_strings(canonical_entry):
1872
+ admit_public_fact_bytes(canonical_json_bytes({path: value}))
1873
+
1874
+
1875
+ def _publishable_strings(value: Any, *, path: tuple[str, ...] = ()) -> Iterator[tuple[str, str]]:
1876
+ """Yield source-controlled strings, excluding only validated digests and derived labels."""
1877
+
1878
+ if isinstance(value, dict):
1879
+ for key, item in value.items():
1880
+ if key in {"schema_version", "entry_id"}:
1881
+ continue
1882
+ if key.endswith(("_sha256", "_digest")):
1883
+ if item is not None and not _is_digest(item):
1884
+ raise _refuse(
1885
+ "PACKED_INPUT_ENTRY",
1886
+ ".".join((*path, key)),
1887
+ "a digest field is not a lowercase SHA-256 digest",
1888
+ )
1889
+ continue
1890
+ yield from _publishable_strings(item, path=(*path, key))
1891
+ return
1892
+ if isinstance(value, list):
1893
+ for item in value:
1894
+ yield from _publishable_strings(item, path=path)
1895
+ return
1896
+ if isinstance(value, str):
1897
+ yield (".".join(path), value)
1898
+
1899
+
1900
+ def _posting_backend(backends: Sequence[PackedBackend]) -> PackedBackend:
1901
+ query_safe = [backend for backend in backends if backend.query_safe]
1902
+ if len(query_safe) != 1:
1903
+ raise _refuse(
1904
+ "PACKED_POSTING_BACKEND",
1905
+ "backends",
1906
+ "postings are derived from exactly one query-safe backend",
1907
+ )
1908
+ return query_safe[0]
1909
+
1910
+
1911
+ def _read_predecessor_head(
1912
+ directory: PackedDirectory,
1913
+ expected: str | None,
1914
+ expected_generation_sha256: str | None,
1915
+ ) -> dict[str, Any]:
1916
+ raw = directory.read(PACKED_HEAD_FILENAME, maximum=MAX_PACKED_HEAD_BYTES)
1917
+ if raw is None:
1918
+ raise _refuse(
1919
+ "PACKED_PREDECESSOR_REQUIRED",
1920
+ "predecessor.head",
1921
+ "no readable packed head at the predecessor root",
1922
+ )
1923
+ if sha256_bytes(raw) != expected:
1924
+ raise _refuse(
1925
+ "PACKED_PREDECESSOR_DIGEST",
1926
+ "predecessor.head",
1927
+ "the predecessor generation is not the caller's exact head",
1928
+ )
1929
+ head = parse_canonical_json(raw)
1930
+ if not isinstance(head, dict) or not isinstance(head.get("ranges"), list):
1931
+ raise _refuse(
1932
+ "PACKED_PREDECESSOR_DIGEST", "predecessor.head", "predecessor head contract differs"
1933
+ )
1934
+ if head.get("generation_sha256") != expected_generation_sha256:
1935
+ raise _refuse(
1936
+ "PACKED_PREDECESSOR_GENERATION",
1937
+ "predecessor.head.generation_sha256",
1938
+ "the predecessor packed head was not published from the named history generation",
1939
+ )
1940
+ return head
1941
+
1942
+
1943
+ def _publish(
1944
+ *,
1945
+ directory: PackedDirectory,
1946
+ journal: _Journal,
1947
+ delta_root: Path,
1948
+ delta_manifest: Mapping[str, Any],
1949
+ delta_shards_descriptor: int | None,
1950
+ identity_descriptor: int | None,
1951
+ authoring_root: Path,
1952
+ authoring_manifest: Mapping[str, Any],
1953
+ identity_manifest_sha256: str,
1954
+ packed_backends: tuple[PackedBackend, ...],
1955
+ seam_backends: tuple[EmbeddingBackend, ...],
1956
+ posting_backend: PackedBackend,
1957
+ predecessor: _PredecessorPacks | None,
1958
+ provider_id: str,
1959
+ generation: Mapping[str, Any],
1960
+ limits: PackedLimits,
1961
+ deadline: float,
1962
+ after_range: Callable[[PackedRangeProgress], None] | None,
1963
+ ) -> PackedPublicationResult:
1964
+ counters = _Counters()
1965
+ seams = {
1966
+ backend.descriptor.coordinate: _CountingSeam(backend, counters) for backend in seam_backends
1967
+ }
1968
+ classification = _ClassificationStream(
1969
+ delta_root, delta_manifest, shards_descriptor=delta_shards_descriptor
1970
+ )
1971
+ authored = _AuthoredStream(authoring_root, authoring_manifest)
1972
+ identity = _HistoryStream(
1973
+ IdentityHistoryStore.open(
1974
+ delta_root / DELTA_IDENTITY_DIRNAME,
1975
+ expected_manifest_sha256=identity_manifest_sha256,
1976
+ root_descriptor=identity_descriptor,
1977
+ )
1978
+ )
1979
+ already = len(journal.completed_ranges)
1980
+ counts = {
1981
+ "ranges": 0,
1982
+ "entries": 0,
1983
+ "fresh_ranges": 0,
1984
+ "new": 0,
1985
+ "changed": 0,
1986
+ "unchanged": 0,
1987
+ "members": 0,
1988
+ }
1989
+ for record in journal.invocations:
1990
+ counters.absorb_record(record)
1991
+ invocation_start = counters.to_dict()
1992
+ descriptors: list[PackedRangeDescriptor] = []
1993
+ vector_payload_sha256s: list[str] = []
1994
+ for range_index, entries in enumerate(_ranges(classification, authored, identity)):
1995
+ if range_index >= limits.max_ranges:
1996
+ raise _refuse(
1997
+ "PACKED_RANGE_LIMIT", "packed.ranges", "the generation exceeds its range bound"
1998
+ )
1999
+ range_input = PackedRangeInput(
2000
+ range_index=range_index, entries=tuple(item.to_entry() for item in entries)
2001
+ )
2002
+ fresh = tuple(
2003
+ position
2004
+ for position, item in enumerate(entries)
2005
+ if item.classification in FRESH_CLASSIFICATIONS
2006
+ )
2007
+ encoded_positions = set(fresh)
2008
+ reused = tuple(
2009
+ position for position in range(len(entries)) if position not in encoded_positions
2010
+ )
2011
+ # Every range this generation contains is counted here, whether this invocation built it
2012
+ # or a previous one did, so the counter equations hold across a resume as written.
2013
+ counts["ranges"] += 1
2014
+ counts["entries"] += len(entries)
2015
+ counts["fresh_ranges"] += 1 if fresh else 0
2016
+ for item in entries:
2017
+ counts[item.classification] += 1
2018
+ if range_index < already:
2019
+ completed = journal.completed_ranges[range_index]
2020
+ _verify_completed_range(directory, completed, range_input)
2021
+ descriptors.append(_completed_descriptor(completed))
2022
+ vector_payload_sha256s.extend(completed["vector_payload_sha256s"])
2023
+ counts["members"] += len(completed["member_sha256s"])
2024
+ continue
2025
+ if time.monotonic() >= deadline:
2026
+ raise _refuse(
2027
+ "PACKED_WALL_LIMIT", "packed.output", "publication exceeded its wall bound"
2028
+ )
2029
+ if reused and predecessor is None:
2030
+ raise _refuse(
2031
+ "PACKED_PREDECESSOR_REQUIRED",
2032
+ "predecessor",
2033
+ "an unchanged entry can only be published against the packs it reuses",
2034
+ )
2035
+ recovered: dict[int, dict[str, dict[str, tuple[int, ...]]]] = {}
2036
+ for position in reused:
2037
+ assert predecessor is not None
2038
+ recovered[position] = predecessor.lookup(entries[position].key)
2039
+
2040
+ vectors: dict[str, dict[str, tuple[tuple[int, ...], ...]]] = {}
2041
+ for backend in packed_backends:
2042
+ seam = seams[backend.coordinate]
2043
+ layers: dict[str, tuple[tuple[int, ...], ...]] = {}
2044
+ for layer in EMBEDDING_LAYERS:
2045
+ column: list[tuple[int, ...] | None] = [None] * len(entries)
2046
+ if fresh:
2047
+ encoded = seam.encode_many(
2048
+ tuple(entries[position].layer_texts[layer] for position in fresh)
2049
+ )
2050
+ for position, vector in zip(fresh, encoded, strict=True):
2051
+ column[position] = vector
2052
+ for position in reused:
2053
+ vector = recovered[position].get(backend.coordinate, {}).get(layer)
2054
+ if vector is None or len(vector) != backend.dimensions:
2055
+ raise _refuse(
2056
+ "PACKED_REUSE_MISSING",
2057
+ "predecessor.ranges",
2058
+ "an unchanged entry has no reusable layer vector for this backend",
2059
+ )
2060
+ column[position] = vector
2061
+ layers[layer] = tuple(value for value in column if value is not None)
2062
+ if len(layers[layer]) != len(entries):
2063
+ raise _refuse( # pragma: no cover - every position is filled above
2064
+ "PACKED_RANGE_ENTRY",
2065
+ "packed.range.vectors",
2066
+ "a layer member does not cover every entry in its range",
2067
+ )
2068
+ vectors[backend.coordinate] = layers
2069
+
2070
+ built = build_packed_range(
2071
+ range_input,
2072
+ vectors,
2073
+ backends=packed_backends,
2074
+ posting_backend_coordinate=posting_backend.coordinate,
2075
+ max_terms_per_segment=limits.max_terms_per_segment,
2076
+ )
2077
+ member_sha256s = _install_range(directory, built)
2078
+ if directory.bytes_written > limits.max_disk_bytes:
2079
+ raise _refuse(
2080
+ "PACKED_DISK_LIMIT", "packed.output", "publication exceeds its local disk bound"
2081
+ )
2082
+ counters.reused_layer_items += len(reused) * len(EMBEDDING_LAYERS) * len(packed_backends)
2083
+ payloads = _vector_payload_sha256s(built)
2084
+ completed = _completed_range_record(built)
2085
+ if completed["member_sha256s"] != [list(item) for item in member_sha256s]:
2086
+ raise _refuse( # pragma: no cover - installation checks every built descriptor
2087
+ "PACKED_OUTPUT_READBACK",
2088
+ "packed.output.member",
2089
+ "installed member coordinates differ from the range that produced them",
2090
+ )
2091
+ journal.completed_ranges.append(completed)
2092
+ descriptors.append(built.descriptor)
2093
+ vector_payload_sha256s.extend(payloads)
2094
+ counts["members"] += len(built.members)
2095
+ _write_journal(directory, journal, counters, invocation_start, status="incomplete")
2096
+ if after_range is not None:
2097
+ after_range(
2098
+ PackedRangeProgress(
2099
+ range_index=range_index,
2100
+ first_key=built.first_key,
2101
+ last_key=built.last_key,
2102
+ entry_count=built.entry_count,
2103
+ fresh_entries=len(fresh),
2104
+ reused_entries=len(reused),
2105
+ range_binding_sha256=built.range_binding_sha256,
2106
+ posting_backend_coordinate=posting_backend.coordinate,
2107
+ manifest_sha256=built.manifest_sha256,
2108
+ member_sha256s=member_sha256s,
2109
+ )
2110
+ )
2111
+
2112
+ if not descriptors:
2113
+ raise _refuse(
2114
+ "PACKED_INPUT_EMPTY",
2115
+ "delta.shards",
2116
+ "a classification with no publishable entry publishes no generation",
2117
+ )
2118
+ if counts["ranges"] < already:
2119
+ raise _refuse(
2120
+ "PACKED_JOURNAL_RANGE",
2121
+ "packed.journal.completed_ranges",
2122
+ "the journal completed more ranges than this classification produces",
2123
+ )
2124
+ if counts["entries"] > limits.max_entries:
2125
+ raise _refuse(
2126
+ "PACKED_ENTRY_LIMIT", "packed.entries", "the generation exceeds its entry bound"
2127
+ )
2128
+
2129
+ # The independent final reader: every emitted pack is re-read from disk and its accelerators
2130
+ # recomputed, before any head exists to point at them.
2131
+ posting_term_count = 0
2132
+ posting_member_sha256s: list[str] = []
2133
+ bound_member_sha256s: list[str] = []
2134
+ for descriptor in descriptors:
2135
+ verified = verify_packed_range(
2136
+ directory,
2137
+ descriptor.manifest_sha256,
2138
+ backends=packed_backends,
2139
+ posting_backend_coordinate=posting_backend.coordinate,
2140
+ )
2141
+ if (
2142
+ verified.entry_count != descriptor.entry_count
2143
+ or verified.facts_sha256 != descriptor.facts_sha256
2144
+ or verified.first_key != descriptor.first_key
2145
+ or verified.last_key != descriptor.last_key
2146
+ ):
2147
+ raise _refuse(
2148
+ "PACKED_RANGE_DIGEST",
2149
+ "packed.range",
2150
+ "an emitted range does not read back as the range its descriptor names",
2151
+ )
2152
+ posting_term_count += verified.posting_term_count
2153
+ posting_member_sha256s.extend(verified.posting_sha256s)
2154
+ bound_member_sha256s.extend(verified.bound_sha256s)
2155
+
2156
+ _verify_counter_equations(counters, counts, backend_count=len(packed_backends))
2157
+
2158
+ # Invocation boundaries are process history and cannot be authenticated after a hard kill.
2159
+ # The durable representation is therefore one canonical aggregate, which is independently
2160
+ # reproducible from the exact inputs and the classifications in the immutable facts members.
2161
+ invocations = [counters.to_dict()]
2162
+ residency = _canonical_residency(counts, generation)
2163
+ terminal = _terminal_document(
2164
+ provider_id=provider_id,
2165
+ generation=generation,
2166
+ coverage=authoring_manifest["coverage"],
2167
+ delta_manifest=delta_manifest,
2168
+ packed_backends=packed_backends,
2169
+ posting_backend=posting_backend,
2170
+ counters=counters,
2171
+ counts=counts,
2172
+ invocations=invocations,
2173
+ descriptors=descriptors,
2174
+ posting_term_count=posting_term_count,
2175
+ vector_payload_sha256s=vector_payload_sha256s,
2176
+ posting_member_sha256s=posting_member_sha256s,
2177
+ bound_member_sha256s=bound_member_sha256s,
2178
+ limits=limits,
2179
+ residency=residency,
2180
+ journal=journal,
2181
+ )
2182
+ terminal_sha256 = sha256_bytes(canonical_json_bytes(terminal))
2183
+ head = build_packed_head(
2184
+ provider_id=provider_id,
2185
+ generation_sha256=generation["delta_manifest_sha256"],
2186
+ predecessor_head_sha256=generation["predecessor_head_sha256"],
2187
+ backends=packed_backends,
2188
+ posting_backend_coordinate=posting_backend.coordinate,
2189
+ ranges=descriptors,
2190
+ coverage=authoring_manifest["coverage"],
2191
+ counters=counters.to_dict(),
2192
+ invocations=invocations,
2193
+ terminal_sha256=terminal_sha256,
2194
+ )
2195
+ head_sha256 = sha256_bytes(canonical_json_bytes(head))
2196
+ complete_journal = _Journal(
2197
+ output_root=journal.output_root,
2198
+ generation=journal.generation,
2199
+ completed_ranges=journal.completed_ranges,
2200
+ invocations=invocations,
2201
+ ).document(status="complete", head_sha256=head_sha256)
2202
+ journal_sha256 = sha256_bytes(canonical_json_bytes(complete_journal))
2203
+ receipt = _receipt_from_terminal(
2204
+ terminal, head_sha256=head_sha256, journal_sha256=journal_sha256
2205
+ )
2206
+
2207
+ final_incomplete_journal = _Journal(
2208
+ output_root=journal.output_root,
2209
+ generation=journal.generation,
2210
+ completed_ranges=journal.completed_ranges,
2211
+ invocations=invocations,
2212
+ ).document(status="incomplete", head_sha256=None)
2213
+ directory.publish(
2214
+ PACKED_JOURNAL_FILENAME,
2215
+ final_incomplete_journal,
2216
+ maximum=MAX_PACKED_JOURNAL_BYTES,
2217
+ )
2218
+ installed_terminal_sha256 = directory.publish(
2219
+ PACKED_TERMINAL_FILENAME, terminal, maximum=MAX_PACKED_TERMINAL_BYTES
2220
+ )
2221
+ if installed_terminal_sha256 != terminal_sha256:
2222
+ raise _refuse(
2223
+ "PACKED_TERMINAL_DIGEST",
2224
+ "packed.terminal",
2225
+ "installed terminal record differs from the record the head names",
2226
+ )
2227
+ installed_head_sha256 = directory.publish(
2228
+ PACKED_HEAD_FILENAME, head, maximum=MAX_PACKED_HEAD_BYTES
2229
+ )
2230
+ if installed_head_sha256 != head_sha256:
2231
+ raise _refuse(
2232
+ "PACKED_HEAD_MISMATCH", "packed.head", "installed head differs from its coordinate"
2233
+ )
2234
+ installed = verify_packed_generation(
2235
+ directory.root,
2236
+ expected_head_sha256=head_sha256,
2237
+ root_descriptor=directory.root_descriptor,
2238
+ )
2239
+ if (
2240
+ installed.entry_count != counts["entries"]
2241
+ or installed.range_count != counts["ranges"]
2242
+ or installed.member_count != counts["members"]
2243
+ ):
2244
+ raise _refuse(
2245
+ "PACKED_HEAD_EXPECTED",
2246
+ "packed.head",
2247
+ "the installed head does not read back as the generation just written",
2248
+ )
2249
+
2250
+ installed_journal_sha256 = directory.publish(
2251
+ PACKED_JOURNAL_FILENAME, complete_journal, maximum=MAX_PACKED_JOURNAL_BYTES
2252
+ )
2253
+ if installed_journal_sha256 != journal_sha256:
2254
+ raise _refuse(
2255
+ "PACKED_JOURNAL_DIGEST",
2256
+ "packed.journal",
2257
+ "installed complete journal differs from the terminal record",
2258
+ )
2259
+ return PackedPublicationResult(
2260
+ status="complete",
2261
+ head_sha256=head_sha256,
2262
+ journal_sha256=journal_sha256,
2263
+ counters=counters.to_dict(),
2264
+ counts=counts,
2265
+ receipt=receipt,
2266
+ )
2267
+
2268
+
2269
+ @dataclass(frozen=True)
2270
+ class _PublishableEntry:
2271
+ """One classified entry with everything the packed schema needs and nothing it does not."""
2272
+
2273
+ key: str
2274
+ entry_id: str
2275
+ classification: str
2276
+ semantic_facts_digest: str
2277
+ entry: dict[str, Any]
2278
+ history: dict[str, Any]
2279
+ layer_texts: dict[str, str]
2280
+
2281
+ def to_entry(self) -> dict[str, Any]:
2282
+ return {
2283
+ "key": self.key,
2284
+ "entry_id": self.entry_id,
2285
+ "classification": self.classification,
2286
+ "semantic_facts_digest": self.semantic_facts_digest,
2287
+ "entry": self.entry,
2288
+ "history": self.history,
2289
+ }
2290
+
2291
+
2292
+ def _ranges(
2293
+ classification: _ClassificationStream,
2294
+ authored: _AuthoredStream,
2295
+ identity: _HistoryStream,
2296
+ ) -> Iterator[list[_PublishableEntry]]:
2297
+ buffer: list[_PublishableEntry] = []
2298
+ for item in classification:
2299
+ buffer.append(_publishable(item, authored, identity))
2300
+ if len(buffer) == RANGE_ENTRIES:
2301
+ yield buffer
2302
+ buffer = []
2303
+ if buffer:
2304
+ yield buffer
2305
+
2306
+
2307
+ def _publishable(
2308
+ item: Mapping[str, Any], authored: _AuthoredStream, identity: _HistoryStream
2309
+ ) -> _PublishableEntry:
2310
+ record_id = item["provider_record_id"]
2311
+ authored_item = authored.advance_to(record_id)
2312
+ history = identity.advance_to(record_id)
2313
+ entry = authored_item.get("entry")
2314
+ if not isinstance(entry, dict) or entry.get("entry_id") != item["entry_id"]:
2315
+ raise _refuse(
2316
+ "PACKED_INPUT_ENTRY",
2317
+ "authoring.shard.items[].entry",
2318
+ "a classified entry does not match the authored entry it names",
2319
+ )
2320
+ if history.identity.entry_id != item["entry_id"]:
2321
+ raise _refuse(
2322
+ "PACKED_INPUT_IDENTITY",
2323
+ "identity.records",
2324
+ "a classified entry does not match the identity history it names",
2325
+ )
2326
+ layer_texts = _layer_texts(authored_item)
2327
+ return _PublishableEntry(
2328
+ key=record_id,
2329
+ entry_id=item["entry_id"],
2330
+ classification=item["classification"],
2331
+ semantic_facts_digest=item["semantic_facts_digest"],
2332
+ entry=entry,
2333
+ history=history.to_dict(),
2334
+ layer_texts=layer_texts,
2335
+ )
2336
+
2337
+
2338
+ def _layer_texts(authored_item: Mapping[str, Any]) -> dict[str, str]:
2339
+ """Read the exact four authored texts whose UTF-8 bytes the encoding seam meters."""
2340
+
2341
+ texts = authored_item.get("layer_texts")
2342
+ if not isinstance(texts, list) or len(texts) != len(EMBEDDING_LAYERS):
2343
+ raise _refuse(
2344
+ "PACKED_LAYER_TEXTS",
2345
+ "authoring.shard.items[].layer_texts",
2346
+ f"a published entry carries exactly the four layer texts {list(EMBEDDING_LAYERS)}",
2347
+ )
2348
+ layer_texts: dict[str, str] = {}
2349
+ for text in texts:
2350
+ if (
2351
+ not isinstance(text, dict)
2352
+ or set(text) != {"layer", "text"}
2353
+ or text.get("layer") not in EMBEDDING_LAYERS
2354
+ or not isinstance(text.get("text"), str)
2355
+ or not text["text"]
2356
+ or len(text["text"].encode("utf-8")) > MAX_LAYER_TEXT_BYTES
2357
+ ):
2358
+ raise _refuse(
2359
+ "PACKED_LAYER_TEXTS",
2360
+ "authoring.shard.items[].layer_texts[]",
2361
+ "each layer text names its layer and carries text",
2362
+ )
2363
+ layer_texts[text["layer"]] = text["text"]
2364
+ if set(layer_texts) != set(EMBEDDING_LAYERS) or [text["layer"] for text in texts] != list(
2365
+ EMBEDDING_LAYERS
2366
+ ):
2367
+ raise _refuse(
2368
+ "PACKED_LAYER_TEXTS",
2369
+ "authoring.shard.items[].layer_texts",
2370
+ f"a published entry carries exactly the four layer texts {list(EMBEDDING_LAYERS)}",
2371
+ )
2372
+ return layer_texts
2373
+
2374
+
2375
+ def _install_range(
2376
+ directory: PackedDirectory, built: BuiltPackedRange
2377
+ ) -> tuple[tuple[str, str], ...]:
2378
+ installed: list[tuple[str, str]] = []
2379
+ for member in built.members:
2380
+ digest, size = directory.write_member_bytes(member.raw, kind=member.descriptor.kind)
2381
+ if digest != member.descriptor.sha256 or size != member.descriptor.bytes:
2382
+ raise _refuse( # pragma: no cover - content addressing makes this unreachable
2383
+ "PACKED_OUTPUT_READBACK",
2384
+ "packed.output.member",
2385
+ "an installed member is not the member it was built as",
2386
+ )
2387
+ installed.append((member.descriptor.kind, digest))
2388
+ manifest_sha256, _size = directory.write_member_bytes(built.manifest_raw, kind="range")
2389
+ if manifest_sha256 != built.manifest_sha256:
2390
+ raise _refuse( # pragma: no cover - content addressing makes this unreachable
2391
+ "PACKED_OUTPUT_READBACK",
2392
+ "packed.output.manifest",
2393
+ "an installed range manifest is not the manifest it was built as",
2394
+ )
2395
+ return tuple(installed)
2396
+
2397
+
2398
+ def _vector_payload_sha256s(built: BuiltPackedRange) -> tuple[str, ...]:
2399
+ """Digest each layer member's fixed-width vector payload, header excluded, in published order.
2400
+
2401
+ The header binds a member to its range, so two generations that reuse the same vectors publish
2402
+ different member digests: the facts a range carries move even when its vectors do not. The
2403
+ payload is the part reuse is a claim about, so this is the digest a receipt states when it says
2404
+ an unchanged entry's packs are the predecessor's bytes.
2405
+ """
2406
+
2407
+ return tuple(
2408
+ sha256_bytes(member.raw[MEMBER_HEADER_BYTES:])
2409
+ for member in built.members
2410
+ if member.descriptor.kind == "vector"
2411
+ )
2412
+
2413
+
2414
+ def _completed_range_record(built: BuiltPackedRange) -> dict[str, Any]:
2415
+ """State the exact durable range record derived from one deterministic build."""
2416
+
2417
+ return {
2418
+ "range_index": built.range_index,
2419
+ "first_key": built.first_key,
2420
+ "last_key": built.last_key,
2421
+ "entry_count": built.entry_count,
2422
+ "facts_sha256": built.facts_sha256,
2423
+ "range_binding_sha256": built.range_binding_sha256,
2424
+ "range_manifest_sha256": built.manifest_sha256,
2425
+ "range_manifest_bytes": len(built.manifest_raw),
2426
+ "member_sha256s": [
2427
+ [member.descriptor.kind, member.descriptor.sha256] for member in built.members
2428
+ ],
2429
+ "vector_payload_sha256s": list(_vector_payload_sha256s(built)),
2430
+ }
2431
+
2432
+
2433
+ def _completed_descriptor(completed: Mapping[str, Any]) -> PackedRangeDescriptor:
2434
+ return PackedRangeDescriptor(
2435
+ range_index=completed["range_index"],
2436
+ first_key=completed["first_key"],
2437
+ last_key=completed["last_key"],
2438
+ entry_count=completed["entry_count"],
2439
+ facts_sha256=completed["facts_sha256"],
2440
+ range_binding_sha256=completed["range_binding_sha256"],
2441
+ manifest_sha256=completed["range_manifest_sha256"],
2442
+ manifest_bytes=completed["range_manifest_bytes"],
2443
+ member_count=len(completed["member_sha256s"]),
2444
+ )
2445
+
2446
+
2447
+ def _verify_completed_range(
2448
+ directory: PackedDirectory, completed: Mapping[str, Any], range_input: PackedRangeInput
2449
+ ) -> None:
2450
+ """Prove one already-emitted range is the range this stream produces, encoding nothing.
2451
+
2452
+ An ordinary pre-head resume must not re-encode a complete range, so this first proof is drawn
2453
+ entirely from bytes that already exist: the journal states the interval, size and facts order;
2454
+ the range binding has to reproduce from exactly those; and the installed facts member has to
2455
+ list exactly these entries carrying exactly these canonical documents. Terminal recovery adds
2456
+ a deterministic full-range replay after this inexpensive facts check.
2457
+ """
2458
+
2459
+ path = f"packed.journal.completed_ranges[{range_input.range_index}]"
2460
+ if (
2461
+ completed["range_index"] != range_input.range_index
2462
+ or completed["first_key"] != range_input.first_key
2463
+ or completed["last_key"] != range_input.last_key
2464
+ or completed["entry_count"] != len(range_input.entries)
2465
+ ):
2466
+ raise _refuse(
2467
+ "PACKED_JOURNAL_RANGE",
2468
+ path,
2469
+ "a completed range does not describe the range this stream produces",
2470
+ )
2471
+ if completed["range_binding_sha256"] != range_binding_sha256(
2472
+ range_index=range_input.range_index,
2473
+ first_key=range_input.first_key,
2474
+ last_key=range_input.last_key,
2475
+ entry_count=len(range_input.entries),
2476
+ facts_sha256=completed["facts_sha256"],
2477
+ ):
2478
+ raise _refuse(
2479
+ "PACKED_JOURNAL_RANGE",
2480
+ f"{path}.range_binding_sha256",
2481
+ "a completed range's binding does not reproduce from its own interval and facts order",
2482
+ )
2483
+ raw = directory.read_member(
2484
+ completed["facts_sha256"], kind="facts", maximum=MAX_FACTS_MEMBER_BYTES
2485
+ )
2486
+ if raw is None or sha256_bytes(raw) != completed["facts_sha256"]:
2487
+ raise _refuse(
2488
+ "PACKED_JOURNAL_MEMBER",
2489
+ f"{path}.facts_sha256",
2490
+ "an already-emitted immutable member does not read back exactly",
2491
+ )
2492
+ try:
2493
+ payload = parse_canonical_json(raw)
2494
+ except CanonicalJSONError as error:
2495
+ raise _refuse(
2496
+ "PACKED_JOURNAL_RANGE", f"{path}.facts_sha256", "a facts member is not canonical"
2497
+ ) from error
2498
+ items = payload.get("entries") if isinstance(payload, dict) else None
2499
+ if not isinstance(items, list) or len(items) != len(range_input.entries):
2500
+ raise _refuse(
2501
+ "PACKED_JOURNAL_RANGE",
2502
+ f"{path}.facts_sha256",
2503
+ "a completed range's facts member does not cover this stream's entries",
2504
+ )
2505
+ for position, (item, entry) in enumerate(zip(items, range_input.entries, strict=True)):
2506
+ try:
2507
+ entry_sha256 = sha256_bytes(canonical_json_bytes(entry["entry"]))
2508
+ except CanonicalJSONError as error:
2509
+ raise _refuse(
2510
+ "PACKED_FACTS_LIMIT",
2511
+ f"packed.range.entries[{position}].entry",
2512
+ "a range document is not canonically representable",
2513
+ ) from error
2514
+ if (
2515
+ not isinstance(item, dict)
2516
+ or item.get("key") != entry["key"]
2517
+ or item.get("entry_id") != entry["entry_id"]
2518
+ or item.get("classification") != entry["classification"]
2519
+ or item.get("semantic_facts_digest") != entry["semantic_facts_digest"]
2520
+ or item.get("entry_sha256") != entry_sha256
2521
+ ):
2522
+ raise _refuse(
2523
+ "PACKED_JOURNAL_RANGE",
2524
+ f"{path}.entries[{position}]",
2525
+ "a completed range does not carry the entry this stream produces",
2526
+ )
2527
+
2528
+
2529
+ def _invocation_record(counters: _Counters, start: Mapping[str, int]) -> dict[str, int]:
2530
+ return {name: getattr(counters, name) - start[name] for name in COUNTER_MEMBERS}
2531
+
2532
+
2533
+ def _write_journal(
2534
+ directory: PackedDirectory,
2535
+ journal: _Journal,
2536
+ counters: _Counters,
2537
+ start: Mapping[str, int],
2538
+ *,
2539
+ status: str,
2540
+ head_sha256: str | None = None,
2541
+ ) -> str:
2542
+ document = _journal_document(journal, counters, start, status=status, head_sha256=head_sha256)
2543
+ return directory.publish(PACKED_JOURNAL_FILENAME, document, maximum=MAX_PACKED_JOURNAL_BYTES)
2544
+
2545
+
2546
+ def _journal_document(
2547
+ journal: _Journal,
2548
+ counters: _Counters,
2549
+ start: Mapping[str, int],
2550
+ *,
2551
+ status: str,
2552
+ head_sha256: str | None = None,
2553
+ ) -> dict[str, Any]:
2554
+ """Build the exact journal document before choosing when to install it."""
2555
+
2556
+ invocations = list(journal.invocations)
2557
+ if status == "incomplete":
2558
+ invocations.append(_invocation_record(counters, start))
2559
+ return _Journal(
2560
+ output_root=journal.output_root,
2561
+ generation=journal.generation,
2562
+ completed_ranges=journal.completed_ranges,
2563
+ invocations=invocations,
2564
+ ).document(status=status, head_sha256=head_sha256)
2565
+
2566
+
2567
+ def _verify_counter_equations(
2568
+ counters: _Counters, counts: Mapping[str, int], *, backend_count: int
2569
+ ) -> None:
2570
+ fresh_entries = counts["new"] + counts["changed"]
2571
+ layers = len(EMBEDDING_LAYERS)
2572
+ expected = {
2573
+ "batch_invocations": counts["fresh_ranges"] * layers * backend_count,
2574
+ "encoded_layer_items": fresh_entries * layers * backend_count,
2575
+ "backend_scalar_invocations": fresh_entries * layers * backend_count,
2576
+ "reused_layer_items": counts["unchanged"] * layers * backend_count,
2577
+ }
2578
+ for name, value in expected.items():
2579
+ if getattr(counters, name) != value:
2580
+ raise _refuse(
2581
+ "PACKED_COUNTER_MISMATCH",
2582
+ f"packed.counters.{name}",
2583
+ f"the publisher's own work does not satisfy its equation: "
2584
+ f"{getattr(counters, name)} != {value}",
2585
+ )
2586
+
2587
+
2588
+ def _terminal_document(
2589
+ *,
2590
+ provider_id: str,
2591
+ generation: Mapping[str, Any],
2592
+ coverage: Mapping[str, Any],
2593
+ delta_manifest: Mapping[str, Any],
2594
+ packed_backends: Sequence[PackedBackend],
2595
+ posting_backend: PackedBackend,
2596
+ counters: _Counters,
2597
+ counts: Mapping[str, int],
2598
+ invocations: Sequence[Mapping[str, int]],
2599
+ descriptors: Sequence[PackedRangeDescriptor],
2600
+ posting_term_count: int,
2601
+ vector_payload_sha256s: Sequence[str],
2602
+ posting_member_sha256s: Sequence[str],
2603
+ bound_member_sha256s: Sequence[str],
2604
+ limits: PackedLimits,
2605
+ residency: Mapping[str, int],
2606
+ journal: _Journal,
2607
+ ) -> dict[str, Any]:
2608
+ """Persist everything needed to finish evidence after the packed head is durable.
2609
+
2610
+ ``outputs.vector_member_sha256s`` is the ordered digest of every layer member's vector payload
2611
+ -- see :func:`_vector_payload_sha256s` -- so a successor that reused a predecessor's vectors
2612
+ states the same list the predecessor did, and one that silently re-encoded cannot. The posting
2613
+ and bound digests beside it are the independent final reader's, not the builder's: they name
2614
+ the accelerator members that reader recomputed from the pack bytes it read back.
2615
+
2616
+ ``adapter_mode`` is stated rather than inferred. Every vector in this generation came back on a
2617
+ :class:`BatchEncodeResult` that could only have been minted by the bounded scalar adapter, so
2618
+ the receipt says which implementation produced them instead of leaving a reader to assume.
2619
+ """
2620
+
2621
+ receipt = {
2622
+ "schema_version": PACKED_RECEIPT_SCHEMA,
2623
+ "status": "complete",
2624
+ "provider_id": provider_id,
2625
+ "adapter_mode": BOUNDED_SCALAR_ADAPTER,
2626
+ # The head and this receipt both state the coverage, and the installed-catalogue reader
2627
+ # requires them to agree. A duplicated summary nothing reconciles is the weak kind; this
2628
+ # one is compared member-for-member in `packed_retrieval._bound_receipt`.
2629
+ "coverage": dict(coverage),
2630
+ "inputs": {
2631
+ "delta_manifest_sha256": generation["delta_manifest_sha256"],
2632
+ "delta_root_sha256": delta_manifest["root_sha256"],
2633
+ "authoring_manifest_sha256": generation["authoring_manifest_sha256"],
2634
+ "identity_manifest_sha256": generation["identity_manifest_sha256"],
2635
+ "predecessor_head_sha256": generation["predecessor_head_sha256"],
2636
+ "publication_binding_sha256": generation.get("publication_binding_sha256"),
2637
+ },
2638
+ "backends": [backend.to_dict() for backend in packed_backends],
2639
+ "posting_backend_coordinate": posting_backend.coordinate,
2640
+ "counts": dict(counts),
2641
+ "counters": counters.to_dict(),
2642
+ "invocations": [dict(record) for record in invocations],
2643
+ "outputs": {
2644
+ "range_count": counts["ranges"],
2645
+ "entry_count": counts["entries"],
2646
+ "member_count": counts["members"],
2647
+ "posting_term_count": posting_term_count,
2648
+ "ranges": [descriptor.to_dict() for descriptor in descriptors],
2649
+ "ranges_chain_sha256": ranges_chain_sha256(descriptors),
2650
+ "vector_member_sha256s": list(vector_payload_sha256s),
2651
+ "posting_member_sha256s": list(posting_member_sha256s),
2652
+ "bound_member_sha256s": list(bound_member_sha256s),
2653
+ },
2654
+ "residency": dict(residency),
2655
+ "limits": limits.to_dict(),
2656
+ }
2657
+ body = {
2658
+ "schema_version": PACKED_TERMINAL_SCHEMA,
2659
+ "generation": dict(generation),
2660
+ "journal": {
2661
+ "output_root": journal.output_root,
2662
+ "generation": dict(journal.generation),
2663
+ "completed_ranges": list(journal.completed_ranges),
2664
+ "invocations": [dict(record) for record in invocations],
2665
+ },
2666
+ "receipt": receipt,
2667
+ }
2668
+ return {**body, "root_sha256": canonical_sha256(body)}
2669
+
2670
+
2671
+ def _receipt_from_terminal(
2672
+ terminal: Mapping[str, Any], *, head_sha256: str, journal_sha256: str
2673
+ ) -> dict[str, Any]:
2674
+ """Complete the packed receipt from the head-bound terminal record."""
2675
+
2676
+ stated = terminal.get("receipt")
2677
+ if not isinstance(stated, Mapping):
2678
+ raise _refuse(
2679
+ "PACKED_TERMINAL_SCHEMA", "packed.terminal.receipt", "terminal receipt differs"
2680
+ )
2681
+ outputs = stated.get("outputs")
2682
+ if not isinstance(outputs, Mapping):
2683
+ raise _refuse("PACKED_TERMINAL_SCHEMA", "packed.terminal.receipt.outputs", "outputs differ")
2684
+ body = {
2685
+ **dict(stated),
2686
+ "outputs": {
2687
+ "head_sha256": head_sha256,
2688
+ "journal_sha256": journal_sha256,
2689
+ **dict(outputs),
2690
+ },
2691
+ }
2692
+ return {**body, "coordinate_sha256": canonical_sha256(body)}
2693
+
2694
+
2695
+ def _parse_terminal(raw: bytes) -> dict[str, Any]:
2696
+ """Read one head-bound terminal record without accepting an alternate spelling."""
2697
+
2698
+ if len(raw) > MAX_PACKED_TERMINAL_BYTES:
2699
+ raise _refuse("PACKED_TERMINAL_LIMIT", "packed.terminal", "terminal record exceeds its cap")
2700
+ try:
2701
+ terminal = parse_canonical_json(raw)
2702
+ except CanonicalJSONError as error:
2703
+ raise _refuse(
2704
+ "PACKED_TERMINAL_SCHEMA", "packed.terminal", "terminal record is not canonical"
2705
+ ) from error
2706
+ if (
2707
+ not isinstance(terminal, dict)
2708
+ or set(terminal) != {"schema_version", "generation", "journal", "receipt", "root_sha256"}
2709
+ or terminal.get("schema_version") != PACKED_TERMINAL_SCHEMA
2710
+ or not isinstance(terminal.get("generation"), dict)
2711
+ or not isinstance(terminal.get("journal"), dict)
2712
+ or not isinstance(terminal.get("receipt"), dict)
2713
+ or not _is_digest(terminal.get("root_sha256"))
2714
+ ):
2715
+ raise _refuse(
2716
+ "PACKED_TERMINAL_SCHEMA", "packed.terminal", "terminal record contract differs"
2717
+ )
2718
+ body = {key: value for key, value in terminal.items() if key != "root_sha256"}
2719
+ if canonical_sha256(body) != terminal["root_sha256"]:
2720
+ raise _refuse(
2721
+ "PACKED_TERMINAL_DIGEST",
2722
+ "packed.terminal.root_sha256",
2723
+ "terminal record does not reproduce its own digest",
2724
+ )
2725
+ return terminal
2726
+
2727
+
2728
+ def verify_packed_receipt(
2729
+ receipt: Mapping[str, Any], *, root: Path, root_descriptor: int | None = None
2730
+ ) -> None:
2731
+ """Refuse a receipt that is not the exact coordinate of the generation installed at ``root``."""
2732
+
2733
+ if not isinstance(receipt, Mapping) or receipt.get("schema_version") != PACKED_RECEIPT_SCHEMA:
2734
+ raise _refuse("PACKED_RECEIPT_SCHEMA", "packed.receipt", "receipt contract differs")
2735
+ if receipt.get("adapter_mode") != BOUNDED_SCALAR_ADAPTER:
2736
+ raise _refuse(
2737
+ "PACKED_RECEIPT_ADAPTER",
2738
+ "packed.receipt.adapter_mode",
2739
+ f"a packed generation may claim only the {BOUNDED_SCALAR_ADAPTER!r} adapter",
2740
+ )
2741
+ body = {key: value for key, value in receipt.items() if key != "coordinate_sha256"}
2742
+ if canonical_sha256(body) != receipt.get("coordinate_sha256"):
2743
+ raise _refuse(
2744
+ "PACKED_RECEIPT_COORDINATE",
2745
+ "packed.receipt.coordinate_sha256",
2746
+ "the receipt does not reproduce its own coordinate",
2747
+ )
2748
+ installed = verify_packed_generation(
2749
+ Path(root),
2750
+ expected_head_sha256=receipt["outputs"]["head_sha256"],
2751
+ root_descriptor=root_descriptor,
2752
+ )
2753
+ directory = PackedDirectory(Path(root), root_descriptor=root_descriptor)
2754
+ with directory.opened():
2755
+ terminal_raw = directory.read(PACKED_TERMINAL_FILENAME, maximum=MAX_PACKED_TERMINAL_BYTES)
2756
+ journal_raw = directory.read(PACKED_JOURNAL_FILENAME, maximum=MAX_PACKED_JOURNAL_BYTES)
2757
+ if (
2758
+ terminal_raw is None
2759
+ or journal_raw is None
2760
+ or sha256_bytes(terminal_raw) != installed.head["terminal_sha256"]
2761
+ or sha256_bytes(journal_raw) != receipt["outputs"]["journal_sha256"]
2762
+ ):
2763
+ raise _refuse(
2764
+ "PACKED_RECEIPT_MISMATCH",
2765
+ "packed.receipt.outputs",
2766
+ "the receipt does not bind the installed terminal record and journal",
2767
+ )
2768
+ terminal = _parse_terminal(terminal_raw)
2769
+ journal = _read_journal(directory)
2770
+ if (
2771
+ journal is None
2772
+ or journal.get("status") != "complete"
2773
+ or journal.get("head_sha256") != installed.head_sha256
2774
+ or _receipt_from_terminal(
2775
+ terminal,
2776
+ head_sha256=installed.head_sha256,
2777
+ journal_sha256=sha256_bytes(journal_raw),
2778
+ )
2779
+ != dict(receipt)
2780
+ ):
2781
+ raise _refuse(
2782
+ "PACKED_RECEIPT_MISMATCH",
2783
+ "packed.receipt",
2784
+ "the receipt is not the head-bound terminal publication",
2785
+ )
2786
+ if (
2787
+ receipt.get("outputs")
2788
+ != {
2789
+ "head_sha256": installed.head_sha256,
2790
+ "journal_sha256": receipt["outputs"]["journal_sha256"],
2791
+ **_verified_receipt_outputs(installed),
2792
+ }
2793
+ or installed.head["counters"] != receipt["counters"]
2794
+ or installed.head["backends"] != receipt.get("backends")
2795
+ or installed.head["posting_backend_coordinate"] != receipt.get("posting_backend_coordinate")
2796
+ or installed.head["provider_id"] != receipt.get("provider_id")
2797
+ ):
2798
+ raise _refuse(
2799
+ "PACKED_RECEIPT_MISMATCH",
2800
+ "packed.receipt.outputs",
2801
+ "the receipt does not describe the generation installed at this root",
2802
+ )