mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,2345 @@
1
+ """Range-local packed catalogs: fixed-width vector packs, exact postings, conservative bounds.
2
+
3
+ The v2 retrieval manifest in :mod:`retrieval_manifest` describes one representation as a flat member
4
+ list. At provider scale that shape stops working: four corpus-wide packs would each be gigabytes,
5
+ no reader could bound its own allocation from a descriptor, and a single mutated byte anywhere would
6
+ be indistinguishable from a legitimate republication. This module owns packed generations; the
7
+ flat v1/v2 contract and its constants remain scoped to :mod:`retrieval_manifest`.
8
+
9
+ *A range is the unit.* Sorted entries are grouped into consecutive ranges of exactly
10
+ ``RANGE_ENTRIES`` -- only the final range is short -- and each range owns one facts member, one
11
+ history member, four layer-specific vector members per available backend, one to eight lexical
12
+ posting segments, and one bound segment per backend. There are never four corpus-wide packs.
13
+
14
+ *Every member has an exact byte law.* A vector member is exactly
15
+ ``64 + entry_count * dimension * 4`` bytes; a bound segment is exactly
16
+ ``64 + 4 * dimension * 2 * 4`` bytes; a posting term is exactly ten bytes. That is what lets a
17
+ reader refuse a member from its 64-byte header before it allocates anything the member claims to
18
+ contain, and it is why the ceilings in this module are laws rather than guidance: a 1000-entry
19
+ range at the 384-dimension ceiling is 1,536,064 bytes per layer member and cannot be anything
20
+ else.
21
+
22
+ *Nothing derived is trusted.* Posting segments and bound segments are accelerators, and an
23
+ accelerator that lies changes which entries a bounded search ever looks at. So
24
+ :func:`verify_packed_generation` recomputes both from the exact vector member bytes and refuses a
25
+ summary that is not the publisher's exact one. It refuses a *non-conservative* bound with its own
26
+ code, because a bound that is too narrow hides a true hit while a merely wrong one only wastes a
27
+ scan.
28
+
29
+ *The bound is the one number this module computes.* Exact entry scores belong to the ranker and die
30
+ there. What a published range can carry is the summary side of that score:
31
+ :func:`conservative_upper_bound` is ``sum_i(q_i * max_i)`` where ``q_i >= 0`` and
32
+ ``sum_i(q_i * min_i)`` otherwise, and :func:`range_upper_bound` maximizes it across the four
33
+ layers. Pruning on it must be strict -- an equal bound still exact-scans, because a tie is broken
34
+ by entry id and a pruned range cannot present its ids.
35
+
36
+ Fixed-endian tables and canonical JSON only. No pickle, no memory mapping, no length a reader
37
+ learns after it has already allocated.
38
+ """
39
+
40
+ from __future__ import annotations
41
+
42
+ import contextlib
43
+ import os
44
+ import re
45
+ import secrets
46
+ import stat
47
+ import struct
48
+ from collections.abc import Iterator, Mapping, Sequence
49
+ from dataclasses import dataclass, replace
50
+ from pathlib import Path
51
+ from typing import Any
52
+
53
+ from mostlyright.data_harness.canonical import (
54
+ CanonicalJSONError,
55
+ canonical_json_bytes,
56
+ canonical_sha256,
57
+ parse_canonical_json,
58
+ sha256_bytes,
59
+ )
60
+ from mostlyright.data_harness.sources.catalog.bounded_io import (
61
+ BoundedReadFailure,
62
+ read_bounded_at,
63
+ )
64
+ from mostlyright.data_harness.sources.catalog.contracts import EMBEDDING_LAYERS
65
+ from mostlyright.data_harness.sources.catalog.coverage import coverage_is_valid
66
+ from mostlyright.data_harness.sources.catalog.embedding import BatchEncodeResult, EmbeddingBackend
67
+ from mostlyright.data_harness.sources.contracts import SourceContractError
68
+
69
+ PACKED_HEAD_SCHEMA = "harness-catalog-packed-head.v1"
70
+ PACKED_RANGE_MANIFEST_SCHEMA = "harness-catalog-packed-range.v1"
71
+ PACKED_FACTS_SCHEMA = "harness-catalog-packed-facts.v1"
72
+ PACKED_HISTORY_SCHEMA = "harness-catalog-packed-history.v1"
73
+
74
+ PACKED_HEAD_FILENAME = "packed-head.json"
75
+ PACKED_MEMBERS_DIRNAME = "members"
76
+
77
+ #: One range is exactly this many entries. Only the final range of a generation is shorter, and a
78
+ #: publisher that would rather split a range than refuse an oversized member changes the call bound
79
+ #: the whole design rests on -- so it refuses instead.
80
+ RANGE_ENTRIES = 1_000
81
+
82
+ MAX_RANGE_DESCRIPTORS = 1_024
83
+ MAX_RANGE_DESCRIPTOR_BYTES = 2_048
84
+ MAX_PACKED_HEAD_BYTES = 4 * 1024 * 1024
85
+
86
+ MAX_FACTS_MEMBER_BYTES = 8 * 1024 * 1024
87
+ MAX_HISTORY_MEMBER_BYTES = 8 * 1024 * 1024
88
+
89
+ MAX_RANGE_MANIFEST_BYTES = 64 * 1024
90
+ MAX_RANGE_MEMBER_DESCRIPTORS = 24
91
+ MAX_MEMBER_DESCRIPTOR_BYTES = 1_024
92
+
93
+ #: The widest backend this format admits. 384 is the MiniLM coordinate's dimension, and the number
94
+ #: is a law rather than a default: it is what makes the vector-member ceiling a constant.
95
+ MAX_BACKEND_DIMENSION = 384
96
+ MEMBER_HEADER_BYTES = 64
97
+ MAX_VECTOR_MEMBER_BYTES = MEMBER_HEADER_BYTES + RANGE_ENTRIES * MAX_BACKEND_DIMENSION * 4
98
+
99
+ BOUND_SEGMENT_LAYERS = len(EMBEDDING_LAYERS)
100
+ MIN_POSTING_SEGMENTS = 1
101
+ MAX_POSTING_SEGMENTS = 8
102
+ MAX_POSTING_SEGMENT_BYTES = 4 * 1024 * 1024
103
+ POSTING_TERM_FORMAT = ">HHBxi"
104
+ POSTING_TERM_BYTES = struct.calcsize(POSTING_TERM_FORMAT)
105
+ MAX_POSTING_SEGMENT_TERMS = (MAX_POSTING_SEGMENT_BYTES - MEMBER_HEADER_BYTES) // POSTING_TERM_BYTES
106
+
107
+ #: Two backends (one query-safe lexical, one governed neural) is what the 24-descriptor range
108
+ #: manifest admits alongside eight posting segments. A third would not fit, and widening the
109
+ #: manifest to make it fit would widen every reader's allocation ceiling.
110
+ MAX_PACKED_BACKENDS = 2
111
+
112
+ PACKED_CLASSIFICATIONS = ("new", "changed", "unchanged")
113
+ MEMBER_KINDS = ("facts", "history", "vector", "posting", "bound", "range")
114
+ _BINARY_KINDS = ("vector", "posting", "bound")
115
+
116
+ MIN_INT32 = -(1 << 31)
117
+ MAX_INT32 = (1 << 31) - 1
118
+
119
+ PACKED_FORMAT_VERSION = 1
120
+ MEMBER_MAGIC = b"MRPACKD1"
121
+ _HEADER_FORMAT = ">8sHHHHIIII16s16s"
122
+ _KIND_CODES = {"vector": 1, "posting": 2, "bound": 3}
123
+ _KIND_NAMES = {code: name for name, code in _KIND_CODES.items()}
124
+ SUBINDEX_NONE = 0xFFFF
125
+
126
+ _OPEN_DIRECTORY = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0)
127
+ _OPEN_MEMBER = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0)
128
+
129
+ _SHA256 = re.compile(r"^[0-9a-f]{64}$")
130
+
131
+
132
+ class CatalogPackedRefused(SourceContractError):
133
+ """A stable refusal of a packed member, descriptor, manifest, head, or durable output."""
134
+
135
+
136
+ def _refuse(code: str, path: str, detail: str) -> CatalogPackedRefused:
137
+ return CatalogPackedRefused(code, path, detail)
138
+
139
+
140
+ def _is_digest(value: Any) -> bool:
141
+ return isinstance(value, str) and _SHA256.fullmatch(value) is not None
142
+
143
+
144
+ # ------------------------------------------------------------------------------------------
145
+ # Exact byte laws
146
+ # ------------------------------------------------------------------------------------------
147
+
148
+
149
+ def vector_member_bytes(*, entry_count: int, dimensions: int) -> int:
150
+ """The exact size of one layer-specific vector member. There is no other admissible size."""
151
+
152
+ return MEMBER_HEADER_BYTES + entry_count * dimensions * 4
153
+
154
+
155
+ def bound_segment_bytes(*, dimensions: int) -> int:
156
+ """The exact size of one backend's bound segment: four layers of per-dimension min and max."""
157
+
158
+ return MEMBER_HEADER_BYTES + BOUND_SEGMENT_LAYERS * dimensions * 2 * 4
159
+
160
+
161
+ def member_filename(kind: str, sha256: str) -> str:
162
+ """Content-addressed member name. Binary tables and canonical JSON never share an extension."""
163
+
164
+ if kind not in MEMBER_KINDS:
165
+ raise _refuse("PACKED_MEMBER_KIND", "packed.member.kind", f"unknown member kind {kind!r}")
166
+ if not _is_digest(sha256):
167
+ raise _refuse("PACKED_MEMBER_DIGEST", "packed.member.sha256", "must be a lowercase SHA-256")
168
+ return f"{sha256}.bin" if kind in _BINARY_KINDS else f"{sha256}.json"
169
+
170
+
171
+ def _check_dimensions(dimensions: Any, path: str) -> int:
172
+ if type(dimensions) is not int or not 1 <= dimensions <= MAX_BACKEND_DIMENSION:
173
+ raise _refuse(
174
+ "PACKED_MEMBER_DIMENSION",
175
+ path,
176
+ f"a packed backend dimension must be an integer in [1, {MAX_BACKEND_DIMENSION}]",
177
+ )
178
+ return dimensions
179
+
180
+
181
+ def _check_entry_count(entry_count: Any, path: str) -> int:
182
+ if type(entry_count) is not int or not 1 <= entry_count <= RANGE_ENTRIES:
183
+ raise _refuse(
184
+ "PACKED_MEMBER_ENTRY_COUNT",
185
+ path,
186
+ f"a range carries 1 to {RANGE_ENTRIES} entries",
187
+ )
188
+ return entry_count
189
+
190
+
191
+ # ------------------------------------------------------------------------------------------
192
+ # The 64-byte fixed-endian member header
193
+ # ------------------------------------------------------------------------------------------
194
+
195
+
196
+ @dataclass(frozen=True)
197
+ class PackedMemberHeader:
198
+ """Everything a reader must know before it allocates one byte of a member's payload."""
199
+
200
+ kind: str
201
+ subindex: int
202
+ dimensions: int
203
+ entry_count: int
204
+ term_count: int
205
+ range_index: int
206
+ payload_bytes: int
207
+ binding_prefix: bytes
208
+ backend_prefix: bytes
209
+
210
+
211
+ def _binding_prefix(binding_sha256: str) -> bytes:
212
+ if not _is_digest(binding_sha256):
213
+ raise _refuse(
214
+ "PACKED_MEMBER_BINDING",
215
+ "packed.member.range_binding_sha256",
216
+ "a member binds its range by that range's exact binding digest",
217
+ )
218
+ return bytes.fromhex(binding_sha256)[:16]
219
+
220
+
221
+ def _backend_prefix(backend_coordinate: str) -> bytes:
222
+ if not isinstance(backend_coordinate, str) or not backend_coordinate:
223
+ raise _refuse(
224
+ "PACKED_MEMBER_BINDING",
225
+ "packed.member.backend_coordinate",
226
+ "a member binds the exact backend coordinate that produced it",
227
+ )
228
+ return bytes.fromhex(sha256_bytes(backend_coordinate.encode("utf-8")))[:16]
229
+
230
+
231
+ def pack_member_header(
232
+ *,
233
+ kind: str,
234
+ subindex: int,
235
+ dimensions: int,
236
+ entry_count: int,
237
+ term_count: int,
238
+ range_index: int,
239
+ payload_bytes: int,
240
+ binding_sha256: str,
241
+ backend_coordinate: str,
242
+ ) -> bytes:
243
+ """Build one exact 64-byte member header. Fixed big-endian, no padding a reader must guess."""
244
+
245
+ if kind not in _KIND_CODES:
246
+ raise _refuse("PACKED_MEMBER_KIND", "packed.member.kind", f"unknown member kind {kind!r}")
247
+ if type(range_index) is not int or not 0 <= range_index < MAX_RANGE_DESCRIPTORS:
248
+ raise _refuse(
249
+ "PACKED_MEMBER_HEADER",
250
+ "packed.member.range_index",
251
+ f"a range index must be an integer in [0, {MAX_RANGE_DESCRIPTORS - 1}]",
252
+ )
253
+ if type(subindex) is not int or not 0 <= subindex <= SUBINDEX_NONE:
254
+ raise _refuse(
255
+ "PACKED_MEMBER_HEADER", "packed.member.subindex", "a member subindex is a uint16"
256
+ )
257
+ raw = struct.pack(
258
+ _HEADER_FORMAT,
259
+ MEMBER_MAGIC,
260
+ PACKED_FORMAT_VERSION,
261
+ _KIND_CODES[kind],
262
+ subindex,
263
+ dimensions,
264
+ entry_count,
265
+ term_count,
266
+ range_index,
267
+ payload_bytes,
268
+ _binding_prefix(binding_sha256),
269
+ _backend_prefix(backend_coordinate),
270
+ )
271
+ if len(raw) != MEMBER_HEADER_BYTES: # pragma: no cover - the format is fixed
272
+ raise _refuse("PACKED_MEMBER_HEADER", "packed.member", "header is not exactly 64 bytes")
273
+ return raw
274
+
275
+
276
+ def parse_member_header(raw: bytes) -> PackedMemberHeader:
277
+ """Read the fixed header, or refuse. Nothing is allocated from a value read after this."""
278
+
279
+ if not isinstance(raw, bytes | bytearray | memoryview):
280
+ raise _refuse("PACKED_MEMBER_HEADER", "packed.member", "a member is raw bytes")
281
+ raw = bytes(raw)
282
+ if len(raw) < MEMBER_HEADER_BYTES:
283
+ raise _refuse(
284
+ "PACKED_MEMBER_TRUNCATED",
285
+ "packed.member",
286
+ f"a member is at least its {MEMBER_HEADER_BYTES}-byte header",
287
+ )
288
+ (
289
+ magic,
290
+ version,
291
+ kind_code,
292
+ subindex,
293
+ dimensions,
294
+ entry_count,
295
+ term_count,
296
+ range_index,
297
+ payload_bytes,
298
+ binding_prefix,
299
+ backend_prefix,
300
+ ) = struct.unpack(_HEADER_FORMAT, raw[:MEMBER_HEADER_BYTES])
301
+ if magic != MEMBER_MAGIC:
302
+ raise _refuse("PACKED_MEMBER_MAGIC", "packed.member.magic", "not a packed catalog member")
303
+ if version != PACKED_FORMAT_VERSION:
304
+ raise _refuse(
305
+ "PACKED_MEMBER_VERSION",
306
+ "packed.member.format_version",
307
+ f"this reader reads format version {PACKED_FORMAT_VERSION} only",
308
+ )
309
+ if kind_code not in _KIND_NAMES:
310
+ raise _refuse("PACKED_MEMBER_KIND", "packed.member.kind", "unknown member kind code")
311
+ _check_dimensions(dimensions, "packed.member.dimensions")
312
+ _check_entry_count(entry_count, "packed.member.entry_count")
313
+ if range_index >= MAX_RANGE_DESCRIPTORS:
314
+ raise _refuse(
315
+ "PACKED_MEMBER_HEADER", "packed.member.range_index", "range index exceeds its bound"
316
+ )
317
+ return PackedMemberHeader(
318
+ kind=_KIND_NAMES[kind_code],
319
+ subindex=subindex,
320
+ dimensions=dimensions,
321
+ entry_count=entry_count,
322
+ term_count=term_count,
323
+ range_index=range_index,
324
+ payload_bytes=payload_bytes,
325
+ binding_prefix=binding_prefix,
326
+ backend_prefix=backend_prefix,
327
+ )
328
+
329
+
330
+ def _payload(raw: bytes, header: PackedMemberHeader, *, expected_payload: int) -> bytes:
331
+ if header.payload_bytes != expected_payload:
332
+ raise _refuse(
333
+ "PACKED_MEMBER_HEADER",
334
+ "packed.member.payload_bytes",
335
+ "the declared payload length disagrees with this member's exact byte law",
336
+ )
337
+ total = MEMBER_HEADER_BYTES + expected_payload
338
+ if len(raw) < total:
339
+ raise _refuse(
340
+ "PACKED_MEMBER_TRUNCATED",
341
+ "packed.member",
342
+ "the member ends before its declared payload",
343
+ )
344
+ if len(raw) > total:
345
+ raise _refuse(
346
+ "PACKED_MEMBER_EXCESS",
347
+ "packed.member",
348
+ "a member is exactly its header plus its payload; trailing bytes are refused",
349
+ )
350
+ return raw[MEMBER_HEADER_BYTES:]
351
+
352
+
353
+ def _check_binding(
354
+ header: PackedMemberHeader,
355
+ descriptor: PackedMemberDescriptor | None,
356
+ *,
357
+ kind: str,
358
+ ) -> None:
359
+ if header.kind != kind:
360
+ raise _refuse(
361
+ "PACKED_MEMBER_KIND",
362
+ "packed.member.kind",
363
+ f"this member is a {header.kind} member, not a {kind} member",
364
+ )
365
+ if descriptor is None:
366
+ return
367
+ if descriptor.kind != kind:
368
+ raise _refuse(
369
+ "PACKED_MEMBER_KIND", "packed.member.kind", "the descriptor names another member kind"
370
+ )
371
+ if kind == "vector" and descriptor.layer != EMBEDDING_LAYERS[header.subindex]:
372
+ raise _refuse(
373
+ "PACKED_MEMBER_LAYER",
374
+ "packed.member.layer",
375
+ "this member carries another layer than the descriptor binds",
376
+ )
377
+ if kind == "posting" and descriptor.segment_index != header.subindex:
378
+ raise _refuse(
379
+ "PACKED_MEMBER_BINDING",
380
+ "packed.member.segment_index",
381
+ "this posting segment is not the segment the descriptor binds",
382
+ )
383
+ if (
384
+ descriptor.range_index != header.range_index
385
+ or descriptor.entry_count != header.entry_count
386
+ or descriptor.dimensions != header.dimensions
387
+ or _binding_prefix(descriptor.range_binding_sha256) != header.binding_prefix
388
+ or _backend_prefix(descriptor.backend_coordinate or "") != header.backend_prefix
389
+ ):
390
+ raise _refuse(
391
+ "PACKED_MEMBER_BINDING",
392
+ "packed.member",
393
+ "this member is bound to another range, entry order, or backend",
394
+ )
395
+
396
+
397
+ # ------------------------------------------------------------------------------------------
398
+ # Vector members
399
+ # ------------------------------------------------------------------------------------------
400
+
401
+
402
+ @dataclass(frozen=True)
403
+ class PackedVectorMember:
404
+ """One layer's exact integer vectors for one range and one backend, in range entry order."""
405
+
406
+ layer: str
407
+ dimensions: int
408
+ entry_count: int
409
+ range_index: int
410
+ vectors: tuple[tuple[int, ...], ...]
411
+
412
+
413
+ def pack_vector_member(
414
+ vectors: Sequence[Sequence[int]],
415
+ *,
416
+ layer: str,
417
+ dimensions: int,
418
+ range_index: int,
419
+ binding_sha256: str,
420
+ backend_coordinate: str,
421
+ ) -> bytes:
422
+ """Serialize one layer's vectors as a fixed-width big-endian int32 table."""
423
+
424
+ _check_dimensions(dimensions, "packed.vector.dimensions")
425
+ if layer not in EMBEDDING_LAYERS:
426
+ raise _refuse(
427
+ "PACKED_MEMBER_LAYER",
428
+ "packed.vector.layer",
429
+ f"a layer member names one of {list(EMBEDDING_LAYERS)}",
430
+ )
431
+ entry_count = _check_entry_count(len(vectors), "packed.vector.entry_count")
432
+ payload = bytearray()
433
+ for ordinal, vector in enumerate(vectors):
434
+ if len(vector) != dimensions:
435
+ raise _refuse(
436
+ "PACKED_MEMBER_DIMENSION",
437
+ f"packed.vector[{ordinal}]",
438
+ "every vector in a member has exactly the backend's dimension",
439
+ )
440
+ for value in vector:
441
+ if type(value) is not int or not MIN_INT32 <= value <= MAX_INT32:
442
+ raise _refuse(
443
+ "PACKED_VECTOR_VALUE",
444
+ f"packed.vector[{ordinal}]",
445
+ "a packed vector holds exact int32 values only",
446
+ )
447
+ payload += struct.pack(f">{dimensions}i", *vector)
448
+ header = pack_member_header(
449
+ kind="vector",
450
+ subindex=EMBEDDING_LAYERS.index(layer),
451
+ dimensions=dimensions,
452
+ entry_count=entry_count,
453
+ term_count=0,
454
+ range_index=range_index,
455
+ payload_bytes=len(payload),
456
+ binding_sha256=binding_sha256,
457
+ backend_coordinate=backend_coordinate,
458
+ )
459
+ raw = header + bytes(payload)
460
+ if len(raw) != vector_member_bytes(entry_count=entry_count, dimensions=dimensions):
461
+ raise _refuse( # pragma: no cover - the arithmetic above is the law
462
+ "PACKED_MEMBER_HEADER", "packed.vector", "vector member violated its exact byte law"
463
+ )
464
+ if len(raw) > MAX_VECTOR_MEMBER_BYTES: # pragma: no cover - implied by the two ceilings
465
+ raise _refuse("PACKED_MEMBER_LIMIT", "packed.vector", "vector member exceeds its ceiling")
466
+ return raw
467
+
468
+
469
+ def parse_vector_member(
470
+ raw: bytes, *, descriptor: PackedMemberDescriptor | None = None
471
+ ) -> PackedVectorMember:
472
+ """Read one vector member exactly, or refuse it. Never partially."""
473
+
474
+ header = parse_member_header(raw)
475
+ _check_binding(header, descriptor, kind="vector")
476
+ if header.subindex >= len(EMBEDDING_LAYERS):
477
+ raise _refuse(
478
+ "PACKED_MEMBER_LAYER", "packed.vector.layer", "layer ordinal is outside layer closure"
479
+ )
480
+ if header.term_count != 0:
481
+ raise _refuse(
482
+ "PACKED_MEMBER_HEADER", "packed.vector.term_count", "a vector member carries no terms"
483
+ )
484
+ expected = header.entry_count * header.dimensions * 4
485
+ payload = _payload(bytes(raw), header, expected_payload=expected)
486
+ vectors: list[tuple[int, ...]] = []
487
+ stride = header.dimensions * 4
488
+ for ordinal in range(header.entry_count):
489
+ chunk = payload[ordinal * stride : (ordinal + 1) * stride]
490
+ vectors.append(struct.unpack(f">{header.dimensions}i", chunk))
491
+ return PackedVectorMember(
492
+ layer=EMBEDDING_LAYERS[header.subindex],
493
+ dimensions=header.dimensions,
494
+ entry_count=header.entry_count,
495
+ range_index=header.range_index,
496
+ vectors=tuple(vectors),
497
+ )
498
+
499
+
500
+ # ------------------------------------------------------------------------------------------
501
+ # Lexical posting segments
502
+ # ------------------------------------------------------------------------------------------
503
+
504
+
505
+ PostingTerm = tuple[int, int, int, int]
506
+
507
+
508
+ @dataclass(frozen=True)
509
+ class PackedPostingSegment:
510
+ """One bounded run of canonical ``(dimension, entry ordinal, layer ordinal, value)`` terms."""
511
+
512
+ segment_index: int
513
+ dimensions: int
514
+ entry_count: int
515
+ range_index: int
516
+ terms: tuple[PostingTerm, ...]
517
+
518
+
519
+ def compute_layer_postings(
520
+ vectors_by_layer: Mapping[str, Sequence[Sequence[int]]], *, dimensions: int
521
+ ) -> tuple[PostingTerm, ...]:
522
+ """Derive this range's canonical posting terms from its exact layer vectors.
523
+
524
+ A zero contributes nothing to any dot product, so it is not a term. Everything else is, and the
525
+ order is total: dimension, then entry ordinal, then layer ordinal.
526
+ """
527
+
528
+ _check_dimensions(dimensions, "packed.postings.dimensions")
529
+ terms: list[PostingTerm] = []
530
+ for layer_ordinal, layer in enumerate(EMBEDDING_LAYERS):
531
+ vectors = vectors_by_layer.get(layer)
532
+ if vectors is None:
533
+ raise _refuse(
534
+ "PACKED_MEMBER_LAYER",
535
+ "packed.postings.layers",
536
+ f"postings require exactly the four layers {list(EMBEDDING_LAYERS)}",
537
+ )
538
+ for ordinal, vector in enumerate(vectors):
539
+ for dimension, value in enumerate(vector):
540
+ if value:
541
+ terms.append((dimension, ordinal, layer_ordinal, value))
542
+ terms.sort()
543
+ return tuple(terms)
544
+
545
+
546
+ def segment_posting_terms(
547
+ terms: Sequence[PostingTerm], *, max_terms_per_segment: int | None = None
548
+ ) -> tuple[tuple[PostingTerm, ...], ...]:
549
+ """Split canonical terms into 1-8 bounded segments, or refuse rather than widen the cap."""
550
+
551
+ limit = MAX_POSTING_SEGMENT_TERMS if max_terms_per_segment is None else max_terms_per_segment
552
+ if type(limit) is not int or not 1 <= limit <= MAX_POSTING_SEGMENT_TERMS:
553
+ raise _refuse(
554
+ "PACKED_POSTING_SEGMENTS",
555
+ "packed.postings.max_terms_per_segment",
556
+ f"a segment carries 1 to {MAX_POSTING_SEGMENT_TERMS} terms",
557
+ )
558
+ segments = tuple(
559
+ tuple(terms[start : start + limit]) for start in range(0, max(len(terms), 1), limit)
560
+ )
561
+ if not MIN_POSTING_SEGMENTS <= len(segments) <= MAX_POSTING_SEGMENTS:
562
+ raise _refuse(
563
+ "PACKED_POSTING_SEGMENTS",
564
+ "packed.postings.segments",
565
+ f"a range carries {MIN_POSTING_SEGMENTS} to {MAX_POSTING_SEGMENTS} posting segments; "
566
+ f"{len(segments)} would be needed",
567
+ )
568
+ return segments
569
+
570
+
571
+ def pack_posting_segment(
572
+ terms: Sequence[PostingTerm],
573
+ *,
574
+ segment_index: int,
575
+ dimensions: int,
576
+ entry_count: int,
577
+ range_index: int,
578
+ binding_sha256: str,
579
+ backend_coordinate: str,
580
+ ) -> bytes:
581
+ """Serialize one posting segment as a fixed-width big-endian term table."""
582
+
583
+ _check_dimensions(dimensions, "packed.postings.dimensions")
584
+ _check_entry_count(entry_count, "packed.postings.entry_count")
585
+ _validate_posting_terms(terms, dimensions=dimensions, entry_count=entry_count)
586
+ payload = bytearray()
587
+ for dimension, ordinal, layer_ordinal, value in terms:
588
+ payload += struct.pack(POSTING_TERM_FORMAT, dimension, ordinal, layer_ordinal, value)
589
+ if MEMBER_HEADER_BYTES + len(payload) > MAX_POSTING_SEGMENT_BYTES:
590
+ raise _refuse(
591
+ "PACKED_POSTING_LIMIT",
592
+ "packed.postings.segment",
593
+ f"one posting segment holds at most {MAX_POSTING_SEGMENT_BYTES} bytes",
594
+ )
595
+ header = pack_member_header(
596
+ kind="posting",
597
+ subindex=segment_index,
598
+ dimensions=dimensions,
599
+ entry_count=entry_count,
600
+ term_count=len(terms),
601
+ range_index=range_index,
602
+ payload_bytes=len(payload),
603
+ binding_sha256=binding_sha256,
604
+ backend_coordinate=backend_coordinate,
605
+ )
606
+ return header + bytes(payload)
607
+
608
+
609
+ def _validate_posting_terms(
610
+ terms: Sequence[PostingTerm], *, dimensions: int, entry_count: int
611
+ ) -> None:
612
+ previous: PostingTerm | None = None
613
+ for position, term in enumerate(terms):
614
+ if len(term) != 4:
615
+ raise _refuse(
616
+ "PACKED_POSTING_TERM", f"packed.postings[{position}]", "a term is exactly four ints"
617
+ )
618
+ dimension, ordinal, layer_ordinal, value = term
619
+ if (
620
+ type(dimension) is not int
621
+ or not 0 <= dimension < dimensions
622
+ or type(ordinal) is not int
623
+ or not 0 <= ordinal < entry_count
624
+ or type(layer_ordinal) is not int
625
+ or not 0 <= layer_ordinal < len(EMBEDDING_LAYERS)
626
+ or type(value) is not int
627
+ or not MIN_INT32 <= value <= MAX_INT32
628
+ or value == 0
629
+ ):
630
+ raise _refuse(
631
+ "PACKED_POSTING_TERM",
632
+ f"packed.postings[{position}]",
633
+ "a term names an in-range dimension, entry ordinal, layer ordinal and int32",
634
+ )
635
+ if previous is not None and term <= previous:
636
+ raise _refuse(
637
+ "PACKED_POSTING_ORDER",
638
+ f"packed.postings[{position}]",
639
+ "posting terms ascend by dimension, entry ordinal, then layer ordinal",
640
+ )
641
+ previous = term
642
+
643
+
644
+ def parse_posting_segment(
645
+ raw: bytes, *, descriptor: PackedMemberDescriptor | None = None
646
+ ) -> PackedPostingSegment:
647
+ """Read one posting segment exactly, refusing a term outside the range's own closure."""
648
+
649
+ header = parse_member_header(raw)
650
+ _check_binding(header, descriptor, kind="posting")
651
+ if len(raw) > MAX_POSTING_SEGMENT_BYTES:
652
+ raise _refuse(
653
+ "PACKED_POSTING_LIMIT", "packed.postings.segment", "segment exceeds its byte ceiling"
654
+ )
655
+ if header.term_count > MAX_POSTING_SEGMENT_TERMS:
656
+ raise _refuse(
657
+ "PACKED_POSTING_LIMIT", "packed.postings.term_count", "segment exceeds its term ceiling"
658
+ )
659
+ expected = header.term_count * POSTING_TERM_BYTES
660
+ payload = _payload(bytes(raw), header, expected_payload=expected)
661
+ terms = tuple(
662
+ struct.unpack_from(POSTING_TERM_FORMAT, payload, position * POSTING_TERM_BYTES)
663
+ for position in range(header.term_count)
664
+ )
665
+ _validate_posting_terms(terms, dimensions=header.dimensions, entry_count=header.entry_count)
666
+ return PackedPostingSegment(
667
+ segment_index=header.subindex,
668
+ dimensions=header.dimensions,
669
+ entry_count=header.entry_count,
670
+ range_index=header.range_index,
671
+ terms=terms,
672
+ )
673
+
674
+
675
+ # ------------------------------------------------------------------------------------------
676
+ # Bound segments and the conservative bound
677
+ # ------------------------------------------------------------------------------------------
678
+
679
+
680
+ LayerBounds = tuple[tuple[int, ...], tuple[int, ...]]
681
+
682
+
683
+ @dataclass(frozen=True)
684
+ class PackedBoundSegment:
685
+ """One backend's exact per-layer, per-dimension minima and maxima for one range."""
686
+
687
+ dimensions: int
688
+ entry_count: int
689
+ range_index: int
690
+ bounds: tuple[LayerBounds, ...]
691
+
692
+
693
+ def compute_layer_bounds(
694
+ vectors_by_layer: Mapping[str, Sequence[Sequence[int]]], *, dimensions: int
695
+ ) -> tuple[LayerBounds, ...]:
696
+ """Derive the exact per-dimension minima and maxima of each layer in this range."""
697
+
698
+ _check_dimensions(dimensions, "packed.bounds.dimensions")
699
+ bounds: list[LayerBounds] = []
700
+ for layer in EMBEDDING_LAYERS:
701
+ vectors = vectors_by_layer.get(layer)
702
+ if not vectors:
703
+ raise _refuse(
704
+ "PACKED_MEMBER_LAYER",
705
+ "packed.bounds.layers",
706
+ f"bounds require exactly the four layers {list(EMBEDDING_LAYERS)}",
707
+ )
708
+ minima = [MAX_INT32] * dimensions
709
+ maxima = [MIN_INT32] * dimensions
710
+ for vector in vectors:
711
+ if len(vector) != dimensions:
712
+ raise _refuse(
713
+ "PACKED_MEMBER_DIMENSION",
714
+ "packed.bounds.vectors",
715
+ "every vector summarized has exactly the backend's dimension",
716
+ )
717
+ for dimension, value in enumerate(vector):
718
+ if value < minima[dimension]:
719
+ minima[dimension] = value
720
+ if value > maxima[dimension]:
721
+ maxima[dimension] = value
722
+ bounds.append((tuple(minima), tuple(maxima)))
723
+ return tuple(bounds)
724
+
725
+
726
+ def pack_bound_segment(
727
+ bounds: Sequence[LayerBounds],
728
+ *,
729
+ dimensions: int,
730
+ entry_count: int,
731
+ range_index: int,
732
+ binding_sha256: str,
733
+ backend_coordinate: str,
734
+ ) -> bytes:
735
+ """Serialize one backend's four-layer min/max summary as a fixed-width int32 table."""
736
+
737
+ _check_dimensions(dimensions, "packed.bounds.dimensions")
738
+ _check_entry_count(entry_count, "packed.bounds.entry_count")
739
+ if len(bounds) != BOUND_SEGMENT_LAYERS:
740
+ raise _refuse(
741
+ "PACKED_BOUND_SEGMENT",
742
+ "packed.bounds",
743
+ f"a bound segment summarizes exactly {BOUND_SEGMENT_LAYERS} layers",
744
+ )
745
+ payload = bytearray()
746
+ for layer_ordinal, layer_bounds in enumerate(bounds):
747
+ if len(layer_bounds) != 2:
748
+ raise _refuse(
749
+ "PACKED_BOUND_SEGMENT",
750
+ f"packed.bounds[{layer_ordinal}]",
751
+ "each layer summary is exactly one minima and one maxima table",
752
+ )
753
+ for values in layer_bounds:
754
+ if len(values) != dimensions:
755
+ raise _refuse(
756
+ "PACKED_BOUND_SEGMENT",
757
+ f"packed.bounds[{layer_ordinal}]",
758
+ "each summary table has exactly the backend's dimension",
759
+ )
760
+ for value in values:
761
+ if type(value) is not int or not MIN_INT32 <= value <= MAX_INT32:
762
+ raise _refuse(
763
+ "PACKED_BOUND_SEGMENT",
764
+ f"packed.bounds[{layer_ordinal}]",
765
+ "a bound summary holds exact int32 values only",
766
+ )
767
+ payload += struct.pack(f">{dimensions}i", *values)
768
+ header = pack_member_header(
769
+ kind="bound",
770
+ subindex=SUBINDEX_NONE,
771
+ dimensions=dimensions,
772
+ entry_count=entry_count,
773
+ term_count=0,
774
+ range_index=range_index,
775
+ payload_bytes=len(payload),
776
+ binding_sha256=binding_sha256,
777
+ backend_coordinate=backend_coordinate,
778
+ )
779
+ raw = header + bytes(payload)
780
+ if len(raw) != bound_segment_bytes(dimensions=dimensions):
781
+ raise _refuse( # pragma: no cover - the arithmetic above is the law
782
+ "PACKED_BOUND_SEGMENT", "packed.bounds", "bound segment violated its exact byte law"
783
+ )
784
+ return raw
785
+
786
+
787
+ def parse_bound_segment(
788
+ raw: bytes, *, descriptor: PackedMemberDescriptor | None = None
789
+ ) -> PackedBoundSegment:
790
+ """Read one bound segment exactly, refusing an inverted or mis-shaped summary."""
791
+
792
+ header = parse_member_header(raw)
793
+ _check_binding(header, descriptor, kind="bound")
794
+ if header.subindex != SUBINDEX_NONE:
795
+ raise _refuse(
796
+ "PACKED_BOUND_SEGMENT",
797
+ "packed.bounds.subindex",
798
+ "a bound segment summarizes every layer and carries no subindex",
799
+ )
800
+ expected = BOUND_SEGMENT_LAYERS * header.dimensions * 2 * 4
801
+ payload = _payload(bytes(raw), header, expected_payload=expected)
802
+ stride = header.dimensions * 4
803
+ bounds: list[LayerBounds] = []
804
+ for layer_ordinal in range(BOUND_SEGMENT_LAYERS):
805
+ base = layer_ordinal * stride * 2
806
+ minima = struct.unpack_from(f">{header.dimensions}i", payload, base)
807
+ maxima = struct.unpack_from(f">{header.dimensions}i", payload, base + stride)
808
+ for dimension, (low, high) in enumerate(zip(minima, maxima, strict=True)):
809
+ if low > high:
810
+ raise _refuse(
811
+ "PACKED_BOUND_ORDER",
812
+ f"packed.bounds[{layer_ordinal}][{dimension}]",
813
+ "a summarized minimum may never exceed its maximum",
814
+ )
815
+ bounds.append((minima, maxima))
816
+ return PackedBoundSegment(
817
+ dimensions=header.dimensions,
818
+ entry_count=header.entry_count,
819
+ range_index=header.range_index,
820
+ bounds=tuple(bounds),
821
+ )
822
+
823
+
824
+ def conservative_upper_bound(
825
+ query: Sequence[int], minima: Sequence[int], maxima: Sequence[int]
826
+ ) -> int:
827
+ """The largest dot product any vector inside this summary could have with ``query``.
828
+
829
+ Exact integers throughout: ``sum_i(q_i * max_i)`` where ``q_i >= 0`` and ``sum_i(q_i * min_i)``
830
+ otherwise. It is an upper bound on the *layer* score, never a score, and it is the only number
831
+ this module computes from a query.
832
+ """
833
+
834
+ total = 0
835
+ for value, low, high in zip(query, minima, maxima, strict=True):
836
+ total += value * (high if value >= 0 else low)
837
+ return total
838
+
839
+
840
+ def range_upper_bound(query: Sequence[int], bounds: Sequence[LayerBounds]) -> int:
841
+ """The range's bound: the maximum of its four layer bounds.
842
+
843
+ Prune only when this is *strictly* below the current kth score. Equality must exact-scan: a
844
+ range that ties on score may still win on entry id, and a pruned range cannot present an id.
845
+ """
846
+
847
+ if len(bounds) != BOUND_SEGMENT_LAYERS:
848
+ raise _refuse(
849
+ "PACKED_BOUND_SEGMENT",
850
+ "packed.bounds",
851
+ f"a range bound maximizes exactly {BOUND_SEGMENT_LAYERS} layer bounds",
852
+ )
853
+ return max(conservative_upper_bound(query, minima, maxima) for minima, maxima in bounds)
854
+
855
+
856
+ # ------------------------------------------------------------------------------------------
857
+ # Descriptors, range manifests, and the outer head
858
+ # ------------------------------------------------------------------------------------------
859
+
860
+
861
+ @dataclass(frozen=True)
862
+ class PackedBackend:
863
+ """One available backend's published coordinate, exactly as the vectors were produced under."""
864
+
865
+ coordinate: str
866
+ dimensions: int
867
+ quantization: str
868
+ query_safe: bool
869
+
870
+ def __post_init__(self) -> None:
871
+ if not isinstance(self.coordinate, str) or not 1 <= len(self.coordinate) <= 256:
872
+ raise _refuse(
873
+ "PACKED_BACKEND", "packed.backend.coordinate", "must be a bounded coordinate string"
874
+ )
875
+ _check_dimensions(self.dimensions, "packed.backend.dimensions")
876
+ if not isinstance(self.quantization, str) or not self.quantization:
877
+ raise _refuse(
878
+ "PACKED_BACKEND", "packed.backend.quantization", "must name its exact quantization"
879
+ )
880
+ if type(self.query_safe) is not bool:
881
+ raise _refuse("PACKED_BACKEND", "packed.backend.query_safe", "must be a boolean")
882
+
883
+ def to_dict(self) -> dict[str, Any]:
884
+ return {
885
+ "coordinate": self.coordinate,
886
+ "dimensions": self.dimensions,
887
+ "quantization": self.quantization,
888
+ "query_safe": self.query_safe,
889
+ }
890
+
891
+
892
+ @dataclass(frozen=True)
893
+ class PackedMemberDescriptor:
894
+ """One member's exact identity: what it is, where it belongs, and its byte-for-byte digest."""
895
+
896
+ kind: str
897
+ sha256: str
898
+ bytes: int
899
+ range_index: int
900
+ first_key: str
901
+ last_key: str
902
+ entry_count: int
903
+ facts_sha256: str
904
+ range_binding_sha256: str
905
+ backend_coordinate: str | None = None
906
+ dimensions: int | None = None
907
+ layer: str | None = None
908
+ segment_index: int | None = None
909
+ term_count: int | None = None
910
+
911
+ def __post_init__(self) -> None:
912
+ if self.kind not in MEMBER_KINDS:
913
+ raise _refuse("PACKED_MEMBER_KIND", "packed.descriptor.kind", "unknown member kind")
914
+ if not _is_digest(self.sha256) or not _is_digest(self.range_binding_sha256):
915
+ raise _refuse(
916
+ "PACKED_MEMBER_DIGEST", "packed.descriptor", "member digests are lowercase SHA-256"
917
+ )
918
+ if not _is_digest(self.facts_sha256):
919
+ raise _refuse(
920
+ "PACKED_MEMBER_DIGEST",
921
+ "packed.descriptor.facts_sha256",
922
+ "a descriptor binds its range's exact facts order digest",
923
+ )
924
+ if type(self.bytes) is not int or self.bytes < 0:
925
+ raise _refuse("PACKED_MEMBER_HEADER", "packed.descriptor.bytes", "must be a byte count")
926
+ if self.layer is not None and self.layer not in EMBEDDING_LAYERS:
927
+ raise _refuse("PACKED_MEMBER_LAYER", "packed.descriptor.layer", "unknown layer")
928
+
929
+ def to_dict(self) -> dict[str, Any]:
930
+ return {
931
+ "kind": self.kind,
932
+ "sha256": self.sha256,
933
+ "bytes": self.bytes,
934
+ "range_index": self.range_index,
935
+ "first_key": self.first_key,
936
+ "last_key": self.last_key,
937
+ "entry_count": self.entry_count,
938
+ "facts_sha256": self.facts_sha256,
939
+ "range_binding_sha256": self.range_binding_sha256,
940
+ "backend_coordinate": self.backend_coordinate,
941
+ "dimensions": self.dimensions,
942
+ "layer": self.layer,
943
+ "segment_index": self.segment_index,
944
+ "term_count": self.term_count,
945
+ }
946
+
947
+ def replace(self, **changes: Any) -> PackedMemberDescriptor:
948
+ return replace(self, **changes)
949
+
950
+
951
+ _MEMBER_DESCRIPTOR_MEMBERS = frozenset(PackedMemberDescriptor.__dataclass_fields__)
952
+
953
+
954
+ def member_descriptor_from_dict(value: Any, *, path: str) -> PackedMemberDescriptor:
955
+ if not isinstance(value, dict) or set(value) != _MEMBER_DESCRIPTOR_MEMBERS:
956
+ raise _refuse("PACKED_MEMBER_HEADER", path, "a member descriptor contract differs")
957
+ if len(canonical_json_bytes(value)) > MAX_MEMBER_DESCRIPTOR_BYTES:
958
+ raise _refuse(
959
+ "PACKED_DESCRIPTOR_LIMIT",
960
+ path,
961
+ f"a member descriptor holds at most {MAX_MEMBER_DESCRIPTOR_BYTES} bytes",
962
+ )
963
+ return PackedMemberDescriptor(**value)
964
+
965
+
966
+ @dataclass(frozen=True)
967
+ class PackedRangeDescriptor:
968
+ """One range's entry in the outer head: its key interval and its manifest's exact digest."""
969
+
970
+ range_index: int
971
+ first_key: str
972
+ last_key: str
973
+ entry_count: int
974
+ facts_sha256: str
975
+ range_binding_sha256: str
976
+ manifest_sha256: str
977
+ manifest_bytes: int
978
+ member_count: int
979
+
980
+ def to_dict(self) -> dict[str, Any]:
981
+ return {
982
+ "range_index": self.range_index,
983
+ "first_key": self.first_key,
984
+ "last_key": self.last_key,
985
+ "entry_count": self.entry_count,
986
+ "facts_sha256": self.facts_sha256,
987
+ "range_binding_sha256": self.range_binding_sha256,
988
+ "manifest_sha256": self.manifest_sha256,
989
+ "manifest_bytes": self.manifest_bytes,
990
+ "member_count": self.member_count,
991
+ }
992
+
993
+ def replace(self, **changes: Any) -> PackedRangeDescriptor:
994
+ return replace(self, **changes)
995
+
996
+
997
+ _RANGE_DESCRIPTOR_MEMBERS = frozenset(PackedRangeDescriptor.__dataclass_fields__)
998
+
999
+
1000
+ def range_descriptor_from_dict(value: Any, *, path: str) -> PackedRangeDescriptor:
1001
+ if not isinstance(value, dict) or set(value) != _RANGE_DESCRIPTOR_MEMBERS:
1002
+ raise _refuse("PACKED_HEAD_DESCRIPTORS", path, "a range descriptor contract differs")
1003
+ if len(canonical_json_bytes(value)) > MAX_RANGE_DESCRIPTOR_BYTES:
1004
+ raise _refuse(
1005
+ "PACKED_DESCRIPTOR_LIMIT",
1006
+ path,
1007
+ f"a range descriptor holds at most {MAX_RANGE_DESCRIPTOR_BYTES} bytes",
1008
+ )
1009
+ if not _is_digest(value["manifest_sha256"]) or not _is_digest(value["facts_sha256"]):
1010
+ raise _refuse("PACKED_HEAD_DESCRIPTORS", path, "a range descriptor binds exact digests")
1011
+ return PackedRangeDescriptor(**value)
1012
+
1013
+
1014
+ def range_binding_sha256(
1015
+ *,
1016
+ range_index: int,
1017
+ first_key: str,
1018
+ last_key: str,
1019
+ entry_count: int,
1020
+ facts_sha256: str,
1021
+ ) -> str:
1022
+ """The digest every member of one range carries: its interval, size, and exact facts order."""
1023
+
1024
+ return canonical_sha256(
1025
+ {
1026
+ "entry_count": entry_count,
1027
+ "facts_sha256": facts_sha256,
1028
+ "first_key": first_key,
1029
+ "last_key": last_key,
1030
+ "range_index": range_index,
1031
+ }
1032
+ )
1033
+
1034
+
1035
+ # ------------------------------------------------------------------------------------------
1036
+ # The one encoding seam a packed publisher may reach
1037
+ # ------------------------------------------------------------------------------------------
1038
+
1039
+
1040
+ def encode_layer_batch(
1041
+ backend: EmbeddingBackend,
1042
+ texts: Sequence[str],
1043
+ *,
1044
+ entry_count: int,
1045
+ dimensions: int,
1046
+ ) -> tuple[tuple[tuple[int, ...], ...], BatchEncodeResult]:
1047
+ """Encode one range/layer/backend's fresh texts through the bounded seam, and verify the result.
1048
+
1049
+ This is the only place in the packed pair that reaches an encoder, and it reaches exactly one
1050
+ entry point. The counter record comes back untouched: the publisher may sum ordered records but
1051
+ may never author one, so nothing here rewrites a field of it.
1052
+ """
1053
+
1054
+ result = backend.encode_many(tuple(texts))
1055
+ if not isinstance(result, BatchEncodeResult):
1056
+ raise _refuse(
1057
+ "PACKED_ENCODE_RESULT",
1058
+ "packed.encode.result",
1059
+ "the bounded seam must return its own counter record",
1060
+ )
1061
+ if len(result.vectors) != entry_count or result.scalar_invocations != entry_count:
1062
+ raise _refuse(
1063
+ "PACKED_ENCODE_RESULT",
1064
+ "packed.encode.result",
1065
+ "the seam returned another count than the items this range/layer asked for",
1066
+ )
1067
+ for ordinal, vector in enumerate(result.vectors):
1068
+ if len(vector) != dimensions or any(
1069
+ type(value) is not int or not MIN_INT32 <= value <= MAX_INT32 for value in vector
1070
+ ):
1071
+ raise _refuse(
1072
+ "PACKED_ENCODE_RESULT",
1073
+ f"packed.encode.vectors[{ordinal}]",
1074
+ "the seam returned a vector outside the backend's exact integer coordinate",
1075
+ )
1076
+ return result.vectors, result
1077
+
1078
+
1079
+ # ------------------------------------------------------------------------------------------
1080
+ # Building one range
1081
+ # ------------------------------------------------------------------------------------------
1082
+
1083
+
1084
+ @dataclass(frozen=True)
1085
+ class PackedRangeInput:
1086
+ """One range's ordered publishable entries, exactly as classification produced them."""
1087
+
1088
+ range_index: int
1089
+ entries: tuple[Mapping[str, Any], ...]
1090
+
1091
+ def __post_init__(self) -> None:
1092
+ if type(self.range_index) is not int or not 0 <= self.range_index < MAX_RANGE_DESCRIPTORS:
1093
+ raise _refuse(
1094
+ "PACKED_RANGE_ORDER",
1095
+ "packed.range.range_index",
1096
+ f"a generation holds at most {MAX_RANGE_DESCRIPTORS} ranges",
1097
+ )
1098
+ _check_entry_count(len(self.entries), "packed.range.entries")
1099
+ previous: bytes | None = None
1100
+ for position, entry in enumerate(self.entries):
1101
+ path = f"packed.range.entries[{position}]"
1102
+ if not isinstance(entry, Mapping) or not {
1103
+ "key",
1104
+ "entry_id",
1105
+ "classification",
1106
+ "semantic_facts_digest",
1107
+ "entry",
1108
+ "history",
1109
+ } <= set(entry):
1110
+ raise _refuse("PACKED_RANGE_ENTRY", path, "a range entry contract differs")
1111
+ if entry["classification"] not in PACKED_CLASSIFICATIONS:
1112
+ raise _refuse(
1113
+ "PACKED_RANGE_ENTRY",
1114
+ path,
1115
+ f"a published entry is one of {list(PACKED_CLASSIFICATIONS)}",
1116
+ )
1117
+ if not isinstance(entry["key"], str) or not entry["key"]:
1118
+ raise _refuse("PACKED_RANGE_ENTRY", path, "a range entry carries its exact key")
1119
+ key = entry["key"].encode("utf-8")
1120
+ if previous is not None and key <= previous:
1121
+ raise _refuse(
1122
+ "PACKED_RANGE_ORDER", path, "a range ascends by key with no repeated key"
1123
+ )
1124
+ previous = key
1125
+
1126
+ @property
1127
+ def keys(self) -> tuple[str, ...]:
1128
+ return tuple(str(entry["key"]) for entry in self.entries)
1129
+
1130
+ @property
1131
+ def first_key(self) -> str:
1132
+ return str(self.entries[0]["key"])
1133
+
1134
+ @property
1135
+ def last_key(self) -> str:
1136
+ return str(self.entries[-1]["key"])
1137
+
1138
+
1139
+ @dataclass(frozen=True)
1140
+ class BuiltPackedMember:
1141
+ """One member's exact bytes beside the descriptor that binds them."""
1142
+
1143
+ raw: bytes
1144
+ descriptor: PackedMemberDescriptor
1145
+
1146
+
1147
+ @dataclass(frozen=True)
1148
+ class BuiltPackedRange:
1149
+ """One complete range: every member's bytes, its manifest, and its head descriptor."""
1150
+
1151
+ range_index: int
1152
+ first_key: str
1153
+ last_key: str
1154
+ entry_count: int
1155
+ facts_sha256: str
1156
+ history_sha256: str
1157
+ range_binding_sha256: str
1158
+ members: tuple[BuiltPackedMember, ...]
1159
+ manifest: dict[str, Any]
1160
+ manifest_raw: bytes
1161
+ manifest_sha256: str
1162
+ descriptor: PackedRangeDescriptor
1163
+ bounds: dict[str, tuple[LayerBounds, ...]]
1164
+ posting_term_count: int
1165
+
1166
+
1167
+ def _canonical_text(payload: Any, *, code: str, path: str, maximum: int) -> tuple[str, str]:
1168
+ """Serialize one nested document to canonical text carried as one string.
1169
+
1170
+ A range holds up to a thousand v2 entries and each one is itself a deep document; nesting them
1171
+ as JSON objects would exceed the canonical member ceiling long before it exceeded the byte
1172
+ ceiling. Carrying each as its exact canonical text keeps the member's own shape flat and keeps
1173
+ the entry byte-for-byte reproducible.
1174
+ """
1175
+
1176
+ try:
1177
+ raw = canonical_json_bytes(payload)
1178
+ except CanonicalJSONError as error:
1179
+ raise _refuse(code, path, "a range document is not canonically representable") from error
1180
+ if len(raw) > maximum:
1181
+ raise _refuse(code, path, "a range document exceeds its byte bound")
1182
+ return raw.decode("utf-8"), sha256_bytes(raw)
1183
+
1184
+
1185
+ def build_packed_range(
1186
+ range_input: PackedRangeInput,
1187
+ vectors: Mapping[str, Mapping[str, Sequence[Sequence[int]]]],
1188
+ *,
1189
+ backends: Sequence[PackedBackend],
1190
+ posting_backend_coordinate: str,
1191
+ max_terms_per_segment: int | None = None,
1192
+ ) -> BuiltPackedRange:
1193
+ """Assemble every member of one range from its entries and its per-backend layer vectors."""
1194
+
1195
+ if not 1 <= len(backends) <= MAX_PACKED_BACKENDS:
1196
+ raise _refuse(
1197
+ "PACKED_BACKEND",
1198
+ "packed.range.backends",
1199
+ f"a packed generation carries 1 to {MAX_PACKED_BACKENDS} backends",
1200
+ )
1201
+ coordinates = [backend.coordinate for backend in backends]
1202
+ if len(set(coordinates)) != len(coordinates) or coordinates != sorted(coordinates):
1203
+ raise _refuse(
1204
+ "PACKED_BACKEND",
1205
+ "packed.range.backends",
1206
+ "backends are distinct and published in ascending coordinate order",
1207
+ )
1208
+ if posting_backend_coordinate not in coordinates:
1209
+ raise _refuse(
1210
+ "PACKED_POSTING_BACKEND",
1211
+ "packed.range.posting_backend_coordinate",
1212
+ "postings are derived from one of this generation's own backends",
1213
+ )
1214
+
1215
+ entry_count = len(range_input.entries)
1216
+ first_key = range_input.first_key
1217
+ last_key = range_input.last_key
1218
+
1219
+ facts_items = []
1220
+ history_items = []
1221
+ for entry in range_input.entries:
1222
+ entry_text, entry_sha256 = _canonical_text(
1223
+ entry["entry"],
1224
+ code="PACKED_FACTS_LIMIT",
1225
+ path="packed.range.entries[].entry",
1226
+ maximum=MAX_FACTS_MEMBER_BYTES,
1227
+ )
1228
+ history_text, history_sha256 = _canonical_text(
1229
+ entry["history"],
1230
+ code="PACKED_HISTORY_LIMIT",
1231
+ path="packed.range.entries[].history",
1232
+ maximum=MAX_HISTORY_MEMBER_BYTES,
1233
+ )
1234
+ facts_items.append(
1235
+ {
1236
+ "key": entry["key"],
1237
+ "entry_id": entry["entry_id"],
1238
+ "classification": entry["classification"],
1239
+ "semantic_facts_digest": entry["semantic_facts_digest"],
1240
+ "entry_sha256": entry_sha256,
1241
+ "entry_json": entry_text,
1242
+ }
1243
+ )
1244
+ history_items.append(
1245
+ {
1246
+ "key": entry["key"],
1247
+ "entry_id": entry["entry_id"],
1248
+ "history_sha256": history_sha256,
1249
+ "history_json": history_text,
1250
+ }
1251
+ )
1252
+
1253
+ facts_payload = {
1254
+ "schema_version": PACKED_FACTS_SCHEMA,
1255
+ "range_index": range_input.range_index,
1256
+ "first_key": first_key,
1257
+ "last_key": last_key,
1258
+ "entry_count": entry_count,
1259
+ "entries": facts_items,
1260
+ }
1261
+ facts_raw = _member_bytes(
1262
+ facts_payload,
1263
+ code="PACKED_FACTS_LIMIT",
1264
+ path="packed.range.facts",
1265
+ maximum=MAX_FACTS_MEMBER_BYTES,
1266
+ )
1267
+ facts_sha256 = sha256_bytes(facts_raw)
1268
+
1269
+ history_payload = {
1270
+ "schema_version": PACKED_HISTORY_SCHEMA,
1271
+ "range_index": range_input.range_index,
1272
+ "first_key": first_key,
1273
+ "last_key": last_key,
1274
+ "entry_count": entry_count,
1275
+ "entries": history_items,
1276
+ }
1277
+ history_raw = _member_bytes(
1278
+ history_payload,
1279
+ code="PACKED_HISTORY_LIMIT",
1280
+ path="packed.range.history",
1281
+ maximum=MAX_HISTORY_MEMBER_BYTES,
1282
+ )
1283
+ history_sha256 = sha256_bytes(history_raw)
1284
+
1285
+ binding = range_binding_sha256(
1286
+ range_index=range_input.range_index,
1287
+ first_key=first_key,
1288
+ last_key=last_key,
1289
+ entry_count=entry_count,
1290
+ facts_sha256=facts_sha256,
1291
+ )
1292
+
1293
+ def descriptor(**changes: Any) -> PackedMemberDescriptor:
1294
+ return PackedMemberDescriptor(
1295
+ range_index=range_input.range_index,
1296
+ first_key=first_key,
1297
+ last_key=last_key,
1298
+ entry_count=entry_count,
1299
+ facts_sha256=facts_sha256,
1300
+ range_binding_sha256=binding,
1301
+ **changes,
1302
+ )
1303
+
1304
+ members: list[BuiltPackedMember] = [
1305
+ BuiltPackedMember(
1306
+ raw=facts_raw,
1307
+ descriptor=descriptor(kind="facts", sha256=facts_sha256, bytes=len(facts_raw)),
1308
+ ),
1309
+ BuiltPackedMember(
1310
+ raw=history_raw,
1311
+ descriptor=descriptor(kind="history", sha256=history_sha256, bytes=len(history_raw)),
1312
+ ),
1313
+ ]
1314
+
1315
+ bounds_by_backend: dict[str, tuple[LayerBounds, ...]] = {}
1316
+ posting_term_count = 0
1317
+ for backend in backends:
1318
+ layers = vectors.get(backend.coordinate)
1319
+ if layers is None or set(layers) != set(EMBEDDING_LAYERS):
1320
+ raise _refuse(
1321
+ "PACKED_MEMBER_LAYER",
1322
+ "packed.range.vectors",
1323
+ f"every backend supplies exactly the four layers {list(EMBEDDING_LAYERS)}",
1324
+ )
1325
+ for layer in EMBEDDING_LAYERS:
1326
+ layer_vectors = layers[layer]
1327
+ if len(layer_vectors) != entry_count:
1328
+ raise _refuse(
1329
+ "PACKED_MEMBER_ENTRY_COUNT",
1330
+ "packed.range.vectors",
1331
+ "every layer member covers exactly this range's entries",
1332
+ )
1333
+ raw = pack_vector_member(
1334
+ layer_vectors,
1335
+ layer=layer,
1336
+ dimensions=backend.dimensions,
1337
+ range_index=range_input.range_index,
1338
+ binding_sha256=binding,
1339
+ backend_coordinate=backend.coordinate,
1340
+ )
1341
+ members.append(
1342
+ BuiltPackedMember(
1343
+ raw=raw,
1344
+ descriptor=descriptor(
1345
+ kind="vector",
1346
+ sha256=sha256_bytes(raw),
1347
+ bytes=len(raw),
1348
+ backend_coordinate=backend.coordinate,
1349
+ dimensions=backend.dimensions,
1350
+ layer=layer,
1351
+ ),
1352
+ )
1353
+ )
1354
+
1355
+ bounds = compute_layer_bounds(layers, dimensions=backend.dimensions)
1356
+ bounds_by_backend[backend.coordinate] = bounds
1357
+ raw = pack_bound_segment(
1358
+ bounds,
1359
+ dimensions=backend.dimensions,
1360
+ entry_count=entry_count,
1361
+ range_index=range_input.range_index,
1362
+ binding_sha256=binding,
1363
+ backend_coordinate=backend.coordinate,
1364
+ )
1365
+ members.append(
1366
+ BuiltPackedMember(
1367
+ raw=raw,
1368
+ descriptor=descriptor(
1369
+ kind="bound",
1370
+ sha256=sha256_bytes(raw),
1371
+ bytes=len(raw),
1372
+ backend_coordinate=backend.coordinate,
1373
+ dimensions=backend.dimensions,
1374
+ ),
1375
+ )
1376
+ )
1377
+
1378
+ if backend.coordinate == posting_backend_coordinate:
1379
+ terms = compute_layer_postings(layers, dimensions=backend.dimensions)
1380
+ posting_term_count = len(terms)
1381
+ for segment_index, segment in enumerate(
1382
+ segment_posting_terms(terms, max_terms_per_segment=max_terms_per_segment)
1383
+ ):
1384
+ raw = pack_posting_segment(
1385
+ segment,
1386
+ segment_index=segment_index,
1387
+ dimensions=backend.dimensions,
1388
+ entry_count=entry_count,
1389
+ range_index=range_input.range_index,
1390
+ binding_sha256=binding,
1391
+ backend_coordinate=backend.coordinate,
1392
+ )
1393
+ members.append(
1394
+ BuiltPackedMember(
1395
+ raw=raw,
1396
+ descriptor=descriptor(
1397
+ kind="posting",
1398
+ sha256=sha256_bytes(raw),
1399
+ bytes=len(raw),
1400
+ backend_coordinate=backend.coordinate,
1401
+ dimensions=backend.dimensions,
1402
+ segment_index=segment_index,
1403
+ term_count=len(segment),
1404
+ ),
1405
+ )
1406
+ )
1407
+
1408
+ if len(members) > MAX_RANGE_MEMBER_DESCRIPTORS:
1409
+ raise _refuse(
1410
+ "PACKED_RANGE_MEMBERS",
1411
+ "packed.range.members",
1412
+ f"a range manifest holds at most {MAX_RANGE_MEMBER_DESCRIPTORS} member descriptors",
1413
+ )
1414
+ for member in members:
1415
+ if len(canonical_json_bytes(member.descriptor.to_dict())) > MAX_MEMBER_DESCRIPTOR_BYTES:
1416
+ raise _refuse(
1417
+ "PACKED_DESCRIPTOR_LIMIT",
1418
+ "packed.range.members[]",
1419
+ f"a member descriptor holds at most {MAX_MEMBER_DESCRIPTOR_BYTES} bytes",
1420
+ )
1421
+
1422
+ body = {
1423
+ "schema_version": PACKED_RANGE_MANIFEST_SCHEMA,
1424
+ "range_index": range_input.range_index,
1425
+ "first_key": first_key,
1426
+ "last_key": last_key,
1427
+ "entry_count": entry_count,
1428
+ "facts_sha256": facts_sha256,
1429
+ "history_sha256": history_sha256,
1430
+ "range_binding_sha256": binding,
1431
+ "backends": list(coordinates),
1432
+ "posting_backend_coordinate": posting_backend_coordinate,
1433
+ "members": [member.descriptor.to_dict() for member in members],
1434
+ }
1435
+ manifest = {**body, "root_sha256": canonical_sha256(body)}
1436
+ manifest_raw = _member_bytes(
1437
+ manifest,
1438
+ code="PACKED_RANGE_MANIFEST",
1439
+ path="packed.range.manifest",
1440
+ maximum=MAX_RANGE_MANIFEST_BYTES,
1441
+ )
1442
+ return BuiltPackedRange(
1443
+ range_index=range_input.range_index,
1444
+ first_key=first_key,
1445
+ last_key=last_key,
1446
+ entry_count=entry_count,
1447
+ facts_sha256=facts_sha256,
1448
+ history_sha256=history_sha256,
1449
+ range_binding_sha256=binding,
1450
+ members=tuple(members),
1451
+ manifest=manifest,
1452
+ manifest_raw=manifest_raw,
1453
+ manifest_sha256=sha256_bytes(manifest_raw),
1454
+ descriptor=PackedRangeDescriptor(
1455
+ range_index=range_input.range_index,
1456
+ first_key=first_key,
1457
+ last_key=last_key,
1458
+ entry_count=entry_count,
1459
+ facts_sha256=facts_sha256,
1460
+ range_binding_sha256=binding,
1461
+ manifest_sha256=sha256_bytes(manifest_raw),
1462
+ manifest_bytes=len(manifest_raw),
1463
+ member_count=len(members),
1464
+ ),
1465
+ bounds=bounds_by_backend,
1466
+ posting_term_count=posting_term_count,
1467
+ )
1468
+
1469
+
1470
+ def _member_bytes(payload: Any, *, code: str, path: str, maximum: int) -> bytes:
1471
+ try:
1472
+ raw = canonical_json_bytes(payload)
1473
+ except CanonicalJSONError as error:
1474
+ raise _refuse(code, path, "a packed document is not canonically representable") from error
1475
+ if len(raw) > maximum:
1476
+ raise _refuse(
1477
+ code,
1478
+ path,
1479
+ f"the member holds at most {maximum} bytes; publication refuses rather than splitting",
1480
+ )
1481
+ return raw
1482
+
1483
+
1484
+ # ------------------------------------------------------------------------------------------
1485
+ # The outer head
1486
+ # ------------------------------------------------------------------------------------------
1487
+
1488
+
1489
+ COUNTER_MEMBERS = (
1490
+ "batch_invocations",
1491
+ "encoded_layer_items",
1492
+ "backend_scalar_invocations",
1493
+ "encoded_utf8_bytes",
1494
+ "reused_layer_items",
1495
+ )
1496
+
1497
+
1498
+ def ranges_chain_sha256(descriptors: Sequence[PackedRangeDescriptor]) -> str:
1499
+ """One chained digest over every range descriptor, in published order."""
1500
+
1501
+ chain: str | None = None
1502
+ for descriptor in descriptors:
1503
+ chain = canonical_sha256({"previous_sha256": chain, "range": descriptor.to_dict()})
1504
+ return chain or canonical_sha256({"previous_sha256": None, "range": None})
1505
+
1506
+
1507
+ def build_packed_head(
1508
+ *,
1509
+ provider_id: str,
1510
+ generation_sha256: str,
1511
+ predecessor_head_sha256: str | None,
1512
+ backends: Sequence[PackedBackend],
1513
+ posting_backend_coordinate: str,
1514
+ ranges: Sequence[PackedRangeDescriptor],
1515
+ coverage: Mapping[str, Any],
1516
+ counters: Mapping[str, int],
1517
+ invocations: Sequence[Mapping[str, Any]],
1518
+ terminal_sha256: str,
1519
+ ) -> dict[str, Any]:
1520
+ """Assemble the outer head. Every law it carries is enforced by the reader, not by this call."""
1521
+
1522
+ if len(ranges) > MAX_RANGE_DESCRIPTORS:
1523
+ raise _refuse(
1524
+ "PACKED_HEAD_DESCRIPTORS",
1525
+ "packed.head.ranges",
1526
+ f"an outer head holds at most {MAX_RANGE_DESCRIPTORS} range descriptors",
1527
+ )
1528
+ if not ranges:
1529
+ raise _refuse(
1530
+ "PACKED_HEAD_DESCRIPTORS", "packed.head.ranges", "a published generation has a range"
1531
+ )
1532
+ if set(counters) != set(COUNTER_MEMBERS):
1533
+ raise _refuse(
1534
+ "PACKED_COUNTER_CONTRACT",
1535
+ "packed.head.counters",
1536
+ f"the head carries exactly {list(COUNTER_MEMBERS)}",
1537
+ )
1538
+ for name in COUNTER_MEMBERS:
1539
+ if type(counters[name]) is not int or counters[name] < 0:
1540
+ raise _refuse(
1541
+ "PACKED_COUNTER_CONTRACT",
1542
+ f"packed.head.counters.{name}",
1543
+ "a work counter is a non-negative integer",
1544
+ )
1545
+ for descriptor in ranges:
1546
+ if len(canonical_json_bytes(descriptor.to_dict())) > MAX_RANGE_DESCRIPTOR_BYTES:
1547
+ raise _refuse(
1548
+ "PACKED_DESCRIPTOR_LIMIT",
1549
+ "packed.head.ranges[]",
1550
+ f"a range descriptor holds at most {MAX_RANGE_DESCRIPTOR_BYTES} bytes",
1551
+ )
1552
+ if not _is_digest(terminal_sha256):
1553
+ raise _refuse(
1554
+ "PACKED_HEAD_DIGEST",
1555
+ "packed.head.terminal_sha256",
1556
+ "the terminal publication record is named by one lowercase SHA-256",
1557
+ )
1558
+ if not coverage_is_valid(coverage):
1559
+ raise _refuse(
1560
+ "PACKED_HEAD_COVERAGE",
1561
+ "packed.head.coverage",
1562
+ "a published generation states what it covers and what it drew from",
1563
+ )
1564
+ body = {
1565
+ "schema_version": PACKED_HEAD_SCHEMA,
1566
+ "provider_id": provider_id,
1567
+ "generation_sha256": generation_sha256,
1568
+ "predecessor_head_sha256": predecessor_head_sha256,
1569
+ "backends": [backend.to_dict() for backend in backends],
1570
+ "posting_backend_coordinate": posting_backend_coordinate,
1571
+ "range_count": len(ranges),
1572
+ "entry_count": sum(descriptor.entry_count for descriptor in ranges),
1573
+ # What this generation covers travels with the head rather than only with the receipts
1574
+ # behind it, because the head is what a release archive is built from and what an
1575
+ # installed catalogue reads back. A consumer that holds only the published bytes can
1576
+ # still tell a bounded catalogue from an exhaustive one.
1577
+ "coverage": dict(coverage),
1578
+ "ranges": [descriptor.to_dict() for descriptor in ranges],
1579
+ "ranges_chain_sha256": ranges_chain_sha256(ranges),
1580
+ "counters": {name: counters[name] for name in COUNTER_MEMBERS},
1581
+ "invocations": [dict(record) for record in invocations],
1582
+ "terminal_sha256": terminal_sha256,
1583
+ }
1584
+ head = {**body, "root_sha256": canonical_sha256(body)}
1585
+ if len(canonical_json_bytes(head)) > MAX_PACKED_HEAD_BYTES:
1586
+ raise _refuse(
1587
+ "PACKED_HEAD_LIMIT",
1588
+ "packed.head",
1589
+ f"an outer head holds at most {MAX_PACKED_HEAD_BYTES} bytes",
1590
+ )
1591
+ return head
1592
+
1593
+
1594
+ # ------------------------------------------------------------------------------------------
1595
+ # The durable output root
1596
+ # ------------------------------------------------------------------------------------------
1597
+
1598
+
1599
+ class PackedDirectory:
1600
+ """One locally confined packed root whose members install atomically and read back exactly."""
1601
+
1602
+ def __init__(self, root: Path, *, root_descriptor: int | None = None) -> None:
1603
+ self.root = Path(root)
1604
+ self._retained_root_descriptor = root_descriptor
1605
+ self.bytes_written = 0
1606
+ self._root_fd = -1
1607
+ self._members_fd = -1
1608
+ self._owned: list[int] = []
1609
+
1610
+ @contextlib.contextmanager
1611
+ def opened(self) -> Iterator[PackedDirectory]:
1612
+ try:
1613
+ if self._retained_root_descriptor is None:
1614
+ self.root.mkdir(parents=True, exist_ok=True, mode=0o700)
1615
+ self._root_fd = self._open_directory(self.root)
1616
+ else:
1617
+ self._root_fd = os.dup(self._retained_root_descriptor)
1618
+ self._owned.append(self._root_fd)
1619
+ self._require_private_directory(self._root_fd)
1620
+ try:
1621
+ os.mkdir(PACKED_MEMBERS_DIRNAME, mode=0o700, dir_fd=self._root_fd)
1622
+ except FileExistsError:
1623
+ pass
1624
+ self._members_fd = self._open_directory_at(self._root_fd, PACKED_MEMBERS_DIRNAME)
1625
+ except OSError as error:
1626
+ self._release()
1627
+ raise _refuse("PACKED_OUTPUT_IO", "packed.output", "output root is unusable") from error
1628
+ except SourceContractError:
1629
+ self._release()
1630
+ raise
1631
+ try:
1632
+ yield self
1633
+ finally:
1634
+ self._release()
1635
+
1636
+ def _release(self) -> None:
1637
+ for descriptor in reversed(self._owned):
1638
+ os.close(descriptor)
1639
+ self._owned = []
1640
+ self._root_fd = -1
1641
+ self._members_fd = -1
1642
+
1643
+ def _open_directory(self, path: Path) -> int:
1644
+ descriptor = os.open(path, _OPEN_DIRECTORY)
1645
+ self._owned.append(descriptor)
1646
+ self._require_private_directory(descriptor)
1647
+ return descriptor
1648
+
1649
+ def _open_directory_at(self, parent_descriptor: int, name: str) -> int:
1650
+ descriptor = os.open(name, _OPEN_DIRECTORY, dir_fd=parent_descriptor)
1651
+ self._owned.append(descriptor)
1652
+ self._require_private_directory(descriptor)
1653
+ return descriptor
1654
+
1655
+ def _require_private_directory(self, descriptor: int) -> None:
1656
+ info = os.fstat(descriptor)
1657
+ if not stat.S_ISDIR(info.st_mode) or stat.S_IMODE(info.st_mode) & 0o077:
1658
+ raise _refuse(
1659
+ "PACKED_OUTPUT_PATH",
1660
+ "packed.output",
1661
+ "output components must be private directories",
1662
+ )
1663
+
1664
+ @property
1665
+ def root_descriptor(self) -> int:
1666
+ """The exact open root used by this directory context."""
1667
+
1668
+ if self._root_fd < 0:
1669
+ raise RuntimeError("packed directory is not open")
1670
+ return self._root_fd
1671
+
1672
+ def read(self, name: str, *, maximum: int) -> bytes | None:
1673
+ try:
1674
+ return _read_regular_at(self._root_fd, name, maximum=maximum)
1675
+ except FileNotFoundError:
1676
+ return None
1677
+
1678
+ def read_member(self, sha256: str, *, kind: str, maximum: int) -> bytes | None:
1679
+ try:
1680
+ return _read_regular_at(
1681
+ self._members_fd, member_filename(kind, sha256), maximum=maximum
1682
+ )
1683
+ except FileNotFoundError:
1684
+ return None
1685
+
1686
+ def has_member(self, sha256: str, *, kind: str) -> bool:
1687
+ try:
1688
+ os.stat(member_filename(kind, sha256), dir_fd=self._members_fd, follow_symlinks=False)
1689
+ except FileNotFoundError:
1690
+ return False
1691
+ except OSError as error:
1692
+ raise _refuse(
1693
+ "PACKED_OUTPUT_IO", "packed.output.member", "member lookup failed"
1694
+ ) from error
1695
+ return True
1696
+
1697
+ def write_member_bytes(self, raw: bytes, *, kind: str) -> tuple[str, int]:
1698
+ digest = sha256_bytes(raw)
1699
+ name = member_filename(kind, digest)
1700
+ try:
1701
+ existing = _read_regular_at(self._members_fd, name, maximum=len(raw))
1702
+ except FileNotFoundError:
1703
+ existing = None
1704
+ except OSError as error:
1705
+ raise _refuse(
1706
+ "PACKED_OUTPUT_IO", "packed.output.member", "member lookup failed"
1707
+ ) from error
1708
+ if existing is not None:
1709
+ if existing != raw:
1710
+ raise _refuse(
1711
+ "PACKED_OUTPUT_READBACK", "packed.output.member", "digest path bytes differ"
1712
+ )
1713
+ return digest, len(raw)
1714
+ temporary = f".publish.{os.getpid()}.{secrets.token_hex(16)}.tmp"
1715
+ try:
1716
+ _write_regular_at(self._members_fd, temporary, raw)
1717
+ with contextlib.suppress(FileExistsError):
1718
+ os.link(
1719
+ temporary,
1720
+ name,
1721
+ src_dir_fd=self._members_fd,
1722
+ dst_dir_fd=self._members_fd,
1723
+ follow_symlinks=False,
1724
+ )
1725
+ except OSError as error:
1726
+ raise _refuse(
1727
+ "PACKED_OUTPUT_IO", "packed.output.member", "member publication failed"
1728
+ ) from error
1729
+ finally:
1730
+ with contextlib.suppress(FileNotFoundError):
1731
+ os.unlink(temporary, dir_fd=self._members_fd)
1732
+ os.fsync(self._members_fd)
1733
+ if _read_regular_at(self._members_fd, name, maximum=len(raw)) != raw:
1734
+ raise _refuse(
1735
+ "PACKED_OUTPUT_READBACK", "packed.output.member", "member readback differs"
1736
+ )
1737
+ self.bytes_written += len(raw)
1738
+ return digest, len(raw)
1739
+
1740
+ def publish(self, name: str, payload: Mapping[str, Any], *, maximum: int) -> str:
1741
+ raw = canonical_json_bytes(dict(payload))
1742
+ if len(raw) > maximum:
1743
+ raise _refuse(
1744
+ "PACKED_OUTPUT_LIMIT", f"packed.output.{name}", "document exceeds its byte bound"
1745
+ )
1746
+ temporary = f".{name}.{os.getpid()}.{secrets.token_hex(8)}.tmp"
1747
+ try:
1748
+ _write_regular_at(self._root_fd, temporary, raw)
1749
+ os.replace(temporary, name, src_dir_fd=self._root_fd, dst_dir_fd=self._root_fd)
1750
+ os.fsync(self._root_fd)
1751
+ except OSError as error:
1752
+ raise _refuse(
1753
+ "PACKED_OUTPUT_IO", f"packed.output.{name}", "document publication failed"
1754
+ ) from error
1755
+ finally:
1756
+ with contextlib.suppress(FileNotFoundError):
1757
+ os.unlink(temporary, dir_fd=self._root_fd)
1758
+ if _read_regular_at(self._root_fd, name, maximum=maximum) != raw:
1759
+ raise _refuse(
1760
+ "PACKED_OUTPUT_READBACK", f"packed.output.{name}", "document readback differs"
1761
+ )
1762
+ self.bytes_written += len(raw)
1763
+ return sha256_bytes(raw)
1764
+
1765
+
1766
+ def _read_regular_at(parent_fd: int, name: str, *, maximum: int) -> bytes:
1767
+ try:
1768
+ raw = read_bounded_at(parent_fd, name, maximum=maximum)
1769
+ except BoundedReadFailure as error:
1770
+ raise _refuse(
1771
+ "PACKED_OUTPUT_PATH",
1772
+ "packed.output",
1773
+ "members must be stable bounded single-link regular files",
1774
+ ) from error
1775
+ assert raw is not None
1776
+ return raw
1777
+
1778
+
1779
+ def _write_regular_at(parent_fd: int, name: str, raw: bytes) -> None:
1780
+ flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0)
1781
+ descriptor = os.open(name, flags, 0o600, dir_fd=parent_fd)
1782
+ try:
1783
+ written = 0
1784
+ while written < len(raw):
1785
+ count = os.write(descriptor, raw[written:])
1786
+ if count < 1:
1787
+ raise _refuse("PACKED_OUTPUT_IO", "packed.output", "write made no progress")
1788
+ written += count
1789
+ os.fsync(descriptor)
1790
+ finally:
1791
+ os.close(descriptor)
1792
+
1793
+
1794
+ # ------------------------------------------------------------------------------------------
1795
+ # The independent reader
1796
+ # ------------------------------------------------------------------------------------------
1797
+
1798
+
1799
+ @dataclass(frozen=True)
1800
+ class VerifiedPackedRange:
1801
+ """What one range verified to, recomputed from its own bytes rather than from a summary."""
1802
+
1803
+ range_index: int
1804
+ first_key: str
1805
+ last_key: str
1806
+ entry_count: int
1807
+ facts_sha256: str
1808
+ history_sha256: str
1809
+ range_binding_sha256: str
1810
+ manifest_bytes: int
1811
+ member_count: int
1812
+ member_sha256s: tuple[str, ...]
1813
+ vector_payload_sha256s: tuple[str, ...]
1814
+ posting_sha256s: tuple[str, ...]
1815
+ bound_sha256s: tuple[str, ...]
1816
+ posting_term_count: int
1817
+ entry_ids: tuple[str, ...]
1818
+ classification_counts: tuple[tuple[str, int], ...]
1819
+
1820
+
1821
+ @dataclass(frozen=True)
1822
+ class VerifiedPackedGeneration:
1823
+ """One complete packed generation's verified shape."""
1824
+
1825
+ head_sha256: str
1826
+ head: dict[str, Any]
1827
+ range_count: int
1828
+ entry_count: int
1829
+ member_count: int
1830
+ posting_term_count: int
1831
+ backends: tuple[str, ...]
1832
+ classification_counts: tuple[tuple[str, int], ...]
1833
+ fresh_range_count: int
1834
+ vector_payload_sha256s: tuple[str, ...]
1835
+ posting_member_sha256s: tuple[str, ...]
1836
+ bound_member_sha256s: tuple[str, ...]
1837
+
1838
+
1839
+ def verify_packed_range(
1840
+ directory: PackedDirectory,
1841
+ manifest_sha256: str,
1842
+ *,
1843
+ backends: Sequence[PackedBackend],
1844
+ posting_backend_coordinate: str,
1845
+ ) -> VerifiedPackedRange:
1846
+ """Scan one emitted range once, recomputing every accelerator from the exact pack bytes."""
1847
+
1848
+ raw = directory.read_member(manifest_sha256, kind="range", maximum=MAX_RANGE_MANIFEST_BYTES)
1849
+ if raw is None or sha256_bytes(raw) != manifest_sha256:
1850
+ raise _refuse(
1851
+ "PACKED_RANGE_DIGEST",
1852
+ "packed.range.manifest",
1853
+ "the installed range manifest is not the manifest the head binds",
1854
+ )
1855
+ try:
1856
+ manifest = parse_canonical_json(raw)
1857
+ except CanonicalJSONError as error:
1858
+ raise _refuse(
1859
+ "PACKED_RANGE_MANIFEST", "packed.range.manifest", "manifest is not canonical"
1860
+ ) from error
1861
+ expected_members = {
1862
+ "schema_version",
1863
+ "range_index",
1864
+ "first_key",
1865
+ "last_key",
1866
+ "entry_count",
1867
+ "facts_sha256",
1868
+ "history_sha256",
1869
+ "range_binding_sha256",
1870
+ "backends",
1871
+ "posting_backend_coordinate",
1872
+ "members",
1873
+ "root_sha256",
1874
+ }
1875
+ if (
1876
+ not isinstance(manifest, dict)
1877
+ or set(manifest) != expected_members
1878
+ or manifest["schema_version"] != PACKED_RANGE_MANIFEST_SCHEMA
1879
+ or not isinstance(manifest["members"], list)
1880
+ ):
1881
+ raise _refuse(
1882
+ "PACKED_RANGE_MANIFEST", "packed.range.manifest", "range manifest contract differs"
1883
+ )
1884
+ body = {key: value for key, value in manifest.items() if key != "root_sha256"}
1885
+ if canonical_sha256(body) != manifest["root_sha256"]:
1886
+ raise _refuse(
1887
+ "PACKED_RANGE_MANIFEST", "packed.range.manifest", "manifest root digest differs"
1888
+ )
1889
+ if not 1 <= len(manifest["members"]) <= MAX_RANGE_MEMBER_DESCRIPTORS:
1890
+ raise _refuse(
1891
+ "PACKED_RANGE_MEMBERS",
1892
+ "packed.range.members",
1893
+ f"a range manifest holds 1 to {MAX_RANGE_MEMBER_DESCRIPTORS} member descriptors",
1894
+ )
1895
+ if manifest["posting_backend_coordinate"] != posting_backend_coordinate or manifest[
1896
+ "backends"
1897
+ ] != [backend.coordinate for backend in backends]:
1898
+ raise _refuse(
1899
+ "PACKED_RANGE_MANIFEST",
1900
+ "packed.range.backends",
1901
+ "a range names exactly the generation's backends and posting backend",
1902
+ )
1903
+
1904
+ descriptors = [
1905
+ member_descriptor_from_dict(value, path=f"packed.range.members[{index}]")
1906
+ for index, value in enumerate(manifest["members"])
1907
+ ]
1908
+ binding = range_binding_sha256(
1909
+ range_index=manifest["range_index"],
1910
+ first_key=manifest["first_key"],
1911
+ last_key=manifest["last_key"],
1912
+ entry_count=manifest["entry_count"],
1913
+ facts_sha256=manifest["facts_sha256"],
1914
+ )
1915
+ if binding != manifest["range_binding_sha256"]:
1916
+ raise _refuse(
1917
+ "PACKED_RANGE_MANIFEST",
1918
+ "packed.range.range_binding_sha256",
1919
+ "the range binding does not reproduce from this range's own interval and facts order",
1920
+ )
1921
+ for descriptor in descriptors:
1922
+ if (
1923
+ descriptor.range_index != manifest["range_index"]
1924
+ or descriptor.first_key != manifest["first_key"]
1925
+ or descriptor.last_key != manifest["last_key"]
1926
+ or descriptor.entry_count != manifest["entry_count"]
1927
+ or descriptor.facts_sha256 != manifest["facts_sha256"]
1928
+ or descriptor.range_binding_sha256 != binding
1929
+ ):
1930
+ raise _refuse(
1931
+ "PACKED_MEMBER_BINDING",
1932
+ "packed.range.members[]",
1933
+ "a member descriptor is bound to another range",
1934
+ )
1935
+
1936
+ facts = _verified_member(directory, _one(descriptors, "facts"), maximum=MAX_FACTS_MEMBER_BYTES)
1937
+ entry_ids, classification_counts = _verify_facts_member(facts, manifest)
1938
+ _verified_member(directory, _one(descriptors, "history"), maximum=MAX_HISTORY_MEMBER_BYTES)
1939
+
1940
+ vectors_by_backend: dict[str, dict[str, tuple[tuple[int, ...], ...]]] = {}
1941
+ vector_payload_sha256s: list[str] = []
1942
+ for backend in backends:
1943
+ layers: dict[str, tuple[tuple[int, ...], ...]] = {}
1944
+ for descriptor in descriptors:
1945
+ if descriptor.kind != "vector" or descriptor.backend_coordinate != backend.coordinate:
1946
+ continue
1947
+ raw_member = _verified_member(directory, descriptor, maximum=MAX_VECTOR_MEMBER_BYTES)
1948
+ member = parse_vector_member(raw_member, descriptor=descriptor)
1949
+ vector_payload_sha256s.append(sha256_bytes(raw_member[MEMBER_HEADER_BYTES:]))
1950
+ if descriptor.dimensions != backend.dimensions:
1951
+ raise _refuse(
1952
+ "PACKED_MEMBER_DIMENSION",
1953
+ "packed.range.members[]",
1954
+ "a vector member does not carry its backend's dimension",
1955
+ )
1956
+ if member.layer in layers:
1957
+ raise _refuse(
1958
+ "PACKED_MEMBER_LAYER",
1959
+ "packed.range.members[]",
1960
+ "a backend carries each layer exactly once",
1961
+ )
1962
+ layers[member.layer] = member.vectors
1963
+ if set(layers) != set(EMBEDDING_LAYERS):
1964
+ raise _refuse(
1965
+ "PACKED_MEMBER_LAYER",
1966
+ "packed.range.members[]",
1967
+ f"every backend closes over the four layers {list(EMBEDDING_LAYERS)}",
1968
+ )
1969
+ vectors_by_backend[backend.coordinate] = layers
1970
+
1971
+ bound_sha256s: list[str] = []
1972
+ for backend in backends:
1973
+ descriptor = _one(
1974
+ [item for item in descriptors if item.backend_coordinate == backend.coordinate], "bound"
1975
+ )
1976
+ raw_member = _verified_member(
1977
+ directory, descriptor, maximum=bound_segment_bytes(dimensions=backend.dimensions)
1978
+ )
1979
+ segment = parse_bound_segment(raw_member, descriptor=descriptor)
1980
+ exact = compute_layer_bounds(
1981
+ vectors_by_backend[backend.coordinate], dimensions=backend.dimensions
1982
+ )
1983
+ _verify_bounds(segment.bounds, exact)
1984
+ bound_sha256s.append(descriptor.sha256)
1985
+
1986
+ posting_descriptors = [item for item in descriptors if item.kind == "posting"]
1987
+ if not MIN_POSTING_SEGMENTS <= len(posting_descriptors) <= MAX_POSTING_SEGMENTS:
1988
+ raise _refuse(
1989
+ "PACKED_POSTING_SEGMENTS",
1990
+ "packed.range.members[]",
1991
+ f"a range carries {MIN_POSTING_SEGMENTS} to {MAX_POSTING_SEGMENTS} posting segments",
1992
+ )
1993
+ observed: list[PostingTerm] = []
1994
+ for expected_index, descriptor in enumerate(posting_descriptors):
1995
+ if descriptor.segment_index != expected_index:
1996
+ raise _refuse(
1997
+ "PACKED_POSTING_SEGMENTS",
1998
+ "packed.range.members[]",
1999
+ "posting segments are published in ascending segment order",
2000
+ )
2001
+ raw_member = _verified_member(directory, descriptor, maximum=MAX_POSTING_SEGMENT_BYTES)
2002
+ segment = parse_posting_segment(raw_member, descriptor=descriptor)
2003
+ if descriptor.term_count != len(segment.terms):
2004
+ raise _refuse(
2005
+ "PACKED_POSTING_TERM",
2006
+ "packed.range.members[]",
2007
+ "a posting descriptor's term count differs from its segment",
2008
+ )
2009
+ if observed and segment.terms and segment.terms[0] <= observed[-1]:
2010
+ raise _refuse(
2011
+ "PACKED_POSTING_ORDER",
2012
+ "packed.range.members[]",
2013
+ "posting terms ascend across segment boundaries too",
2014
+ )
2015
+ observed.extend(segment.terms)
2016
+ posting_dimensions = next(
2017
+ backend.dimensions
2018
+ for backend in backends
2019
+ if backend.coordinate == posting_backend_coordinate
2020
+ )
2021
+ recomputed = compute_layer_postings(
2022
+ vectors_by_backend[posting_backend_coordinate], dimensions=posting_dimensions
2023
+ )
2024
+ if tuple(observed) != recomputed:
2025
+ raise _refuse(
2026
+ "PACKED_POSTING_MISMATCH",
2027
+ "packed.range.postings",
2028
+ "the published postings are not the postings these packs contain",
2029
+ )
2030
+
2031
+ return VerifiedPackedRange(
2032
+ range_index=manifest["range_index"],
2033
+ first_key=manifest["first_key"],
2034
+ last_key=manifest["last_key"],
2035
+ entry_count=manifest["entry_count"],
2036
+ facts_sha256=manifest["facts_sha256"],
2037
+ history_sha256=manifest["history_sha256"],
2038
+ range_binding_sha256=manifest["range_binding_sha256"],
2039
+ manifest_bytes=len(raw),
2040
+ member_count=len(descriptors),
2041
+ member_sha256s=tuple(descriptor.sha256 for descriptor in descriptors),
2042
+ vector_payload_sha256s=tuple(vector_payload_sha256s),
2043
+ posting_sha256s=tuple(descriptor.sha256 for descriptor in posting_descriptors),
2044
+ bound_sha256s=tuple(bound_sha256s),
2045
+ posting_term_count=len(observed),
2046
+ entry_ids=entry_ids,
2047
+ classification_counts=classification_counts,
2048
+ )
2049
+
2050
+
2051
+ def _one(descriptors: Sequence[PackedMemberDescriptor], kind: str) -> PackedMemberDescriptor:
2052
+ found = [descriptor for descriptor in descriptors if descriptor.kind == kind]
2053
+ if len(found) != 1:
2054
+ raise _refuse(
2055
+ "PACKED_RANGE_MEMBERS",
2056
+ "packed.range.members[]",
2057
+ f"a range carries exactly one {kind} member",
2058
+ )
2059
+ return found[0]
2060
+
2061
+
2062
+ def _verified_member(
2063
+ directory: PackedDirectory, descriptor: PackedMemberDescriptor, *, maximum: int
2064
+ ) -> bytes:
2065
+ raw = directory.read_member(descriptor.sha256, kind=descriptor.kind, maximum=maximum)
2066
+ if raw is None or len(raw) != descriptor.bytes or sha256_bytes(raw) != descriptor.sha256:
2067
+ raise _refuse(
2068
+ "PACKED_MEMBER_DIGEST",
2069
+ "packed.range.members[]",
2070
+ "an installed member is not the member its descriptor binds",
2071
+ )
2072
+ return raw
2073
+
2074
+
2075
+ def _verify_facts_member(
2076
+ raw: bytes, manifest: Mapping[str, Any]
2077
+ ) -> tuple[tuple[str, ...], tuple[tuple[str, int], ...]]:
2078
+ try:
2079
+ payload = parse_canonical_json(raw)
2080
+ except CanonicalJSONError as error:
2081
+ raise _refuse(
2082
+ "PACKED_FACTS_MEMBER", "packed.range.facts", "facts member is not canonical"
2083
+ ) from error
2084
+ if (
2085
+ not isinstance(payload, dict)
2086
+ or payload.get("schema_version") != PACKED_FACTS_SCHEMA
2087
+ or payload.get("range_index") != manifest["range_index"]
2088
+ or payload.get("entry_count") != manifest["entry_count"]
2089
+ or not isinstance(payload.get("entries"), list)
2090
+ or len(payload["entries"]) != manifest["entry_count"]
2091
+ ):
2092
+ raise _refuse("PACKED_FACTS_MEMBER", "packed.range.facts", "facts member contract differs")
2093
+ entry_ids: list[str] = []
2094
+ classification_counts = dict.fromkeys(PACKED_CLASSIFICATIONS, 0)
2095
+ keys: list[str] = []
2096
+ previous: bytes | None = None
2097
+ for position, item in enumerate(payload["entries"]):
2098
+ path = f"packed.range.facts.entries[{position}]"
2099
+ if not isinstance(item, dict) or set(item) != {
2100
+ "key",
2101
+ "entry_id",
2102
+ "classification",
2103
+ "semantic_facts_digest",
2104
+ "entry_sha256",
2105
+ "entry_json",
2106
+ }:
2107
+ raise _refuse("PACKED_FACTS_MEMBER", path, "a facts entry contract differs")
2108
+ if sha256_bytes(item["entry_json"].encode("utf-8")) != item["entry_sha256"]:
2109
+ raise _refuse("PACKED_FACTS_MEMBER", path, "an entry is not the entry it digests to")
2110
+ if item["classification"] not in PACKED_CLASSIFICATIONS:
2111
+ raise _refuse(
2112
+ "PACKED_FACTS_MEMBER", path, "an entry carries a publishable classification"
2113
+ )
2114
+ key = item["key"].encode("utf-8")
2115
+ if previous is not None and key <= previous:
2116
+ raise _refuse("PACKED_RANGE_ORDER", path, "a facts member ascends by key")
2117
+ previous = key
2118
+ entry_ids.append(item["entry_id"])
2119
+ classification_counts[item["classification"]] += 1
2120
+ keys.append(item["key"])
2121
+ if keys[0] != manifest["first_key"] or keys[-1] != manifest["last_key"]:
2122
+ raise _refuse(
2123
+ "PACKED_RANGE_ORDER",
2124
+ "packed.range.facts",
2125
+ "the facts member does not span the interval its range claims",
2126
+ )
2127
+ return tuple(entry_ids), tuple(classification_counts.items())
2128
+
2129
+
2130
+ def _verify_bounds(published: Sequence[LayerBounds], exact: Sequence[LayerBounds]) -> None:
2131
+ for layer_ordinal, (summary, truth) in enumerate(zip(published, exact, strict=True)):
2132
+ published_minima, published_maxima = summary
2133
+ exact_minima, exact_maxima = truth
2134
+ for dimension in range(len(exact_minima)):
2135
+ if (
2136
+ published_minima[dimension] > exact_minima[dimension]
2137
+ or published_maxima[dimension] < exact_maxima[dimension]
2138
+ ):
2139
+ raise _refuse(
2140
+ "PACKED_BOUND_NOT_CONSERVATIVE",
2141
+ f"packed.range.bounds[{layer_ordinal}][{dimension}]",
2142
+ "a published bound excludes a vector this range actually contains",
2143
+ )
2144
+ if published_minima != exact_minima or published_maxima != exact_maxima:
2145
+ raise _refuse(
2146
+ "PACKED_BOUND_MISMATCH",
2147
+ f"packed.range.bounds[{layer_ordinal}]",
2148
+ "a published bound is not the exact publisher-derived summary of these packs",
2149
+ )
2150
+
2151
+
2152
+ def verify_packed_generation(
2153
+ root: Path, *, expected_head_sha256: str, root_descriptor: int | None = None
2154
+ ) -> VerifiedPackedGeneration:
2155
+ """Read one published generation end to end and refuse anything that is not exactly it."""
2156
+
2157
+ directory = PackedDirectory(Path(root), root_descriptor=root_descriptor)
2158
+ with directory.opened():
2159
+ raw = directory.read(PACKED_HEAD_FILENAME, maximum=MAX_PACKED_HEAD_BYTES)
2160
+ if raw is None:
2161
+ raise _refuse(
2162
+ "PACKED_HEAD_MISSING", "packed.head", "no readable packed head at this root"
2163
+ )
2164
+ digest = sha256_bytes(raw)
2165
+ if digest != expected_head_sha256:
2166
+ raise _refuse(
2167
+ "PACKED_HEAD_MISMATCH",
2168
+ "packed.head",
2169
+ "the installed head is not the caller's exact generation",
2170
+ )
2171
+ head = _parse_head(raw)
2172
+ backends = tuple(
2173
+ PackedBackend(
2174
+ coordinate=item["coordinate"],
2175
+ dimensions=item["dimensions"],
2176
+ quantization=item["quantization"],
2177
+ query_safe=item["query_safe"],
2178
+ )
2179
+ for item in head["backends"]
2180
+ )
2181
+ posting_backend = head["posting_backend_coordinate"]
2182
+ descriptors = [
2183
+ range_descriptor_from_dict(value, path=f"packed.head.ranges[{index}]")
2184
+ for index, value in enumerate(head["ranges"])
2185
+ ]
2186
+ if ranges_chain_sha256(descriptors) != head["ranges_chain_sha256"]:
2187
+ raise _refuse(
2188
+ "PACKED_HEAD_EXPECTED",
2189
+ "packed.head.ranges_chain_sha256",
2190
+ "the head's expected range chain is not the chain of its own descriptors",
2191
+ )
2192
+ member_count = 0
2193
+ posting_term_count = 0
2194
+ vector_payload_sha256s: list[str] = []
2195
+ posting_member_sha256s: list[str] = []
2196
+ bound_member_sha256s: list[str] = []
2197
+ classification_counts = dict.fromkeys(PACKED_CLASSIFICATIONS, 0)
2198
+ fresh_range_count = 0
2199
+ previous: VerifiedPackedRange | None = None
2200
+ for position, descriptor in enumerate(descriptors):
2201
+ if descriptor.range_index != position:
2202
+ raise _refuse(
2203
+ "PACKED_RANGE_ORDER",
2204
+ f"packed.head.ranges[{position}]",
2205
+ "ranges are consecutive from zero",
2206
+ )
2207
+ verified = verify_packed_range(
2208
+ directory,
2209
+ descriptor.manifest_sha256,
2210
+ backends=backends,
2211
+ posting_backend_coordinate=posting_backend,
2212
+ )
2213
+ if (
2214
+ verified.range_index != descriptor.range_index
2215
+ or verified.first_key != descriptor.first_key
2216
+ or verified.last_key != descriptor.last_key
2217
+ or verified.entry_count != descriptor.entry_count
2218
+ or verified.facts_sha256 != descriptor.facts_sha256
2219
+ or verified.range_binding_sha256 != descriptor.range_binding_sha256
2220
+ or verified.manifest_bytes != descriptor.manifest_bytes
2221
+ or verified.member_count != descriptor.member_count
2222
+ ):
2223
+ raise _refuse(
2224
+ "PACKED_RANGE_DIGEST",
2225
+ f"packed.head.ranges[{position}]",
2226
+ "a range descriptor does not describe the range it names",
2227
+ )
2228
+ is_final = position == len(descriptors) - 1
2229
+ if not is_final and verified.entry_count != RANGE_ENTRIES:
2230
+ raise _refuse(
2231
+ "PACKED_RANGE_SIZE",
2232
+ f"packed.head.ranges[{position}]",
2233
+ f"every range but the final one holds exactly {RANGE_ENTRIES} entries",
2234
+ )
2235
+ if previous is not None and not (
2236
+ previous.last_key.encode("utf-8") < verified.first_key.encode("utf-8")
2237
+ ):
2238
+ raise _refuse(
2239
+ "PACKED_RANGE_ORDER",
2240
+ f"packed.head.ranges[{position}]",
2241
+ "ranges are consecutive intervals that ascend by key",
2242
+ )
2243
+ previous = verified
2244
+ member_count += verified.member_count
2245
+ posting_term_count += verified.posting_term_count
2246
+ vector_payload_sha256s.extend(verified.vector_payload_sha256s)
2247
+ posting_member_sha256s.extend(verified.posting_sha256s)
2248
+ bound_member_sha256s.extend(verified.bound_sha256s)
2249
+ range_counts = dict(verified.classification_counts)
2250
+ for classification in PACKED_CLASSIFICATIONS:
2251
+ classification_counts[classification] += range_counts[classification]
2252
+ if any(range_counts[name] for name in ("changed", "new")):
2253
+ fresh_range_count += 1
2254
+ entry_count = sum(descriptor.entry_count for descriptor in descriptors)
2255
+ if entry_count != head["entry_count"] or len(descriptors) != head["range_count"]:
2256
+ raise _refuse(
2257
+ "PACKED_HEAD_EXPECTED",
2258
+ "packed.head.counts",
2259
+ "the head's own counts disagree with its range descriptors",
2260
+ )
2261
+ return VerifiedPackedGeneration(
2262
+ head_sha256=digest,
2263
+ head=head,
2264
+ range_count=len(descriptors),
2265
+ entry_count=entry_count,
2266
+ member_count=member_count,
2267
+ posting_term_count=posting_term_count,
2268
+ backends=tuple(backend.coordinate for backend in backends),
2269
+ classification_counts=tuple(classification_counts.items()),
2270
+ fresh_range_count=fresh_range_count,
2271
+ vector_payload_sha256s=tuple(vector_payload_sha256s),
2272
+ posting_member_sha256s=tuple(posting_member_sha256s),
2273
+ bound_member_sha256s=tuple(bound_member_sha256s),
2274
+ )
2275
+
2276
+
2277
+ _HEAD_MEMBERS = frozenset(
2278
+ {
2279
+ "schema_version",
2280
+ "provider_id",
2281
+ "generation_sha256",
2282
+ "predecessor_head_sha256",
2283
+ "backends",
2284
+ "posting_backend_coordinate",
2285
+ "range_count",
2286
+ "entry_count",
2287
+ "coverage",
2288
+ "ranges",
2289
+ "ranges_chain_sha256",
2290
+ "counters",
2291
+ "invocations",
2292
+ "terminal_sha256",
2293
+ "root_sha256",
2294
+ }
2295
+ )
2296
+
2297
+
2298
+ def _parse_head(raw: bytes) -> dict[str, Any]:
2299
+ if len(raw) > MAX_PACKED_HEAD_BYTES:
2300
+ raise _refuse("PACKED_HEAD_LIMIT", "packed.head", "head exceeds its byte bound")
2301
+ try:
2302
+ head = parse_canonical_json(raw)
2303
+ except CanonicalJSONError as error:
2304
+ raise _refuse("PACKED_HEAD_SCHEMA", "packed.head", "head is not canonical") from error
2305
+ if (
2306
+ not isinstance(head, dict)
2307
+ or set(head) != _HEAD_MEMBERS
2308
+ or head["schema_version"] != PACKED_HEAD_SCHEMA
2309
+ or not isinstance(head["ranges"], list)
2310
+ or not isinstance(head["backends"], list)
2311
+ or not isinstance(head["counters"], dict)
2312
+ ):
2313
+ raise _refuse("PACKED_HEAD_SCHEMA", "packed.head", "head contract differs")
2314
+ body = {key: value for key, value in head.items() if key != "root_sha256"}
2315
+ if canonical_sha256(body) != head["root_sha256"]:
2316
+ raise _refuse(
2317
+ "PACKED_HEAD_DIGEST",
2318
+ "packed.head.root_sha256",
2319
+ "the head's own digest does not reproduce from the head it signs",
2320
+ )
2321
+ if not 1 <= len(head["ranges"]) <= MAX_RANGE_DESCRIPTORS:
2322
+ raise _refuse(
2323
+ "PACKED_HEAD_DESCRIPTORS",
2324
+ "packed.head.ranges",
2325
+ f"an outer head holds 1 to {MAX_RANGE_DESCRIPTORS} range descriptors",
2326
+ )
2327
+ if not 1 <= len(head["backends"]) <= MAX_PACKED_BACKENDS:
2328
+ raise _refuse(
2329
+ "PACKED_BACKEND",
2330
+ "packed.head.backends",
2331
+ f"a packed generation carries 1 to {MAX_PACKED_BACKENDS} backends",
2332
+ )
2333
+ if set(head["counters"]) != set(COUNTER_MEMBERS):
2334
+ raise _refuse(
2335
+ "PACKED_COUNTER_CONTRACT",
2336
+ "packed.head.counters",
2337
+ f"the head carries exactly {list(COUNTER_MEMBERS)}",
2338
+ )
2339
+ if not coverage_is_valid(head["coverage"]):
2340
+ raise _refuse(
2341
+ "PACKED_HEAD_COVERAGE",
2342
+ "packed.head.coverage",
2343
+ "a published generation states what it covers and what it drew from",
2344
+ )
2345
+ return head