mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,1517 @@
1
+ """Bounded exact retrieval over a range-local packed generation.
2
+
3
+ :mod:`packed_catalog` defines what a packed generation *is* and :mod:`packed_writer` publishes one.
4
+ This module is the only thing that reads one to answer a question, and it is deliberately the
5
+ smallest surface that can do so: identities out, numbers never.
6
+
7
+ Four properties are load-bearing.
8
+
9
+ *Nothing is searched that the caller did not name.* A search takes the caller's exact outer head
10
+ digest **and** the publication receipt whose coordinate re-derives from its own body, and it refuses
11
+ unless that receipt describes the head installed at this root -- the same range descriptors, the
12
+ same range chain, the same counters, the same backends. Every range manifest, bound segment,
13
+ posting segment, vector pack and facts member is then read back and compared against the digest the
14
+ authenticated chain binds it to. An ambient head, an unbound summary, a substituted pack and a
15
+ re-minted receipt that quietly drops an accelerator out of its own binding all refuse rather than
16
+ answer.
17
+
18
+ *The answer is exact whenever it is complete.* An entry's score is the maximum
19
+ over the four layers of the exact integer dot product of the query with that layer's vector. When
20
+ the selected backend is the generation's posting backend the four layer dot products are accumulated
21
+ from the exact posting segments; otherwise the four layer packs are opened and scored directly. The
22
+ two agree term for term, because the postings *are* the nonzero terms of those packs.
23
+
24
+ *Pruning is strict, and the bound is the publisher's.* A range is skipped only when the sign-aware
25
+ conservative bound of :func:`packed_catalog.range_upper_bound`, computed against the independently
26
+ readback-validated bound segment published with the generation, is **strictly** below the current
27
+ kth score. An
28
+ equal bound still exact-scans: a tie is broken by entry id ascending, and a pruned range cannot
29
+ present an id. Nothing recomputes a bound from the packs at query time; that is the publisher's job
30
+ and it was already done by an independent reader before the head was installed.
31
+
32
+ *Work is bounded before it is spent.* :class:`CatalogSearchWorkLimits` is closed and has no
33
+ defaults, so no caller can accidentally ask for unbounded work. Ranges, packs, member bytes,
34
+ posting terms, candidates, open descriptors, query encodes and elapsed time are each checked before
35
+ the work that would consume them. A cap that closes before every unpruned range has closed returns
36
+ a typed *incomplete* carrying the work actuals and **no candidate at all** -- a partial ranking that
37
+ reads like a complete one is the one answer this module may never give.
38
+
39
+ There is no threshold, no approximate neighbour search, no persisted score and no query cache. A
40
+ threshold would be an admission decision taken by similarity; the other three would each make the
41
+ answer a function of something other than the exact bytes the caller authenticated.
42
+ """
43
+
44
+ from __future__ import annotations
45
+
46
+ import contextlib
47
+ import os
48
+ import time
49
+ from collections.abc import Iterator, Mapping, Sequence
50
+ from dataclasses import dataclass
51
+ from pathlib import Path
52
+ from typing import Any
53
+
54
+ from mostlyright.data_harness.canonical import (
55
+ CanonicalJSONError,
56
+ canonical_sha256,
57
+ parse_canonical_json,
58
+ sha256_bytes,
59
+ )
60
+ from mostlyright.data_harness.sources.catalog.bounded_io import (
61
+ BoundedReadFailure,
62
+ read_bounded_descriptor,
63
+ )
64
+ from mostlyright.data_harness.sources.catalog.contracts import EMBEDDING_LAYERS
65
+ from mostlyright.data_harness.sources.catalog.coverage import coverage_is_valid
66
+ from mostlyright.data_harness.sources.catalog.embedding import (
67
+ BOUNDED_SCALAR_ADAPTER,
68
+ EmbeddingBackend,
69
+ )
70
+ from mostlyright.data_harness.sources.catalog.entry_v2 import (
71
+ CatalogEntryV2,
72
+ catalog_entry_v2_from_dict,
73
+ )
74
+ from mostlyright.data_harness.sources.catalog.packed_catalog import (
75
+ MAX_FACTS_MEMBER_BYTES,
76
+ MAX_PACKED_HEAD_BYTES,
77
+ MAX_POSTING_SEGMENT_BYTES,
78
+ MAX_POSTING_SEGMENTS,
79
+ MAX_RANGE_DESCRIPTORS,
80
+ MAX_RANGE_MANIFEST_BYTES,
81
+ MAX_RANGE_MEMBER_DESCRIPTORS,
82
+ MAX_VECTOR_MEMBER_BYTES,
83
+ MEMBER_HEADER_BYTES,
84
+ MIN_POSTING_SEGMENTS,
85
+ PACKED_FACTS_SCHEMA,
86
+ PACKED_HEAD_FILENAME,
87
+ PACKED_HEAD_SCHEMA,
88
+ PACKED_MEMBERS_DIRNAME,
89
+ PACKED_RANGE_MANIFEST_SCHEMA,
90
+ RANGE_ENTRIES,
91
+ CatalogPackedRefused,
92
+ PackedMemberDescriptor,
93
+ PackedRangeDescriptor,
94
+ bound_segment_bytes,
95
+ member_descriptor_from_dict,
96
+ member_filename,
97
+ parse_bound_segment,
98
+ parse_posting_segment,
99
+ parse_vector_member,
100
+ range_binding_sha256,
101
+ range_descriptor_from_dict,
102
+ range_upper_bound,
103
+ ranges_chain_sha256,
104
+ )
105
+ from mostlyright.data_harness.sources.catalog.retrieval import (
106
+ MAX_QUERY_CHARS,
107
+ MAX_RETRIEVAL_LIMIT,
108
+ )
109
+
110
+ #: The publication receipt schema this reader admits. It is stated here rather than imported from
111
+ #: :mod:`packed_writer`, because a read path that imports the publisher would put a writer on the
112
+ #: query call graph; ``test_catalog_packed_retrieval`` asserts the two constants are one string.
113
+ PACKED_RECEIPT_SCHEMA = "mr-data-catalog-packed.v1"
114
+ PACKED_SEARCH_RESULT_SCHEMA = "harness-catalog-packed-search.v1"
115
+ PACKED_WORK_LIMITS_SCHEMA = "harness-catalog-packed-work-limits.v1"
116
+
117
+ RETRIEVAL_V2_FILENAME = "retrieval.json"
118
+ SEALED_V1_FILENAME = "catalog.json"
119
+
120
+ #: The three public-source stores this repository can answer from, newest first. Dispatch is by
121
+ #: exact filename, and the packed head is additionally admitted only under its exact schema.
122
+ PUBLIC_STORE_KINDS = ("packed", "retrieval_v2", "sealed_v1")
123
+
124
+ MAX_WORK_RANGES = MAX_RANGE_DESCRIPTORS
125
+ MAX_WORK_PACKS = MAX_RANGE_DESCRIPTORS * MAX_RANGE_MEMBER_DESCRIPTORS
126
+ MAX_WORK_MEMBER_BYTES = 64 * 1024 * 1024 * 1024
127
+ MAX_WORK_POSTING_TERMS = 1 << 33
128
+ MAX_WORK_CANDIDATES = MAX_RANGE_DESCRIPTORS * RANGE_ENTRIES
129
+
130
+ #: Three descriptors is the whole peak: the root, its members directory, and one member.
131
+ MIN_WORK_OPEN_FILES = 3
132
+ MAX_WORK_OPEN_FILES = 16
133
+ MAX_WORK_QUERY_ENCODES = 8
134
+ MAX_WORK_ELAPSED_SECONDS = 24 * 60 * 60
135
+
136
+ _NANOSECONDS = 1_000_000_000
137
+
138
+ PACKED_INCOMPLETE_CODES = (
139
+ "PACKED_SEARCH_RANGE_BUDGET",
140
+ "PACKED_SEARCH_PACK_BUDGET",
141
+ "PACKED_SEARCH_MEMBER_BYTE_BUDGET",
142
+ "PACKED_SEARCH_POSTING_TERM_BUDGET",
143
+ "PACKED_SEARCH_CANDIDATE_BUDGET",
144
+ "PACKED_SEARCH_OPEN_FILE_BUDGET",
145
+ "PACKED_SEARCH_QUERY_ENCODE_BUDGET",
146
+ "PACKED_SEARCH_TIME_BUDGET",
147
+ )
148
+
149
+ _OPEN_DIRECTORY = (
150
+ os.O_RDONLY
151
+ | getattr(os, "O_DIRECTORY", 0)
152
+ | getattr(os, "O_NOFOLLOW", 0)
153
+ | getattr(os, "O_CLOEXEC", 0)
154
+ )
155
+ _OPEN_MEMBER = (
156
+ os.O_RDONLY
157
+ | getattr(os, "O_NOFOLLOW", 0)
158
+ | getattr(os, "O_CLOEXEC", 0)
159
+ | getattr(os, "O_NONBLOCK", 0)
160
+ )
161
+
162
+ _READ_CHUNK = 1 << 20
163
+
164
+ _RECEIPT_OUTPUT_MEMBERS = frozenset(
165
+ {
166
+ "head_sha256",
167
+ "range_count",
168
+ "entry_count",
169
+ "ranges",
170
+ "ranges_chain_sha256",
171
+ "vector_member_sha256s",
172
+ "posting_member_sha256s",
173
+ "bound_member_sha256s",
174
+ }
175
+ )
176
+
177
+ _FACTS_ENTRY_MEMBERS = frozenset(
178
+ {
179
+ "key",
180
+ "entry_id",
181
+ "classification",
182
+ "semantic_facts_digest",
183
+ "entry_sha256",
184
+ "entry_json",
185
+ }
186
+ )
187
+
188
+ _RANGE_MANIFEST_MEMBERS = frozenset(
189
+ {
190
+ "schema_version",
191
+ "range_index",
192
+ "first_key",
193
+ "last_key",
194
+ "entry_count",
195
+ "facts_sha256",
196
+ "history_sha256",
197
+ "range_binding_sha256",
198
+ "backends",
199
+ "posting_backend_coordinate",
200
+ "members",
201
+ "root_sha256",
202
+ }
203
+ )
204
+
205
+
206
+ def _refuse(code: str, path: str, detail: str) -> CatalogPackedRefused:
207
+ return CatalogPackedRefused(code, path, detail)
208
+
209
+
210
+ def _is_digest(value: Any) -> bool:
211
+ return (
212
+ isinstance(value, str)
213
+ and len(value) == 64
214
+ and all(character in "0123456789abcdef" for character in value)
215
+ )
216
+
217
+
218
+ def _now_ns() -> int:
219
+ """One monotonic clock reading. Named so a test can hold it still."""
220
+
221
+ return time.monotonic_ns()
222
+
223
+
224
+ # ------------------------------------------------------------------------------------------
225
+ # The closed work contract
226
+ # ------------------------------------------------------------------------------------------
227
+
228
+
229
+ _WORK_LIMIT_CEILINGS = {
230
+ "max_ranges": MAX_WORK_RANGES,
231
+ "max_packs": MAX_WORK_PACKS,
232
+ "max_member_bytes": MAX_WORK_MEMBER_BYTES,
233
+ "max_posting_terms": MAX_WORK_POSTING_TERMS,
234
+ "max_candidates": MAX_WORK_CANDIDATES,
235
+ "max_open_files": MAX_WORK_OPEN_FILES,
236
+ "max_query_encodes": MAX_WORK_QUERY_ENCODES,
237
+ "max_elapsed_seconds": MAX_WORK_ELAPSED_SECONDS,
238
+ }
239
+
240
+
241
+ @dataclass(frozen=True)
242
+ class CatalogSearchWorkLimits:
243
+ """Every bound one packed search may consume, stated before it consumes any of them.
244
+
245
+ There is no default anywhere on this contract. A default would be a hidden answer to "how much
246
+ work may this question cost", supplied by whoever forgot to ask rather than by the caller who
247
+ has to live with it.
248
+ """
249
+
250
+ max_ranges: int
251
+ max_packs: int
252
+ max_member_bytes: int
253
+ max_posting_terms: int
254
+ max_candidates: int
255
+ max_open_files: int
256
+ max_query_encodes: int
257
+ max_elapsed_seconds: int
258
+
259
+ def __post_init__(self) -> None:
260
+ for name, ceiling in _WORK_LIMIT_CEILINGS.items():
261
+ value = getattr(self, name)
262
+ if type(value) is not int or not 1 <= value <= ceiling:
263
+ raise _refuse(
264
+ "PACKED_SEARCH_LIMIT",
265
+ f"work_limits.{name}",
266
+ f"must be an integer in [1, {ceiling}]",
267
+ )
268
+
269
+ def to_dict(self) -> dict[str, Any]:
270
+ return {
271
+ "schema_version": PACKED_WORK_LIMITS_SCHEMA,
272
+ **{name: getattr(self, name) for name in _WORK_LIMIT_CEILINGS},
273
+ }
274
+
275
+ @property
276
+ def digest(self) -> str:
277
+ return canonical_sha256(self.to_dict())
278
+
279
+
280
+ @dataclass(frozen=True)
281
+ class PackedSearchWork:
282
+ """What one search actually did. Every number here is deterministic; elapsed time is not one."""
283
+
284
+ ranges_bounded: int
285
+ ranges_scanned: int
286
+ ranges_pruned: int
287
+ packs_opened: int
288
+ member_bytes_read: int
289
+ posting_terms_read: int
290
+ candidates_examined: int
291
+ open_files_peak: int
292
+ query_encodes: int
293
+
294
+ def to_dict(self) -> dict[str, int]:
295
+ return {name: getattr(self, name) for name in self.__dataclass_fields__}
296
+
297
+
298
+ class _Exhausted(Exception):
299
+ """One work cap closed.
300
+
301
+ It is private control flow, not a typed refusal: it never leaves this module as an exception,
302
+ and the budget it names becomes ``PackedRetrievalResult.incomplete_code`` instead. So it
303
+ deliberately carries a ``budget`` rather than a ``code`` -- a workbench reader never sees this
304
+ class, and a class that looks like a typed error while never reaching a person is exactly the
305
+ thing the plain-words gates exist to notice.
306
+ """
307
+
308
+ def __init__(self, budget: str) -> None:
309
+ super().__init__(budget)
310
+ self.budget = budget
311
+
312
+
313
+ class _Budget:
314
+ """Every cap, spent before the work it pays for rather than measured after it."""
315
+
316
+ def __init__(self, limits: CatalogSearchWorkLimits) -> None:
317
+ self._limits = limits
318
+ self._deadline_ns = _now_ns() + limits.max_elapsed_seconds * _NANOSECONDS
319
+ self._open_files = 0
320
+ self.ranges_bounded = 0
321
+ self.ranges_scanned = 0
322
+ self.ranges_pruned = 0
323
+ self.packs_opened = 0
324
+ self.member_bytes_read = 0
325
+ self.posting_terms_read = 0
326
+ self.candidates_examined = 0
327
+ self.open_files_peak = 0
328
+ self.query_encodes = 0
329
+
330
+ def work(self) -> PackedSearchWork:
331
+ return PackedSearchWork(
332
+ ranges_bounded=self.ranges_bounded,
333
+ ranges_scanned=self.ranges_scanned,
334
+ ranges_pruned=self.ranges_pruned,
335
+ packs_opened=self.packs_opened,
336
+ member_bytes_read=self.member_bytes_read,
337
+ posting_terms_read=self.posting_terms_read,
338
+ candidates_examined=self.candidates_examined,
339
+ open_files_peak=self.open_files_peak,
340
+ query_encodes=self.query_encodes,
341
+ )
342
+
343
+ def check_time(self) -> None:
344
+ if _now_ns() > self._deadline_ns:
345
+ raise _Exhausted("PACKED_SEARCH_TIME_BUDGET")
346
+
347
+ def spend_range_bounded(self) -> None:
348
+ if self.ranges_bounded + 1 > self._limits.max_ranges:
349
+ raise _Exhausted("PACKED_SEARCH_RANGE_BUDGET")
350
+ self.ranges_bounded += 1
351
+
352
+ def spend_range_scanned(self) -> None:
353
+ if self.ranges_scanned + 1 > self._limits.max_ranges:
354
+ raise _Exhausted("PACKED_SEARCH_RANGE_BUDGET")
355
+ self.ranges_scanned += 1
356
+
357
+ def note_pruned(self) -> None:
358
+ self.ranges_pruned += 1
359
+
360
+ def spend_member_bytes(self, count: int) -> None:
361
+ if self.member_bytes_read + count > self._limits.max_member_bytes:
362
+ raise _Exhausted("PACKED_SEARCH_MEMBER_BYTE_BUDGET")
363
+ self.member_bytes_read += count
364
+
365
+ def spend_pack(self) -> None:
366
+ if self.packs_opened + 1 > self._limits.max_packs:
367
+ raise _Exhausted("PACKED_SEARCH_PACK_BUDGET")
368
+ self.packs_opened += 1
369
+
370
+ def spend_posting_terms(self, count: int) -> None:
371
+ if self.posting_terms_read + count > self._limits.max_posting_terms:
372
+ raise _Exhausted("PACKED_SEARCH_POSTING_TERM_BUDGET")
373
+ self.posting_terms_read += count
374
+
375
+ def spend_candidates(self, count: int) -> None:
376
+ if self.candidates_examined + count > self._limits.max_candidates:
377
+ raise _Exhausted("PACKED_SEARCH_CANDIDATE_BUDGET")
378
+ self.candidates_examined += count
379
+
380
+ def spend_query_encode(self) -> None:
381
+ if self.query_encodes + 1 > self._limits.max_query_encodes:
382
+ raise _Exhausted("PACKED_SEARCH_QUERY_ENCODE_BUDGET")
383
+ self.query_encodes += 1
384
+
385
+ def open_file(self) -> None:
386
+ if self._open_files + 1 > self._limits.max_open_files:
387
+ raise _Exhausted("PACKED_SEARCH_OPEN_FILE_BUDGET")
388
+ self._open_files += 1
389
+ if self._open_files > self.open_files_peak:
390
+ self.open_files_peak = self._open_files
391
+
392
+ def close_file(self) -> None:
393
+ self._open_files -= 1
394
+
395
+
396
+ # ------------------------------------------------------------------------------------------
397
+ # The read-only packed root
398
+ # ------------------------------------------------------------------------------------------
399
+
400
+
401
+ def _read_regular(parent_fd: int, name: str, *, maximum: int, budget: _Budget | None) -> bytes:
402
+ """Read one bounded single-link regular file relative to a retained directory."""
403
+
404
+ if budget is not None:
405
+ budget.open_file()
406
+ try:
407
+ descriptor = os.open(name, _OPEN_MEMBER, dir_fd=parent_fd)
408
+ except FileNotFoundError:
409
+ if budget is not None:
410
+ budget.close_file()
411
+ raise
412
+ except OSError as error:
413
+ if budget is not None:
414
+ budget.close_file()
415
+ raise _refuse("PACKED_SEARCH_MEMBER", name, "cannot open a plain packed member") from error
416
+ try:
417
+ info = os.fstat(descriptor)
418
+ if budget is not None:
419
+ budget.spend_member_bytes(info.st_size)
420
+ try:
421
+ return read_bounded_descriptor(descriptor, maximum=maximum)
422
+ except BoundedReadFailure as error:
423
+ raise _refuse(
424
+ "PACKED_SEARCH_MEMBER",
425
+ name,
426
+ "a packed member must be stable, bounded, and single-link regular",
427
+ ) from error
428
+ finally:
429
+ os.close(descriptor)
430
+ if budget is not None:
431
+ budget.close_file()
432
+
433
+
434
+ class _PackedRoot:
435
+ """One locally confined packed root, opened read-only and never by pathname twice."""
436
+
437
+ def __init__(self, root: Path, budget: _Budget) -> None:
438
+ self.root = Path(root)
439
+ self._budget = budget
440
+ self._root_fd = -1
441
+ self._members_fd = -1
442
+ self._stack: contextlib.ExitStack | None = None
443
+
444
+ @contextlib.contextmanager
445
+ def opened(self) -> Iterator[_PackedRoot]:
446
+ """Retain the root for the whole search. The members directory opens on first use.
447
+
448
+ The head is a document of the root, so a root with no members directory still refuses on
449
+ its missing head rather than on a directory a caller never asked about.
450
+ """
451
+
452
+ with contextlib.ExitStack() as stack:
453
+ self._budget.open_file()
454
+ stack.callback(self._budget.close_file)
455
+ try:
456
+ root_fd = os.open(self.root, _OPEN_DIRECTORY)
457
+ except OSError as error:
458
+ raise _refuse(
459
+ "PACKED_SEARCH_ROOT", str(self.root), "cannot open a plain packed root"
460
+ ) from error
461
+ stack.callback(os.close, root_fd)
462
+ self._root_fd = root_fd
463
+ self._stack = stack
464
+ try:
465
+ yield self
466
+ finally:
467
+ self._root_fd = -1
468
+ self._members_fd = -1
469
+ self._stack = None
470
+
471
+ def _members(self) -> int:
472
+ if self._members_fd != -1:
473
+ return self._members_fd
474
+ stack = self._stack
475
+ if stack is None: # pragma: no cover - only reachable outside `opened`
476
+ raise _refuse("PACKED_SEARCH_ROOT", str(self.root), "the packed root is not open")
477
+ self._budget.open_file()
478
+ stack.callback(self._budget.close_file)
479
+ try:
480
+ members_fd = os.open(PACKED_MEMBERS_DIRNAME, _OPEN_DIRECTORY, dir_fd=self._root_fd)
481
+ except OSError as error:
482
+ raise _refuse(
483
+ "PACKED_SEARCH_ROOT",
484
+ PACKED_MEMBERS_DIRNAME,
485
+ "cannot open the packed members directory",
486
+ ) from error
487
+ stack.callback(os.close, members_fd)
488
+ self._members_fd = members_fd
489
+ return members_fd
490
+
491
+ def read_document(self, name: str, *, maximum: int) -> bytes | None:
492
+ try:
493
+ return _read_regular(self._root_fd, name, maximum=maximum, budget=self._budget)
494
+ except FileNotFoundError:
495
+ return None
496
+
497
+ def read_member(self, sha256: str, *, kind: str, maximum: int) -> bytes | None:
498
+ members_fd = self._members()
499
+ self._budget.spend_pack()
500
+ try:
501
+ return _read_regular(
502
+ members_fd,
503
+ member_filename(kind, sha256),
504
+ maximum=maximum,
505
+ budget=self._budget,
506
+ )
507
+ except FileNotFoundError:
508
+ return None
509
+
510
+
511
+ def detect_public_store_kind(root: Path) -> str:
512
+ """Name the public store installed at ``root`` by exact filename, never by discovery.
513
+
514
+ A packed head is admitted only under its exact schema: a file with that name that is not a
515
+ packed head refuses rather than silently falling through to the legacy v2 reader.
516
+ """
517
+
518
+ if not isinstance(root, Path):
519
+ raise _refuse("PACKED_SEARCH_STORE", "root", "must be a pathlib.Path")
520
+ try:
521
+ root_fd = os.open(root, _OPEN_DIRECTORY)
522
+ except OSError as error:
523
+ raise _refuse(
524
+ "PACKED_SEARCH_STORE", str(root), "no readable public store at this root"
525
+ ) from error
526
+ try:
527
+ try:
528
+ raw = _read_regular(
529
+ root_fd, PACKED_HEAD_FILENAME, maximum=MAX_PACKED_HEAD_BYTES, budget=None
530
+ )
531
+ except FileNotFoundError:
532
+ raw = None
533
+ if raw is not None:
534
+ try:
535
+ head = parse_canonical_json(raw)
536
+ except CanonicalJSONError as error:
537
+ raise _refuse(
538
+ "PACKED_SEARCH_STORE",
539
+ PACKED_HEAD_FILENAME,
540
+ "a packed head is canonical JSON or it is not a packed head",
541
+ ) from error
542
+ if not isinstance(head, dict) or head.get("schema_version") != PACKED_HEAD_SCHEMA:
543
+ raise _refuse(
544
+ "PACKED_SEARCH_STORE",
545
+ PACKED_HEAD_FILENAME,
546
+ "this root installs a head that is not a packed head",
547
+ )
548
+ return "packed"
549
+ for name, kind in (
550
+ (RETRIEVAL_V2_FILENAME, "retrieval_v2"),
551
+ (SEALED_V1_FILENAME, "sealed_v1"),
552
+ ):
553
+ try:
554
+ os.stat(name, dir_fd=root_fd, follow_symlinks=False)
555
+ except FileNotFoundError:
556
+ continue
557
+ except OSError as error:
558
+ raise _refuse(
559
+ "PACKED_SEARCH_STORE", name, "cannot inspect a public store member"
560
+ ) from error
561
+ return kind
562
+ raise _refuse(
563
+ "PACKED_SEARCH_STORE",
564
+ str(root),
565
+ "this root holds no packed head, v2 retrieval manifest or sealed catalog",
566
+ )
567
+ finally:
568
+ os.close(root_fd)
569
+
570
+
571
+ # ------------------------------------------------------------------------------------------
572
+ # What one search returns
573
+ # ------------------------------------------------------------------------------------------
574
+
575
+
576
+ @dataclass(frozen=True)
577
+ class PackedRankedEntry:
578
+ """One ordered identity and the exact facts it was carried by. No ranking number, ever."""
579
+
580
+ rank: int
581
+ entry_id: str
582
+ entry_coordinate: str
583
+ entry_digest: str
584
+ semantic_facts_digest: str
585
+ range_index: int
586
+ entry: CatalogEntryV2
587
+
588
+ def to_dict(self) -> dict[str, Any]:
589
+ return {
590
+ "rank": self.rank,
591
+ "entry_id": self.entry_id,
592
+ "entry_coordinate": self.entry_coordinate,
593
+ "entry_digest": self.entry_digest,
594
+ "semantic_facts_digest": self.semantic_facts_digest,
595
+ }
596
+
597
+
598
+ @dataclass(frozen=True)
599
+ class PackedRetrievalResult:
600
+ """One search's complete outcome, or a typed incomplete carrying only what it did.
601
+
602
+ An incomplete result carries no entry. That is the whole point of the type: a caller that
603
+ cannot tell a budget-truncated ranking from a finished one would present the first as the
604
+ second, and the difference is exactly the entries the search never looked at.
605
+ """
606
+
607
+ status: str
608
+ incomplete_code: str | None
609
+ head_sha256: str
610
+ receipt_coordinate_sha256: str
611
+ backend_coordinate: str
612
+ question_digest: str
613
+ limit: int
614
+ range_manifest_sha256s: tuple[str, ...]
615
+ vector_member_sha256s: tuple[str, ...]
616
+ posting_member_sha256s: tuple[str, ...]
617
+ bound_member_sha256s: tuple[str, ...]
618
+ entries: tuple[PackedRankedEntry, ...]
619
+ limits: CatalogSearchWorkLimits
620
+ work: PackedSearchWork
621
+
622
+ def __post_init__(self) -> None:
623
+ if self.status not in {"complete", "incomplete"}:
624
+ raise _refuse(
625
+ "PACKED_SEARCH_STATUS", "result.status", "a search is complete or incomplete"
626
+ )
627
+ if self.status == "complete":
628
+ if self.incomplete_code is not None:
629
+ raise _refuse(
630
+ "PACKED_SEARCH_STATUS",
631
+ "result.incomplete_code",
632
+ "a complete search names no exhausted budget",
633
+ )
634
+ elif self.incomplete_code not in PACKED_INCOMPLETE_CODES or self.entries:
635
+ raise _refuse(
636
+ "PACKED_SEARCH_STATUS",
637
+ "result.entries",
638
+ "an incomplete search names the budget that closed and offers no ranked identity",
639
+ )
640
+ if tuple(item.rank for item in self.entries) != tuple(range(1, len(self.entries) + 1)):
641
+ raise _refuse(
642
+ "PACKED_SEARCH_ORDER",
643
+ "result.entries",
644
+ "ranks must be contiguous and ordered from one",
645
+ )
646
+
647
+ def to_dict(self) -> dict[str, Any]:
648
+ return {
649
+ "schema_version": PACKED_SEARCH_RESULT_SCHEMA,
650
+ "status": self.status,
651
+ "incomplete_code": self.incomplete_code,
652
+ "head_sha256": self.head_sha256,
653
+ "receipt_coordinate_sha256": self.receipt_coordinate_sha256,
654
+ "backend_coordinate": self.backend_coordinate,
655
+ "layers": list(EMBEDDING_LAYERS),
656
+ "question_digest": self.question_digest,
657
+ "limit": self.limit,
658
+ "range_manifest_sha256s": list(self.range_manifest_sha256s),
659
+ "vector_member_sha256s": list(self.vector_member_sha256s),
660
+ "posting_member_sha256s": list(self.posting_member_sha256s),
661
+ "bound_member_sha256s": list(self.bound_member_sha256s),
662
+ "entries": [item.to_dict() for item in self.entries],
663
+ "work_limits": self.limits.to_dict(),
664
+ "work_actuals": self.work.to_dict(),
665
+ }
666
+
667
+ @property
668
+ def identity_sha256(self) -> str:
669
+ return canonical_sha256(self.to_dict())
670
+
671
+
672
+ # ------------------------------------------------------------------------------------------
673
+ # Authentication
674
+ # ------------------------------------------------------------------------------------------
675
+
676
+
677
+ @dataclass(frozen=True)
678
+ class _Authenticated:
679
+ """The head and the receipt, agreed with each other before a single range is opened."""
680
+
681
+ head: dict[str, Any]
682
+ head_sha256: str
683
+ receipt_coordinate_sha256: str
684
+ descriptors: tuple[PackedRangeDescriptor, ...]
685
+ backends: tuple[Mapping[str, Any], ...]
686
+ posting_backend_coordinate: str
687
+ vector_payload_sha256s: frozenset[str]
688
+ posting_member_sha256s: frozenset[str]
689
+ bound_member_sha256s: frozenset[str]
690
+
691
+
692
+ def _receipt_coordinate(receipt: Any) -> str:
693
+ """Check the receipt against itself, before a byte of the generation has been read.
694
+
695
+ This half needs no head: a receipt that is not a complete bounded-scalar publication, or that
696
+ does not reproduce its own coordinate, is refused whatever is installed at the root.
697
+ """
698
+
699
+ if not isinstance(receipt, Mapping) or receipt.get("schema_version") != PACKED_RECEIPT_SCHEMA:
700
+ raise _refuse("PACKED_RECEIPT_SCHEMA", "packed.receipt", "receipt contract differs")
701
+ if receipt.get("adapter_mode") != BOUNDED_SCALAR_ADAPTER:
702
+ raise _refuse(
703
+ "PACKED_RECEIPT_ADAPTER",
704
+ "packed.receipt.adapter_mode",
705
+ f"a packed generation may claim only the {BOUNDED_SCALAR_ADAPTER!r} adapter",
706
+ )
707
+ if receipt.get("status") != "complete":
708
+ raise _refuse(
709
+ "PACKED_RECEIPT_STATUS",
710
+ "packed.receipt.status",
711
+ "only a complete publication may be searched",
712
+ )
713
+ body = {key: value for key, value in receipt.items() if key != "coordinate_sha256"}
714
+ try:
715
+ coordinate = canonical_sha256(body)
716
+ except CanonicalJSONError as error:
717
+ raise _refuse(
718
+ "PACKED_RECEIPT_COORDINATE", "packed.receipt", "receipt is not canonical"
719
+ ) from error
720
+ if coordinate != receipt.get("coordinate_sha256"):
721
+ raise _refuse(
722
+ "PACKED_RECEIPT_COORDINATE",
723
+ "packed.receipt.coordinate_sha256",
724
+ "the receipt does not reproduce its own coordinate",
725
+ )
726
+ return coordinate
727
+
728
+
729
+ def _authenticate(
730
+ root: _PackedRoot, *, expected_head_sha256: str, receipt: Mapping[str, Any], coordinate: str
731
+ ) -> _Authenticated:
732
+ raw = root.read_document(PACKED_HEAD_FILENAME, maximum=MAX_PACKED_HEAD_BYTES)
733
+ if raw is None:
734
+ raise _refuse("PACKED_HEAD_MISSING", "packed.head", "no readable packed head at this root")
735
+ observed = sha256_bytes(raw)
736
+ if observed != expected_head_sha256:
737
+ raise _refuse(
738
+ "PACKED_HEAD_MISMATCH",
739
+ "packed.head",
740
+ "the installed head is not the caller's exact generation",
741
+ )
742
+ head = _parse_head(raw)
743
+ descriptors = tuple(
744
+ range_descriptor_from_dict(value, path=f"packed.head.ranges[{index}]")
745
+ for index, value in enumerate(head["ranges"])
746
+ )
747
+ if ranges_chain_sha256(descriptors) != head["ranges_chain_sha256"]:
748
+ raise _refuse(
749
+ "PACKED_HEAD_EXPECTED",
750
+ "packed.head.ranges_chain_sha256",
751
+ "the head's stated range chain is not the chain of its own descriptors",
752
+ )
753
+ for position, descriptor in enumerate(descriptors):
754
+ if descriptor.range_index != position:
755
+ raise _refuse(
756
+ "PACKED_RANGE_ORDER",
757
+ f"packed.head.ranges[{position}]",
758
+ "ranges are consecutive from zero",
759
+ )
760
+ return _bound_receipt(
761
+ receipt,
762
+ head=head,
763
+ head_sha256=observed,
764
+ descriptors=descriptors,
765
+ coordinate=coordinate,
766
+ )
767
+
768
+
769
+ def _parse_head(raw: bytes) -> dict[str, Any]:
770
+ """Read the outer head exactly. The publisher's own laws, enforced by this reader."""
771
+
772
+ if len(raw) > MAX_PACKED_HEAD_BYTES:
773
+ raise _refuse("PACKED_HEAD_LIMIT", "packed.head", "head exceeds its byte bound")
774
+ try:
775
+ head = parse_canonical_json(raw)
776
+ except CanonicalJSONError as error:
777
+ raise _refuse("PACKED_HEAD_SCHEMA", "packed.head", "head is not canonical") from error
778
+ if (
779
+ not isinstance(head, dict)
780
+ or head.get("schema_version") != PACKED_HEAD_SCHEMA
781
+ or not isinstance(head.get("ranges"), list)
782
+ or not isinstance(head.get("backends"), list)
783
+ or not isinstance(head.get("counters"), dict)
784
+ or not isinstance(head.get("posting_backend_coordinate"), str)
785
+ ):
786
+ raise _refuse("PACKED_HEAD_SCHEMA", "packed.head", "head contract differs")
787
+ body = {key: value for key, value in head.items() if key != "root_sha256"}
788
+ if canonical_sha256(body) != head.get("root_sha256"):
789
+ raise _refuse(
790
+ "PACKED_HEAD_DIGEST",
791
+ "packed.head.root_sha256",
792
+ "the head's own digest does not reproduce from the head it signs",
793
+ )
794
+ if not 1 <= len(head["ranges"]) <= MAX_RANGE_DESCRIPTORS:
795
+ raise _refuse(
796
+ "PACKED_HEAD_DESCRIPTORS",
797
+ "packed.head.ranges",
798
+ f"an outer head holds 1 to {MAX_RANGE_DESCRIPTORS} range descriptors",
799
+ )
800
+ # This is the reader an *installed* catalogue is read back through, which makes it the one
801
+ # place the coverage statement has to be required rather than merely carried. A head that
802
+ # omitted it, or stated one that is not a coverage, would otherwise be readable here while
803
+ # the publisher's own reader refused it -- and a catalogue that cannot say whether it holds
804
+ # the whole sweep is exactly the thing the member exists to prevent.
805
+ if not coverage_is_valid(head.get("coverage")):
806
+ raise _refuse(
807
+ "PACKED_HEAD_COVERAGE",
808
+ "packed.head.coverage",
809
+ "an installed generation states what it covers and what it drew from",
810
+ )
811
+ return head
812
+
813
+
814
+ def _bound_receipt(
815
+ receipt: Mapping[str, Any],
816
+ *,
817
+ head: dict[str, Any],
818
+ head_sha256: str,
819
+ descriptors: tuple[PackedRangeDescriptor, ...],
820
+ coordinate: str,
821
+ ) -> _Authenticated:
822
+ outputs = receipt.get("outputs")
823
+ if not isinstance(outputs, Mapping) or not _RECEIPT_OUTPUT_MEMBERS <= set(outputs):
824
+ raise _refuse("PACKED_RECEIPT_SCHEMA", "packed.receipt.outputs", "outputs contract differs")
825
+ if (
826
+ outputs["head_sha256"] != head_sha256
827
+ or outputs["range_count"] != head["range_count"]
828
+ or outputs["entry_count"] != head["entry_count"]
829
+ or outputs["ranges"] != head["ranges"]
830
+ or outputs["ranges_chain_sha256"] != head["ranges_chain_sha256"]
831
+ or receipt.get("counters") != head["counters"]
832
+ or receipt.get("backends") != head["backends"]
833
+ or receipt.get("posting_backend_coordinate") != head["posting_backend_coordinate"]
834
+ or receipt.get("coverage") != head["coverage"]
835
+ ):
836
+ raise _refuse(
837
+ "PACKED_RECEIPT_MISMATCH",
838
+ "packed.receipt.outputs",
839
+ "the receipt does not describe the generation installed at this root",
840
+ )
841
+ summaries: dict[str, frozenset[str]] = {}
842
+ backend_count = len(head["backends"])
843
+ range_count = len(descriptors)
844
+ expected = {
845
+ "vector_member_sha256s": (
846
+ range_count * len(EMBEDDING_LAYERS) * backend_count,
847
+ range_count * len(EMBEDDING_LAYERS) * backend_count,
848
+ ),
849
+ "bound_member_sha256s": (range_count * backend_count, range_count * backend_count),
850
+ "posting_member_sha256s": (
851
+ range_count * MIN_POSTING_SEGMENTS,
852
+ range_count * MAX_POSTING_SEGMENTS,
853
+ ),
854
+ }
855
+ for name, (low, high) in expected.items():
856
+ values = outputs[name]
857
+ if not isinstance(values, list) or any(not _is_digest(value) for value in values):
858
+ raise _refuse(
859
+ "PACKED_RECEIPT_SCHEMA",
860
+ f"packed.receipt.outputs.{name}",
861
+ "a receipt binds each summary by a lowercase SHA-256",
862
+ )
863
+ if not low <= len(values) <= high:
864
+ raise _refuse(
865
+ "PACKED_SEARCH_UNBOUND_MEMBER",
866
+ f"packed.receipt.outputs.{name}",
867
+ "the receipt does not bind one summary per published range, backend and layer",
868
+ )
869
+ summaries[name] = frozenset(values)
870
+ return _Authenticated(
871
+ head=head,
872
+ head_sha256=head_sha256,
873
+ receipt_coordinate_sha256=coordinate,
874
+ descriptors=descriptors,
875
+ backends=tuple(head["backends"]),
876
+ posting_backend_coordinate=str(head["posting_backend_coordinate"]),
877
+ vector_payload_sha256s=summaries["vector_member_sha256s"],
878
+ posting_member_sha256s=summaries["posting_member_sha256s"],
879
+ bound_member_sha256s=summaries["bound_member_sha256s"],
880
+ )
881
+
882
+
883
+ # ------------------------------------------------------------------------------------------
884
+ # One range, read back against the chain that binds it
885
+ # ------------------------------------------------------------------------------------------
886
+
887
+
888
+ @dataclass(frozen=True)
889
+ class _VerifiedRange:
890
+ descriptor: PackedRangeDescriptor
891
+ members: tuple[PackedMemberDescriptor, ...]
892
+
893
+
894
+ def _verified_range(
895
+ root: _PackedRoot, authenticated: _Authenticated, descriptor: PackedRangeDescriptor
896
+ ) -> _VerifiedRange:
897
+ raw = root.read_member(
898
+ descriptor.manifest_sha256, kind="range", maximum=MAX_RANGE_MANIFEST_BYTES
899
+ )
900
+ if (
901
+ raw is None
902
+ or len(raw) != descriptor.manifest_bytes
903
+ or sha256_bytes(raw) != descriptor.manifest_sha256
904
+ ):
905
+ raise _refuse(
906
+ "PACKED_RANGE_DIGEST",
907
+ "packed.range.manifest",
908
+ "the installed range manifest is not the manifest the head binds",
909
+ )
910
+ try:
911
+ manifest = parse_canonical_json(raw)
912
+ except CanonicalJSONError as error:
913
+ raise _refuse(
914
+ "PACKED_RANGE_MANIFEST", "packed.range.manifest", "manifest is not canonical"
915
+ ) from error
916
+ if (
917
+ not isinstance(manifest, dict)
918
+ or set(manifest) != _RANGE_MANIFEST_MEMBERS
919
+ or manifest["schema_version"] != PACKED_RANGE_MANIFEST_SCHEMA
920
+ or not isinstance(manifest["members"], list)
921
+ ):
922
+ raise _refuse(
923
+ "PACKED_RANGE_MANIFEST", "packed.range.manifest", "range manifest contract differs"
924
+ )
925
+ body = {key: value for key, value in manifest.items() if key != "root_sha256"}
926
+ if canonical_sha256(body) != manifest["root_sha256"]:
927
+ raise _refuse(
928
+ "PACKED_RANGE_MANIFEST", "packed.range.manifest", "manifest root digest differs"
929
+ )
930
+ if not 1 <= len(manifest["members"]) <= MAX_RANGE_MEMBER_DESCRIPTORS:
931
+ raise _refuse(
932
+ "PACKED_RANGE_MEMBERS",
933
+ "packed.range.members",
934
+ f"a range manifest holds 1 to {MAX_RANGE_MEMBER_DESCRIPTORS} member descriptors",
935
+ )
936
+ if (
937
+ manifest["backends"] != [backend["coordinate"] for backend in authenticated.backends]
938
+ or manifest["posting_backend_coordinate"] != authenticated.posting_backend_coordinate
939
+ ):
940
+ raise _refuse(
941
+ "PACKED_RANGE_MANIFEST",
942
+ "packed.range.backends",
943
+ "a range names exactly the generation's backends and posting backend",
944
+ )
945
+ binding = range_binding_sha256(
946
+ range_index=manifest["range_index"],
947
+ first_key=manifest["first_key"],
948
+ last_key=manifest["last_key"],
949
+ entry_count=manifest["entry_count"],
950
+ facts_sha256=manifest["facts_sha256"],
951
+ )
952
+ if (
953
+ binding != manifest["range_binding_sha256"]
954
+ or manifest["range_index"] != descriptor.range_index
955
+ or manifest["first_key"] != descriptor.first_key
956
+ or manifest["last_key"] != descriptor.last_key
957
+ or manifest["entry_count"] != descriptor.entry_count
958
+ or manifest["facts_sha256"] != descriptor.facts_sha256
959
+ or binding != descriptor.range_binding_sha256
960
+ or len(manifest["members"]) != descriptor.member_count
961
+ ):
962
+ raise _refuse(
963
+ "PACKED_RANGE_MANIFEST",
964
+ "packed.range.range_binding_sha256",
965
+ "the range manifest does not describe the range its head descriptor names",
966
+ )
967
+ members = tuple(
968
+ member_descriptor_from_dict(value, path=f"packed.range.members[{index}]")
969
+ for index, value in enumerate(manifest["members"])
970
+ )
971
+ for member in members:
972
+ if (
973
+ member.range_index != descriptor.range_index
974
+ or member.first_key != descriptor.first_key
975
+ or member.last_key != descriptor.last_key
976
+ or member.entry_count != descriptor.entry_count
977
+ or member.facts_sha256 != descriptor.facts_sha256
978
+ or member.range_binding_sha256 != binding
979
+ ):
980
+ raise _refuse(
981
+ "PACKED_MEMBER_BINDING",
982
+ "packed.range.members[]",
983
+ "a member descriptor is bound to another range",
984
+ )
985
+ return _VerifiedRange(descriptor=descriptor, members=members)
986
+
987
+
988
+ def _one_member(members: Sequence[PackedMemberDescriptor], kind: str) -> PackedMemberDescriptor:
989
+ found = [member for member in members if member.kind == kind]
990
+ if len(found) != 1:
991
+ raise _refuse(
992
+ "PACKED_RANGE_MEMBERS",
993
+ "packed.range.members[]",
994
+ f"a range carries exactly one {kind} member",
995
+ )
996
+ return found[0]
997
+
998
+
999
+ def _member_bytes(root: _PackedRoot, member: PackedMemberDescriptor, *, maximum: int) -> bytes:
1000
+ raw = root.read_member(member.sha256, kind=member.kind, maximum=maximum)
1001
+ if raw is None or len(raw) != member.bytes or sha256_bytes(raw) != member.sha256:
1002
+ raise _refuse(
1003
+ "PACKED_MEMBER_DIGEST",
1004
+ "packed.range.members[]",
1005
+ "an installed member is not the member its descriptor binds",
1006
+ )
1007
+ return raw
1008
+
1009
+
1010
+ def _require_bound_summary(digest: str, bound: frozenset[str], *, path: str) -> None:
1011
+ if digest not in bound:
1012
+ raise _refuse(
1013
+ "PACKED_SEARCH_UNBOUND_MEMBER",
1014
+ path,
1015
+ "this summary is not one the authenticated receipt binds",
1016
+ )
1017
+
1018
+
1019
+ # ------------------------------------------------------------------------------------------
1020
+ # The facts member
1021
+ # ------------------------------------------------------------------------------------------
1022
+
1023
+
1024
+ @dataclass(frozen=True)
1025
+ class _FactsItem:
1026
+ entry_id: str
1027
+ semantic_facts_digest: str
1028
+ entry_sha256: str
1029
+ entry_json: str
1030
+
1031
+
1032
+ def _verified_facts(raw: bytes, descriptor: PackedRangeDescriptor) -> tuple[_FactsItem, ...]:
1033
+ """Read one facts member under exactly the law ``packed_catalog`` publishes it by.
1034
+
1035
+ ``packed_catalog._verify_facts_member`` enforces the same law and returns only the entry ids;
1036
+ this reader needs each entry's exact canonical text as well, and parsing an eight-mebibyte
1037
+ member twice per scanned range is a cost a provider-scale search cannot pay. The two are
1038
+ proved to agree, and to refuse the same corruptions, in ``test_catalog_packed_retrieval``.
1039
+ """
1040
+
1041
+ try:
1042
+ payload = parse_canonical_json(raw)
1043
+ except CanonicalJSONError as error:
1044
+ raise _refuse(
1045
+ "PACKED_FACTS_MEMBER", "packed.range.facts", "facts member is not canonical"
1046
+ ) from error
1047
+ if (
1048
+ not isinstance(payload, dict)
1049
+ or payload.get("schema_version") != PACKED_FACTS_SCHEMA
1050
+ or payload.get("range_index") != descriptor.range_index
1051
+ or payload.get("entry_count") != descriptor.entry_count
1052
+ or not isinstance(payload.get("entries"), list)
1053
+ or len(payload["entries"]) != descriptor.entry_count
1054
+ ):
1055
+ raise _refuse("PACKED_FACTS_MEMBER", "packed.range.facts", "facts member contract differs")
1056
+ items: list[_FactsItem] = []
1057
+ keys: list[str] = []
1058
+ previous: bytes | None = None
1059
+ for position, item in enumerate(payload["entries"]):
1060
+ path = f"packed.range.facts.entries[{position}]"
1061
+ if not isinstance(item, dict) or set(item) != _FACTS_ENTRY_MEMBERS:
1062
+ raise _refuse("PACKED_FACTS_MEMBER", path, "a facts entry contract differs")
1063
+ if sha256_bytes(item["entry_json"].encode("utf-8")) != item["entry_sha256"]:
1064
+ raise _refuse("PACKED_FACTS_MEMBER", path, "an entry is not the entry it digests to")
1065
+ key = item["key"].encode("utf-8")
1066
+ if previous is not None and key <= previous:
1067
+ raise _refuse("PACKED_RANGE_ORDER", path, "a facts member ascends by key")
1068
+ previous = key
1069
+ keys.append(item["key"])
1070
+ items.append(
1071
+ _FactsItem(
1072
+ entry_id=item["entry_id"],
1073
+ semantic_facts_digest=item["semantic_facts_digest"],
1074
+ entry_sha256=item["entry_sha256"],
1075
+ entry_json=item["entry_json"],
1076
+ )
1077
+ )
1078
+ if keys[0] != descriptor.first_key or keys[-1] != descriptor.last_key:
1079
+ raise _refuse(
1080
+ "PACKED_RANGE_ORDER",
1081
+ "packed.range.facts",
1082
+ "the facts member does not span the interval its range claims",
1083
+ )
1084
+ return tuple(items)
1085
+
1086
+
1087
+ def _rebuilt_entry(item: _FactsItem) -> CatalogEntryV2:
1088
+ """Rebuild one carried entry under the exact v2 contract, and check it is the one named.
1089
+
1090
+ The digest beside it in the facts member is the delta's *classification* digest -- the one
1091
+ ``streaming_delta`` computes over identity, facts and provenance while deliberately excluding
1092
+ the version link, the observations and the vector instances. It is therefore not
1093
+ ``CatalogEntryV2.semantic_facts_digest`` and is carried onward as the publisher stated it,
1094
+ rather than recomputed here under a definition this module does not own.
1095
+ """
1096
+
1097
+ try:
1098
+ document = parse_canonical_json(item.entry_json.encode("utf-8"))
1099
+ except CanonicalJSONError as error:
1100
+ raise _refuse(
1101
+ "PACKED_SEARCH_FACTS_ENTRY",
1102
+ "packed.range.facts.entries[].entry_json",
1103
+ "a carried entry is not canonical",
1104
+ ) from error
1105
+ if not isinstance(document, dict):
1106
+ raise _refuse(
1107
+ "PACKED_SEARCH_FACTS_ENTRY",
1108
+ "packed.range.facts.entries[].entry_json",
1109
+ "a carried entry is an object",
1110
+ )
1111
+ entry = catalog_entry_v2_from_dict(document)
1112
+ if entry.entry_id != item.entry_id:
1113
+ raise _refuse(
1114
+ "PACKED_SEARCH_FACTS_ENTRY",
1115
+ "packed.range.facts.entries[]",
1116
+ "a carried entry is not the entry its facts member names",
1117
+ )
1118
+ return entry
1119
+
1120
+
1121
+ # ------------------------------------------------------------------------------------------
1122
+ # Exact scoring: postings when they are this backend's, packs otherwise
1123
+ # ------------------------------------------------------------------------------------------
1124
+
1125
+
1126
+ def _scores_from_postings(
1127
+ root: _PackedRoot,
1128
+ verified: _VerifiedRange,
1129
+ authenticated: _Authenticated,
1130
+ *,
1131
+ query: Sequence[int],
1132
+ budget: _Budget,
1133
+ opened: list[str],
1134
+ ) -> list[int]:
1135
+ postings = [member for member in verified.members if member.kind == "posting"]
1136
+ if not MIN_POSTING_SEGMENTS <= len(postings) <= MAX_POSTING_SEGMENTS:
1137
+ raise _refuse(
1138
+ "PACKED_POSTING_SEGMENTS",
1139
+ "packed.range.members[]",
1140
+ f"a range carries {MIN_POSTING_SEGMENTS} to {MAX_POSTING_SEGMENTS} posting segments",
1141
+ )
1142
+ entry_count = verified.descriptor.entry_count
1143
+ layers = len(EMBEDDING_LAYERS)
1144
+ accumulated = [[0] * layers for _ in range(entry_count)]
1145
+ previous: tuple[int, int, int, int] | None = None
1146
+ for expected_index, member in enumerate(postings):
1147
+ if member.segment_index != expected_index:
1148
+ raise _refuse(
1149
+ "PACKED_POSTING_SEGMENTS",
1150
+ "packed.range.members[]",
1151
+ "posting segments are published in ascending segment order",
1152
+ )
1153
+ _require_bound_summary(
1154
+ member.sha256,
1155
+ authenticated.posting_member_sha256s,
1156
+ path="packed.range.postings",
1157
+ )
1158
+ budget.spend_posting_terms(int(member.term_count or 0))
1159
+ raw = _member_bytes(root, member, maximum=MAX_POSTING_SEGMENT_BYTES)
1160
+ segment = parse_posting_segment(raw, descriptor=member)
1161
+ if member.term_count != len(segment.terms):
1162
+ raise _refuse(
1163
+ "PACKED_POSTING_TERM",
1164
+ "packed.range.members[]",
1165
+ "a posting descriptor's term count differs from its segment",
1166
+ )
1167
+ if previous is not None and segment.terms and segment.terms[0] <= previous:
1168
+ raise _refuse(
1169
+ "PACKED_POSTING_ORDER",
1170
+ "packed.range.members[]",
1171
+ "posting terms ascend across segment boundaries too",
1172
+ )
1173
+ for dimension, ordinal, layer_ordinal, value in segment.terms:
1174
+ weight = query[dimension]
1175
+ if weight:
1176
+ accumulated[ordinal][layer_ordinal] += weight * value
1177
+ if segment.terms:
1178
+ previous = segment.terms[-1]
1179
+ opened.append(member.sha256)
1180
+ return [max(row) for row in accumulated]
1181
+
1182
+
1183
+ def _scores_from_packs(
1184
+ root: _PackedRoot,
1185
+ verified: _VerifiedRange,
1186
+ authenticated: _Authenticated,
1187
+ *,
1188
+ query: Sequence[int],
1189
+ backend_coordinate: str,
1190
+ opened: list[str],
1191
+ ) -> list[int]:
1192
+ entry_count = verified.descriptor.entry_count
1193
+ best: list[int | None] = [None] * entry_count
1194
+ seen: set[str] = set()
1195
+ for layer in EMBEDDING_LAYERS:
1196
+ member = _one_layer_member(verified.members, layer=layer, coordinate=backend_coordinate)
1197
+ raw = _member_bytes(root, member, maximum=MAX_VECTOR_MEMBER_BYTES)
1198
+ payload_sha256 = sha256_bytes(raw[MEMBER_HEADER_BYTES:])
1199
+ _require_bound_summary(
1200
+ payload_sha256,
1201
+ authenticated.vector_payload_sha256s,
1202
+ path="packed.range.vectors",
1203
+ )
1204
+ parsed = parse_vector_member(raw, descriptor=member)
1205
+ if parsed.entry_count != entry_count or layer in seen:
1206
+ raise _refuse(
1207
+ "PACKED_MEMBER_LAYER",
1208
+ "packed.range.members[]",
1209
+ "a backend carries each layer exactly once, over this range's own entries",
1210
+ )
1211
+ seen.add(layer)
1212
+ for ordinal, vector in enumerate(parsed.vectors):
1213
+ total = 0
1214
+ for weight, value in zip(query, vector, strict=True):
1215
+ total += weight * value
1216
+ current = best[ordinal]
1217
+ if current is None or total > current:
1218
+ best[ordinal] = total
1219
+ opened.append(payload_sha256)
1220
+ return [0 if value is None else value for value in best]
1221
+
1222
+
1223
+ def _one_layer_member(
1224
+ members: Sequence[PackedMemberDescriptor], *, layer: str, coordinate: str
1225
+ ) -> PackedMemberDescriptor:
1226
+ found = [
1227
+ member
1228
+ for member in members
1229
+ if member.kind == "vector"
1230
+ and member.layer == layer
1231
+ and member.backend_coordinate == coordinate
1232
+ ]
1233
+ if len(found) != 1:
1234
+ raise _refuse(
1235
+ "PACKED_MEMBER_LAYER",
1236
+ "packed.range.members[]",
1237
+ f"this range carries no single {layer!r} pack for the selected backend",
1238
+ )
1239
+ return found[0]
1240
+
1241
+
1242
+ # ------------------------------------------------------------------------------------------
1243
+ # The search
1244
+ # ------------------------------------------------------------------------------------------
1245
+
1246
+
1247
+ @dataclass(frozen=True)
1248
+ class _Candidate:
1249
+ """One entry that scored above zero, ordered by score descending then entry id ascending."""
1250
+
1251
+ negated: int
1252
+ entry_id: str
1253
+ range_index: int
1254
+ item: _FactsItem
1255
+
1256
+
1257
+ def rank_packed_entries(
1258
+ *,
1259
+ root: Path,
1260
+ expected_head_sha256: str,
1261
+ receipt: Mapping[str, Any],
1262
+ question: str,
1263
+ backend: EmbeddingBackend,
1264
+ limit: int,
1265
+ work_limits: CatalogSearchWorkLimits,
1266
+ ) -> PackedRetrievalResult:
1267
+ """Answer one question from one authenticated packed generation, or say what it could not do."""
1268
+
1269
+ if not isinstance(root, Path):
1270
+ raise _refuse("PACKED_SEARCH_ROOT", "root", "must be a pathlib.Path")
1271
+ if not isinstance(work_limits, CatalogSearchWorkLimits):
1272
+ raise _refuse("PACKED_SEARCH_LIMIT", "work_limits", "must be a CatalogSearchWorkLimits")
1273
+ if type(limit) is not int or not 1 <= limit <= MAX_RETRIEVAL_LIMIT:
1274
+ raise _refuse(
1275
+ "RETRIEVAL_LIMIT", "limit", f"must be an integer in [1, {MAX_RETRIEVAL_LIMIT}]"
1276
+ )
1277
+ if not isinstance(question, str) or len(question) > MAX_QUERY_CHARS:
1278
+ raise _refuse(
1279
+ "RETRIEVAL_QUERY_LIMIT",
1280
+ "question",
1281
+ f"a question is a string of at most {MAX_QUERY_CHARS} characters",
1282
+ )
1283
+
1284
+ if not _is_digest(expected_head_sha256):
1285
+ raise _refuse("PACKED_HEAD_EXPECTED", "expected_head_sha256", "must be a lowercase SHA-256")
1286
+ receipt_coordinate = _receipt_coordinate(receipt)
1287
+ coordinate = backend.descriptor.coordinate
1288
+
1289
+ budget = _Budget(work_limits)
1290
+ question_digest = canonical_sha256({"question": question})
1291
+ reader = _PackedRoot(root, budget)
1292
+ manifests: list[str] = []
1293
+ vectors: list[str] = []
1294
+ postings: list[str] = []
1295
+ bounds: list[str] = []
1296
+ entries: tuple[PackedRankedEntry, ...] = ()
1297
+ incomplete: str | None = None
1298
+ try:
1299
+ with reader.opened() as opened_root:
1300
+ authenticated = _authenticate(
1301
+ opened_root,
1302
+ expected_head_sha256=expected_head_sha256,
1303
+ receipt=receipt,
1304
+ coordinate=receipt_coordinate,
1305
+ )
1306
+ coordinate = _selected_backend_coordinate(authenticated, backend)
1307
+ query = _encoded_query(backend, question, budget=budget)
1308
+ if any(query):
1309
+ entries = _search(
1310
+ opened_root,
1311
+ authenticated,
1312
+ query=query,
1313
+ coordinate=coordinate,
1314
+ limit=limit,
1315
+ budget=budget,
1316
+ manifests=manifests,
1317
+ vectors=vectors,
1318
+ postings=postings,
1319
+ bounds=bounds,
1320
+ )
1321
+ except _Exhausted as exhausted:
1322
+ incomplete = exhausted.budget
1323
+ entries = ()
1324
+ return PackedRetrievalResult(
1325
+ status="incomplete" if incomplete is not None else "complete",
1326
+ incomplete_code=incomplete,
1327
+ head_sha256=expected_head_sha256,
1328
+ receipt_coordinate_sha256=receipt_coordinate,
1329
+ backend_coordinate=coordinate,
1330
+ question_digest=question_digest,
1331
+ limit=limit,
1332
+ range_manifest_sha256s=tuple(manifests),
1333
+ vector_member_sha256s=tuple(vectors),
1334
+ posting_member_sha256s=tuple(postings),
1335
+ bound_member_sha256s=tuple(bounds),
1336
+ entries=entries,
1337
+ limits=work_limits,
1338
+ work=budget.work(),
1339
+ )
1340
+
1341
+
1342
+ def _selected_backend_coordinate(authenticated: _Authenticated, backend: EmbeddingBackend) -> str:
1343
+ descriptor = backend.descriptor
1344
+ coordinate = descriptor.coordinate
1345
+ for published in authenticated.backends:
1346
+ if published["coordinate"] != coordinate:
1347
+ continue
1348
+ if (
1349
+ published["dimensions"] != descriptor.dimensions
1350
+ or published["quantization"] != descriptor.quantization
1351
+ ):
1352
+ raise _refuse(
1353
+ "PACKED_SEARCH_BACKEND",
1354
+ "backend",
1355
+ "the selected encoder is not the published backend of that coordinate",
1356
+ )
1357
+ return coordinate
1358
+ raise _refuse(
1359
+ "PACKED_SEARCH_BACKEND",
1360
+ "backend",
1361
+ "this generation publishes no packs for the selected backend coordinate",
1362
+ )
1363
+
1364
+
1365
+ def _encoded_query(backend: EmbeddingBackend, question: str, *, budget: _Budget) -> tuple[int, ...]:
1366
+ budget.spend_query_encode()
1367
+ asked = backend.encode(question)
1368
+ dimensions = backend.descriptor.dimensions
1369
+ if (
1370
+ not isinstance(asked, tuple)
1371
+ or len(asked) != dimensions
1372
+ or any(type(value) is not int for value in asked)
1373
+ ):
1374
+ raise _refuse(
1375
+ "RETRIEVAL_QUERY_VECTOR",
1376
+ "question",
1377
+ "the selected backend returned a vector outside its exact integer coordinate",
1378
+ )
1379
+ return asked
1380
+
1381
+
1382
+ def _search(
1383
+ root: _PackedRoot,
1384
+ authenticated: _Authenticated,
1385
+ *,
1386
+ query: tuple[int, ...],
1387
+ coordinate: str,
1388
+ limit: int,
1389
+ budget: _Budget,
1390
+ manifests: list[str],
1391
+ vectors: list[str],
1392
+ postings: list[str],
1393
+ bounds: list[str],
1394
+ ) -> tuple[PackedRankedEntry, ...]:
1395
+ """One bound pass to order the work, one exact pass over whatever survives pruning."""
1396
+
1397
+ range_bounds: list[int] = []
1398
+ for descriptor in authenticated.descriptors:
1399
+ budget.check_time()
1400
+ budget.spend_range_bounded()
1401
+ verified = _verified_range(root, authenticated, descriptor)
1402
+ manifests.append(descriptor.manifest_sha256)
1403
+ member = _one_backend_member(verified.members, kind="bound", coordinate=coordinate)
1404
+ _require_bound_summary(
1405
+ member.sha256, authenticated.bound_member_sha256s, path="packed.range.bounds"
1406
+ )
1407
+ raw = _member_bytes(root, member, maximum=bound_segment_bytes(dimensions=len(query)))
1408
+ segment = parse_bound_segment(raw, descriptor=member)
1409
+ range_bounds.append(range_upper_bound(query, segment.bounds))
1410
+ bounds.append(member.sha256)
1411
+
1412
+ order = sorted(
1413
+ range(len(authenticated.descriptors)),
1414
+ key=lambda index: (-range_bounds[index], index),
1415
+ )
1416
+ selected: list[_Candidate] = []
1417
+ for index in order:
1418
+ if len(selected) == limit and range_bounds[index] < -selected[limit - 1].negated:
1419
+ budget.note_pruned()
1420
+ continue
1421
+ descriptor = authenticated.descriptors[index]
1422
+ budget.check_time()
1423
+ budget.spend_range_scanned()
1424
+ budget.spend_candidates(descriptor.entry_count)
1425
+ verified = _verified_range(root, authenticated, descriptor)
1426
+ facts = _one_member(verified.members, "facts")
1427
+ items = _verified_facts(
1428
+ _member_bytes(root, facts, maximum=MAX_FACTS_MEMBER_BYTES), descriptor
1429
+ )
1430
+ if coordinate == authenticated.posting_backend_coordinate:
1431
+ scores = _scores_from_postings(
1432
+ root, verified, authenticated, query=query, budget=budget, opened=postings
1433
+ )
1434
+ else:
1435
+ scores = _scores_from_packs(
1436
+ root,
1437
+ verified,
1438
+ authenticated,
1439
+ query=query,
1440
+ backend_coordinate=coordinate,
1441
+ opened=vectors,
1442
+ )
1443
+ if len(scores) != len(items):
1444
+ raise _refuse(
1445
+ "PACKED_SEARCH_FACTS_ENTRY",
1446
+ "packed.range.facts",
1447
+ "this range's packs and its facts member describe different entry counts",
1448
+ )
1449
+ found = [
1450
+ _Candidate(
1451
+ negated=-score,
1452
+ entry_id=item.entry_id,
1453
+ range_index=descriptor.range_index,
1454
+ item=item,
1455
+ )
1456
+ for score, item in zip(scores, items, strict=True)
1457
+ if score > 0
1458
+ ]
1459
+ found.sort(key=lambda candidate: (candidate.negated, candidate.entry_id))
1460
+ selected = sorted(
1461
+ [*selected, *found[:limit]],
1462
+ key=lambda candidate: (candidate.negated, candidate.entry_id),
1463
+ )[:limit]
1464
+ return tuple(
1465
+ PackedRankedEntry(
1466
+ rank=rank,
1467
+ entry_id=candidate.entry_id,
1468
+ entry_coordinate=entry.coordinate,
1469
+ entry_digest=entry.digest,
1470
+ semantic_facts_digest=candidate.item.semantic_facts_digest,
1471
+ range_index=candidate.range_index,
1472
+ entry=entry,
1473
+ )
1474
+ for rank, (candidate, entry) in enumerate(
1475
+ ((candidate, _rebuilt_entry(candidate.item)) for candidate in selected), start=1
1476
+ )
1477
+ )
1478
+
1479
+
1480
+ def _one_backend_member(
1481
+ members: Sequence[PackedMemberDescriptor], *, kind: str, coordinate: str
1482
+ ) -> PackedMemberDescriptor:
1483
+ found = [
1484
+ member
1485
+ for member in members
1486
+ if member.kind == kind and member.backend_coordinate == coordinate
1487
+ ]
1488
+ if len(found) != 1:
1489
+ raise _refuse(
1490
+ "PACKED_RANGE_MEMBERS",
1491
+ "packed.range.members[]",
1492
+ f"a range carries exactly one {kind} member per backend",
1493
+ )
1494
+ return found[0]
1495
+
1496
+
1497
+ __all__ = [
1498
+ "MAX_WORK_CANDIDATES",
1499
+ "MAX_WORK_ELAPSED_SECONDS",
1500
+ "MAX_WORK_MEMBER_BYTES",
1501
+ "MAX_WORK_OPEN_FILES",
1502
+ "MAX_WORK_PACKS",
1503
+ "MAX_WORK_POSTING_TERMS",
1504
+ "MAX_WORK_QUERY_ENCODES",
1505
+ "MAX_WORK_RANGES",
1506
+ "MIN_WORK_OPEN_FILES",
1507
+ "PACKED_INCOMPLETE_CODES",
1508
+ "PACKED_RECEIPT_SCHEMA",
1509
+ "PACKED_SEARCH_RESULT_SCHEMA",
1510
+ "PUBLIC_STORE_KINDS",
1511
+ "CatalogSearchWorkLimits",
1512
+ "PackedRankedEntry",
1513
+ "PackedRetrievalResult",
1514
+ "PackedSearchWork",
1515
+ "detect_public_store_kind",
1516
+ "rank_packed_entries",
1517
+ ]