mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,2345 @@
|
|
|
1
|
+
"""Range-local packed catalogs: fixed-width vector packs, exact postings, conservative bounds.
|
|
2
|
+
|
|
3
|
+
The v2 retrieval manifest in :mod:`retrieval_manifest` describes one representation as a flat member
|
|
4
|
+
list. At provider scale that shape stops working: four corpus-wide packs would each be gigabytes,
|
|
5
|
+
no reader could bound its own allocation from a descriptor, and a single mutated byte anywhere would
|
|
6
|
+
be indistinguishable from a legitimate republication. This module owns packed generations; the
|
|
7
|
+
flat v1/v2 contract and its constants remain scoped to :mod:`retrieval_manifest`.
|
|
8
|
+
|
|
9
|
+
*A range is the unit.* Sorted entries are grouped into consecutive ranges of exactly
|
|
10
|
+
``RANGE_ENTRIES`` -- only the final range is short -- and each range owns one facts member, one
|
|
11
|
+
history member, four layer-specific vector members per available backend, one to eight lexical
|
|
12
|
+
posting segments, and one bound segment per backend. There are never four corpus-wide packs.
|
|
13
|
+
|
|
14
|
+
*Every member has an exact byte law.* A vector member is exactly
|
|
15
|
+
``64 + entry_count * dimension * 4`` bytes; a bound segment is exactly
|
|
16
|
+
``64 + 4 * dimension * 2 * 4`` bytes; a posting term is exactly ten bytes. That is what lets a
|
|
17
|
+
reader refuse a member from its 64-byte header before it allocates anything the member claims to
|
|
18
|
+
contain, and it is why the ceilings in this module are laws rather than guidance: a 1000-entry
|
|
19
|
+
range at the 384-dimension ceiling is 1,536,064 bytes per layer member and cannot be anything
|
|
20
|
+
else.
|
|
21
|
+
|
|
22
|
+
*Nothing derived is trusted.* Posting segments and bound segments are accelerators, and an
|
|
23
|
+
accelerator that lies changes which entries a bounded search ever looks at. So
|
|
24
|
+
:func:`verify_packed_generation` recomputes both from the exact vector member bytes and refuses a
|
|
25
|
+
summary that is not the publisher's exact one. It refuses a *non-conservative* bound with its own
|
|
26
|
+
code, because a bound that is too narrow hides a true hit while a merely wrong one only wastes a
|
|
27
|
+
scan.
|
|
28
|
+
|
|
29
|
+
*The bound is the one number this module computes.* Exact entry scores belong to the ranker and die
|
|
30
|
+
there. What a published range can carry is the summary side of that score:
|
|
31
|
+
:func:`conservative_upper_bound` is ``sum_i(q_i * max_i)`` where ``q_i >= 0`` and
|
|
32
|
+
``sum_i(q_i * min_i)`` otherwise, and :func:`range_upper_bound` maximizes it across the four
|
|
33
|
+
layers. Pruning on it must be strict -- an equal bound still exact-scans, because a tie is broken
|
|
34
|
+
by entry id and a pruned range cannot present its ids.
|
|
35
|
+
|
|
36
|
+
Fixed-endian tables and canonical JSON only. No pickle, no memory mapping, no length a reader
|
|
37
|
+
learns after it has already allocated.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
from __future__ import annotations
|
|
41
|
+
|
|
42
|
+
import contextlib
|
|
43
|
+
import os
|
|
44
|
+
import re
|
|
45
|
+
import secrets
|
|
46
|
+
import stat
|
|
47
|
+
import struct
|
|
48
|
+
from collections.abc import Iterator, Mapping, Sequence
|
|
49
|
+
from dataclasses import dataclass, replace
|
|
50
|
+
from pathlib import Path
|
|
51
|
+
from typing import Any
|
|
52
|
+
|
|
53
|
+
from mostlyright.data_harness.canonical import (
|
|
54
|
+
CanonicalJSONError,
|
|
55
|
+
canonical_json_bytes,
|
|
56
|
+
canonical_sha256,
|
|
57
|
+
parse_canonical_json,
|
|
58
|
+
sha256_bytes,
|
|
59
|
+
)
|
|
60
|
+
from mostlyright.data_harness.sources.catalog.bounded_io import (
|
|
61
|
+
BoundedReadFailure,
|
|
62
|
+
read_bounded_at,
|
|
63
|
+
)
|
|
64
|
+
from mostlyright.data_harness.sources.catalog.contracts import EMBEDDING_LAYERS
|
|
65
|
+
from mostlyright.data_harness.sources.catalog.coverage import coverage_is_valid
|
|
66
|
+
from mostlyright.data_harness.sources.catalog.embedding import BatchEncodeResult, EmbeddingBackend
|
|
67
|
+
from mostlyright.data_harness.sources.contracts import SourceContractError
|
|
68
|
+
|
|
69
|
+
PACKED_HEAD_SCHEMA = "harness-catalog-packed-head.v1"
|
|
70
|
+
PACKED_RANGE_MANIFEST_SCHEMA = "harness-catalog-packed-range.v1"
|
|
71
|
+
PACKED_FACTS_SCHEMA = "harness-catalog-packed-facts.v1"
|
|
72
|
+
PACKED_HISTORY_SCHEMA = "harness-catalog-packed-history.v1"
|
|
73
|
+
|
|
74
|
+
PACKED_HEAD_FILENAME = "packed-head.json"
|
|
75
|
+
PACKED_MEMBERS_DIRNAME = "members"
|
|
76
|
+
|
|
77
|
+
#: One range is exactly this many entries. Only the final range of a generation is shorter, and a
|
|
78
|
+
#: publisher that would rather split a range than refuse an oversized member changes the call bound
|
|
79
|
+
#: the whole design rests on -- so it refuses instead.
|
|
80
|
+
RANGE_ENTRIES = 1_000
|
|
81
|
+
|
|
82
|
+
MAX_RANGE_DESCRIPTORS = 1_024
|
|
83
|
+
MAX_RANGE_DESCRIPTOR_BYTES = 2_048
|
|
84
|
+
MAX_PACKED_HEAD_BYTES = 4 * 1024 * 1024
|
|
85
|
+
|
|
86
|
+
MAX_FACTS_MEMBER_BYTES = 8 * 1024 * 1024
|
|
87
|
+
MAX_HISTORY_MEMBER_BYTES = 8 * 1024 * 1024
|
|
88
|
+
|
|
89
|
+
MAX_RANGE_MANIFEST_BYTES = 64 * 1024
|
|
90
|
+
MAX_RANGE_MEMBER_DESCRIPTORS = 24
|
|
91
|
+
MAX_MEMBER_DESCRIPTOR_BYTES = 1_024
|
|
92
|
+
|
|
93
|
+
#: The widest backend this format admits. 384 is the MiniLM coordinate's dimension, and the number
|
|
94
|
+
#: is a law rather than a default: it is what makes the vector-member ceiling a constant.
|
|
95
|
+
MAX_BACKEND_DIMENSION = 384
|
|
96
|
+
MEMBER_HEADER_BYTES = 64
|
|
97
|
+
MAX_VECTOR_MEMBER_BYTES = MEMBER_HEADER_BYTES + RANGE_ENTRIES * MAX_BACKEND_DIMENSION * 4
|
|
98
|
+
|
|
99
|
+
BOUND_SEGMENT_LAYERS = len(EMBEDDING_LAYERS)
|
|
100
|
+
MIN_POSTING_SEGMENTS = 1
|
|
101
|
+
MAX_POSTING_SEGMENTS = 8
|
|
102
|
+
MAX_POSTING_SEGMENT_BYTES = 4 * 1024 * 1024
|
|
103
|
+
POSTING_TERM_FORMAT = ">HHBxi"
|
|
104
|
+
POSTING_TERM_BYTES = struct.calcsize(POSTING_TERM_FORMAT)
|
|
105
|
+
MAX_POSTING_SEGMENT_TERMS = (MAX_POSTING_SEGMENT_BYTES - MEMBER_HEADER_BYTES) // POSTING_TERM_BYTES
|
|
106
|
+
|
|
107
|
+
#: Two backends (one query-safe lexical, one governed neural) is what the 24-descriptor range
|
|
108
|
+
#: manifest admits alongside eight posting segments. A third would not fit, and widening the
|
|
109
|
+
#: manifest to make it fit would widen every reader's allocation ceiling.
|
|
110
|
+
MAX_PACKED_BACKENDS = 2
|
|
111
|
+
|
|
112
|
+
PACKED_CLASSIFICATIONS = ("new", "changed", "unchanged")
|
|
113
|
+
MEMBER_KINDS = ("facts", "history", "vector", "posting", "bound", "range")
|
|
114
|
+
_BINARY_KINDS = ("vector", "posting", "bound")
|
|
115
|
+
|
|
116
|
+
MIN_INT32 = -(1 << 31)
|
|
117
|
+
MAX_INT32 = (1 << 31) - 1
|
|
118
|
+
|
|
119
|
+
PACKED_FORMAT_VERSION = 1
|
|
120
|
+
MEMBER_MAGIC = b"MRPACKD1"
|
|
121
|
+
_HEADER_FORMAT = ">8sHHHHIIII16s16s"
|
|
122
|
+
_KIND_CODES = {"vector": 1, "posting": 2, "bound": 3}
|
|
123
|
+
_KIND_NAMES = {code: name for name, code in _KIND_CODES.items()}
|
|
124
|
+
SUBINDEX_NONE = 0xFFFF
|
|
125
|
+
|
|
126
|
+
_OPEN_DIRECTORY = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0)
|
|
127
|
+
_OPEN_MEMBER = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0)
|
|
128
|
+
|
|
129
|
+
_SHA256 = re.compile(r"^[0-9a-f]{64}$")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
class CatalogPackedRefused(SourceContractError):
|
|
133
|
+
"""A stable refusal of a packed member, descriptor, manifest, head, or durable output."""
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _refuse(code: str, path: str, detail: str) -> CatalogPackedRefused:
|
|
137
|
+
return CatalogPackedRefused(code, path, detail)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _is_digest(value: Any) -> bool:
|
|
141
|
+
return isinstance(value, str) and _SHA256.fullmatch(value) is not None
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
# ------------------------------------------------------------------------------------------
|
|
145
|
+
# Exact byte laws
|
|
146
|
+
# ------------------------------------------------------------------------------------------
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def vector_member_bytes(*, entry_count: int, dimensions: int) -> int:
|
|
150
|
+
"""The exact size of one layer-specific vector member. There is no other admissible size."""
|
|
151
|
+
|
|
152
|
+
return MEMBER_HEADER_BYTES + entry_count * dimensions * 4
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def bound_segment_bytes(*, dimensions: int) -> int:
|
|
156
|
+
"""The exact size of one backend's bound segment: four layers of per-dimension min and max."""
|
|
157
|
+
|
|
158
|
+
return MEMBER_HEADER_BYTES + BOUND_SEGMENT_LAYERS * dimensions * 2 * 4
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def member_filename(kind: str, sha256: str) -> str:
|
|
162
|
+
"""Content-addressed member name. Binary tables and canonical JSON never share an extension."""
|
|
163
|
+
|
|
164
|
+
if kind not in MEMBER_KINDS:
|
|
165
|
+
raise _refuse("PACKED_MEMBER_KIND", "packed.member.kind", f"unknown member kind {kind!r}")
|
|
166
|
+
if not _is_digest(sha256):
|
|
167
|
+
raise _refuse("PACKED_MEMBER_DIGEST", "packed.member.sha256", "must be a lowercase SHA-256")
|
|
168
|
+
return f"{sha256}.bin" if kind in _BINARY_KINDS else f"{sha256}.json"
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _check_dimensions(dimensions: Any, path: str) -> int:
|
|
172
|
+
if type(dimensions) is not int or not 1 <= dimensions <= MAX_BACKEND_DIMENSION:
|
|
173
|
+
raise _refuse(
|
|
174
|
+
"PACKED_MEMBER_DIMENSION",
|
|
175
|
+
path,
|
|
176
|
+
f"a packed backend dimension must be an integer in [1, {MAX_BACKEND_DIMENSION}]",
|
|
177
|
+
)
|
|
178
|
+
return dimensions
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _check_entry_count(entry_count: Any, path: str) -> int:
|
|
182
|
+
if type(entry_count) is not int or not 1 <= entry_count <= RANGE_ENTRIES:
|
|
183
|
+
raise _refuse(
|
|
184
|
+
"PACKED_MEMBER_ENTRY_COUNT",
|
|
185
|
+
path,
|
|
186
|
+
f"a range carries 1 to {RANGE_ENTRIES} entries",
|
|
187
|
+
)
|
|
188
|
+
return entry_count
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
# ------------------------------------------------------------------------------------------
|
|
192
|
+
# The 64-byte fixed-endian member header
|
|
193
|
+
# ------------------------------------------------------------------------------------------
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
@dataclass(frozen=True)
|
|
197
|
+
class PackedMemberHeader:
|
|
198
|
+
"""Everything a reader must know before it allocates one byte of a member's payload."""
|
|
199
|
+
|
|
200
|
+
kind: str
|
|
201
|
+
subindex: int
|
|
202
|
+
dimensions: int
|
|
203
|
+
entry_count: int
|
|
204
|
+
term_count: int
|
|
205
|
+
range_index: int
|
|
206
|
+
payload_bytes: int
|
|
207
|
+
binding_prefix: bytes
|
|
208
|
+
backend_prefix: bytes
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _binding_prefix(binding_sha256: str) -> bytes:
|
|
212
|
+
if not _is_digest(binding_sha256):
|
|
213
|
+
raise _refuse(
|
|
214
|
+
"PACKED_MEMBER_BINDING",
|
|
215
|
+
"packed.member.range_binding_sha256",
|
|
216
|
+
"a member binds its range by that range's exact binding digest",
|
|
217
|
+
)
|
|
218
|
+
return bytes.fromhex(binding_sha256)[:16]
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _backend_prefix(backend_coordinate: str) -> bytes:
|
|
222
|
+
if not isinstance(backend_coordinate, str) or not backend_coordinate:
|
|
223
|
+
raise _refuse(
|
|
224
|
+
"PACKED_MEMBER_BINDING",
|
|
225
|
+
"packed.member.backend_coordinate",
|
|
226
|
+
"a member binds the exact backend coordinate that produced it",
|
|
227
|
+
)
|
|
228
|
+
return bytes.fromhex(sha256_bytes(backend_coordinate.encode("utf-8")))[:16]
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def pack_member_header(
|
|
232
|
+
*,
|
|
233
|
+
kind: str,
|
|
234
|
+
subindex: int,
|
|
235
|
+
dimensions: int,
|
|
236
|
+
entry_count: int,
|
|
237
|
+
term_count: int,
|
|
238
|
+
range_index: int,
|
|
239
|
+
payload_bytes: int,
|
|
240
|
+
binding_sha256: str,
|
|
241
|
+
backend_coordinate: str,
|
|
242
|
+
) -> bytes:
|
|
243
|
+
"""Build one exact 64-byte member header. Fixed big-endian, no padding a reader must guess."""
|
|
244
|
+
|
|
245
|
+
if kind not in _KIND_CODES:
|
|
246
|
+
raise _refuse("PACKED_MEMBER_KIND", "packed.member.kind", f"unknown member kind {kind!r}")
|
|
247
|
+
if type(range_index) is not int or not 0 <= range_index < MAX_RANGE_DESCRIPTORS:
|
|
248
|
+
raise _refuse(
|
|
249
|
+
"PACKED_MEMBER_HEADER",
|
|
250
|
+
"packed.member.range_index",
|
|
251
|
+
f"a range index must be an integer in [0, {MAX_RANGE_DESCRIPTORS - 1}]",
|
|
252
|
+
)
|
|
253
|
+
if type(subindex) is not int or not 0 <= subindex <= SUBINDEX_NONE:
|
|
254
|
+
raise _refuse(
|
|
255
|
+
"PACKED_MEMBER_HEADER", "packed.member.subindex", "a member subindex is a uint16"
|
|
256
|
+
)
|
|
257
|
+
raw = struct.pack(
|
|
258
|
+
_HEADER_FORMAT,
|
|
259
|
+
MEMBER_MAGIC,
|
|
260
|
+
PACKED_FORMAT_VERSION,
|
|
261
|
+
_KIND_CODES[kind],
|
|
262
|
+
subindex,
|
|
263
|
+
dimensions,
|
|
264
|
+
entry_count,
|
|
265
|
+
term_count,
|
|
266
|
+
range_index,
|
|
267
|
+
payload_bytes,
|
|
268
|
+
_binding_prefix(binding_sha256),
|
|
269
|
+
_backend_prefix(backend_coordinate),
|
|
270
|
+
)
|
|
271
|
+
if len(raw) != MEMBER_HEADER_BYTES: # pragma: no cover - the format is fixed
|
|
272
|
+
raise _refuse("PACKED_MEMBER_HEADER", "packed.member", "header is not exactly 64 bytes")
|
|
273
|
+
return raw
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def parse_member_header(raw: bytes) -> PackedMemberHeader:
|
|
277
|
+
"""Read the fixed header, or refuse. Nothing is allocated from a value read after this."""
|
|
278
|
+
|
|
279
|
+
if not isinstance(raw, bytes | bytearray | memoryview):
|
|
280
|
+
raise _refuse("PACKED_MEMBER_HEADER", "packed.member", "a member is raw bytes")
|
|
281
|
+
raw = bytes(raw)
|
|
282
|
+
if len(raw) < MEMBER_HEADER_BYTES:
|
|
283
|
+
raise _refuse(
|
|
284
|
+
"PACKED_MEMBER_TRUNCATED",
|
|
285
|
+
"packed.member",
|
|
286
|
+
f"a member is at least its {MEMBER_HEADER_BYTES}-byte header",
|
|
287
|
+
)
|
|
288
|
+
(
|
|
289
|
+
magic,
|
|
290
|
+
version,
|
|
291
|
+
kind_code,
|
|
292
|
+
subindex,
|
|
293
|
+
dimensions,
|
|
294
|
+
entry_count,
|
|
295
|
+
term_count,
|
|
296
|
+
range_index,
|
|
297
|
+
payload_bytes,
|
|
298
|
+
binding_prefix,
|
|
299
|
+
backend_prefix,
|
|
300
|
+
) = struct.unpack(_HEADER_FORMAT, raw[:MEMBER_HEADER_BYTES])
|
|
301
|
+
if magic != MEMBER_MAGIC:
|
|
302
|
+
raise _refuse("PACKED_MEMBER_MAGIC", "packed.member.magic", "not a packed catalog member")
|
|
303
|
+
if version != PACKED_FORMAT_VERSION:
|
|
304
|
+
raise _refuse(
|
|
305
|
+
"PACKED_MEMBER_VERSION",
|
|
306
|
+
"packed.member.format_version",
|
|
307
|
+
f"this reader reads format version {PACKED_FORMAT_VERSION} only",
|
|
308
|
+
)
|
|
309
|
+
if kind_code not in _KIND_NAMES:
|
|
310
|
+
raise _refuse("PACKED_MEMBER_KIND", "packed.member.kind", "unknown member kind code")
|
|
311
|
+
_check_dimensions(dimensions, "packed.member.dimensions")
|
|
312
|
+
_check_entry_count(entry_count, "packed.member.entry_count")
|
|
313
|
+
if range_index >= MAX_RANGE_DESCRIPTORS:
|
|
314
|
+
raise _refuse(
|
|
315
|
+
"PACKED_MEMBER_HEADER", "packed.member.range_index", "range index exceeds its bound"
|
|
316
|
+
)
|
|
317
|
+
return PackedMemberHeader(
|
|
318
|
+
kind=_KIND_NAMES[kind_code],
|
|
319
|
+
subindex=subindex,
|
|
320
|
+
dimensions=dimensions,
|
|
321
|
+
entry_count=entry_count,
|
|
322
|
+
term_count=term_count,
|
|
323
|
+
range_index=range_index,
|
|
324
|
+
payload_bytes=payload_bytes,
|
|
325
|
+
binding_prefix=binding_prefix,
|
|
326
|
+
backend_prefix=backend_prefix,
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _payload(raw: bytes, header: PackedMemberHeader, *, expected_payload: int) -> bytes:
|
|
331
|
+
if header.payload_bytes != expected_payload:
|
|
332
|
+
raise _refuse(
|
|
333
|
+
"PACKED_MEMBER_HEADER",
|
|
334
|
+
"packed.member.payload_bytes",
|
|
335
|
+
"the declared payload length disagrees with this member's exact byte law",
|
|
336
|
+
)
|
|
337
|
+
total = MEMBER_HEADER_BYTES + expected_payload
|
|
338
|
+
if len(raw) < total:
|
|
339
|
+
raise _refuse(
|
|
340
|
+
"PACKED_MEMBER_TRUNCATED",
|
|
341
|
+
"packed.member",
|
|
342
|
+
"the member ends before its declared payload",
|
|
343
|
+
)
|
|
344
|
+
if len(raw) > total:
|
|
345
|
+
raise _refuse(
|
|
346
|
+
"PACKED_MEMBER_EXCESS",
|
|
347
|
+
"packed.member",
|
|
348
|
+
"a member is exactly its header plus its payload; trailing bytes are refused",
|
|
349
|
+
)
|
|
350
|
+
return raw[MEMBER_HEADER_BYTES:]
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def _check_binding(
|
|
354
|
+
header: PackedMemberHeader,
|
|
355
|
+
descriptor: PackedMemberDescriptor | None,
|
|
356
|
+
*,
|
|
357
|
+
kind: str,
|
|
358
|
+
) -> None:
|
|
359
|
+
if header.kind != kind:
|
|
360
|
+
raise _refuse(
|
|
361
|
+
"PACKED_MEMBER_KIND",
|
|
362
|
+
"packed.member.kind",
|
|
363
|
+
f"this member is a {header.kind} member, not a {kind} member",
|
|
364
|
+
)
|
|
365
|
+
if descriptor is None:
|
|
366
|
+
return
|
|
367
|
+
if descriptor.kind != kind:
|
|
368
|
+
raise _refuse(
|
|
369
|
+
"PACKED_MEMBER_KIND", "packed.member.kind", "the descriptor names another member kind"
|
|
370
|
+
)
|
|
371
|
+
if kind == "vector" and descriptor.layer != EMBEDDING_LAYERS[header.subindex]:
|
|
372
|
+
raise _refuse(
|
|
373
|
+
"PACKED_MEMBER_LAYER",
|
|
374
|
+
"packed.member.layer",
|
|
375
|
+
"this member carries another layer than the descriptor binds",
|
|
376
|
+
)
|
|
377
|
+
if kind == "posting" and descriptor.segment_index != header.subindex:
|
|
378
|
+
raise _refuse(
|
|
379
|
+
"PACKED_MEMBER_BINDING",
|
|
380
|
+
"packed.member.segment_index",
|
|
381
|
+
"this posting segment is not the segment the descriptor binds",
|
|
382
|
+
)
|
|
383
|
+
if (
|
|
384
|
+
descriptor.range_index != header.range_index
|
|
385
|
+
or descriptor.entry_count != header.entry_count
|
|
386
|
+
or descriptor.dimensions != header.dimensions
|
|
387
|
+
or _binding_prefix(descriptor.range_binding_sha256) != header.binding_prefix
|
|
388
|
+
or _backend_prefix(descriptor.backend_coordinate or "") != header.backend_prefix
|
|
389
|
+
):
|
|
390
|
+
raise _refuse(
|
|
391
|
+
"PACKED_MEMBER_BINDING",
|
|
392
|
+
"packed.member",
|
|
393
|
+
"this member is bound to another range, entry order, or backend",
|
|
394
|
+
)
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
# ------------------------------------------------------------------------------------------
|
|
398
|
+
# Vector members
|
|
399
|
+
# ------------------------------------------------------------------------------------------
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
@dataclass(frozen=True)
|
|
403
|
+
class PackedVectorMember:
|
|
404
|
+
"""One layer's exact integer vectors for one range and one backend, in range entry order."""
|
|
405
|
+
|
|
406
|
+
layer: str
|
|
407
|
+
dimensions: int
|
|
408
|
+
entry_count: int
|
|
409
|
+
range_index: int
|
|
410
|
+
vectors: tuple[tuple[int, ...], ...]
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def pack_vector_member(
|
|
414
|
+
vectors: Sequence[Sequence[int]],
|
|
415
|
+
*,
|
|
416
|
+
layer: str,
|
|
417
|
+
dimensions: int,
|
|
418
|
+
range_index: int,
|
|
419
|
+
binding_sha256: str,
|
|
420
|
+
backend_coordinate: str,
|
|
421
|
+
) -> bytes:
|
|
422
|
+
"""Serialize one layer's vectors as a fixed-width big-endian int32 table."""
|
|
423
|
+
|
|
424
|
+
_check_dimensions(dimensions, "packed.vector.dimensions")
|
|
425
|
+
if layer not in EMBEDDING_LAYERS:
|
|
426
|
+
raise _refuse(
|
|
427
|
+
"PACKED_MEMBER_LAYER",
|
|
428
|
+
"packed.vector.layer",
|
|
429
|
+
f"a layer member names one of {list(EMBEDDING_LAYERS)}",
|
|
430
|
+
)
|
|
431
|
+
entry_count = _check_entry_count(len(vectors), "packed.vector.entry_count")
|
|
432
|
+
payload = bytearray()
|
|
433
|
+
for ordinal, vector in enumerate(vectors):
|
|
434
|
+
if len(vector) != dimensions:
|
|
435
|
+
raise _refuse(
|
|
436
|
+
"PACKED_MEMBER_DIMENSION",
|
|
437
|
+
f"packed.vector[{ordinal}]",
|
|
438
|
+
"every vector in a member has exactly the backend's dimension",
|
|
439
|
+
)
|
|
440
|
+
for value in vector:
|
|
441
|
+
if type(value) is not int or not MIN_INT32 <= value <= MAX_INT32:
|
|
442
|
+
raise _refuse(
|
|
443
|
+
"PACKED_VECTOR_VALUE",
|
|
444
|
+
f"packed.vector[{ordinal}]",
|
|
445
|
+
"a packed vector holds exact int32 values only",
|
|
446
|
+
)
|
|
447
|
+
payload += struct.pack(f">{dimensions}i", *vector)
|
|
448
|
+
header = pack_member_header(
|
|
449
|
+
kind="vector",
|
|
450
|
+
subindex=EMBEDDING_LAYERS.index(layer),
|
|
451
|
+
dimensions=dimensions,
|
|
452
|
+
entry_count=entry_count,
|
|
453
|
+
term_count=0,
|
|
454
|
+
range_index=range_index,
|
|
455
|
+
payload_bytes=len(payload),
|
|
456
|
+
binding_sha256=binding_sha256,
|
|
457
|
+
backend_coordinate=backend_coordinate,
|
|
458
|
+
)
|
|
459
|
+
raw = header + bytes(payload)
|
|
460
|
+
if len(raw) != vector_member_bytes(entry_count=entry_count, dimensions=dimensions):
|
|
461
|
+
raise _refuse( # pragma: no cover - the arithmetic above is the law
|
|
462
|
+
"PACKED_MEMBER_HEADER", "packed.vector", "vector member violated its exact byte law"
|
|
463
|
+
)
|
|
464
|
+
if len(raw) > MAX_VECTOR_MEMBER_BYTES: # pragma: no cover - implied by the two ceilings
|
|
465
|
+
raise _refuse("PACKED_MEMBER_LIMIT", "packed.vector", "vector member exceeds its ceiling")
|
|
466
|
+
return raw
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
def parse_vector_member(
|
|
470
|
+
raw: bytes, *, descriptor: PackedMemberDescriptor | None = None
|
|
471
|
+
) -> PackedVectorMember:
|
|
472
|
+
"""Read one vector member exactly, or refuse it. Never partially."""
|
|
473
|
+
|
|
474
|
+
header = parse_member_header(raw)
|
|
475
|
+
_check_binding(header, descriptor, kind="vector")
|
|
476
|
+
if header.subindex >= len(EMBEDDING_LAYERS):
|
|
477
|
+
raise _refuse(
|
|
478
|
+
"PACKED_MEMBER_LAYER", "packed.vector.layer", "layer ordinal is outside layer closure"
|
|
479
|
+
)
|
|
480
|
+
if header.term_count != 0:
|
|
481
|
+
raise _refuse(
|
|
482
|
+
"PACKED_MEMBER_HEADER", "packed.vector.term_count", "a vector member carries no terms"
|
|
483
|
+
)
|
|
484
|
+
expected = header.entry_count * header.dimensions * 4
|
|
485
|
+
payload = _payload(bytes(raw), header, expected_payload=expected)
|
|
486
|
+
vectors: list[tuple[int, ...]] = []
|
|
487
|
+
stride = header.dimensions * 4
|
|
488
|
+
for ordinal in range(header.entry_count):
|
|
489
|
+
chunk = payload[ordinal * stride : (ordinal + 1) * stride]
|
|
490
|
+
vectors.append(struct.unpack(f">{header.dimensions}i", chunk))
|
|
491
|
+
return PackedVectorMember(
|
|
492
|
+
layer=EMBEDDING_LAYERS[header.subindex],
|
|
493
|
+
dimensions=header.dimensions,
|
|
494
|
+
entry_count=header.entry_count,
|
|
495
|
+
range_index=header.range_index,
|
|
496
|
+
vectors=tuple(vectors),
|
|
497
|
+
)
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
# ------------------------------------------------------------------------------------------
|
|
501
|
+
# Lexical posting segments
|
|
502
|
+
# ------------------------------------------------------------------------------------------
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
PostingTerm = tuple[int, int, int, int]
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
@dataclass(frozen=True)
|
|
509
|
+
class PackedPostingSegment:
|
|
510
|
+
"""One bounded run of canonical ``(dimension, entry ordinal, layer ordinal, value)`` terms."""
|
|
511
|
+
|
|
512
|
+
segment_index: int
|
|
513
|
+
dimensions: int
|
|
514
|
+
entry_count: int
|
|
515
|
+
range_index: int
|
|
516
|
+
terms: tuple[PostingTerm, ...]
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def compute_layer_postings(
|
|
520
|
+
vectors_by_layer: Mapping[str, Sequence[Sequence[int]]], *, dimensions: int
|
|
521
|
+
) -> tuple[PostingTerm, ...]:
|
|
522
|
+
"""Derive this range's canonical posting terms from its exact layer vectors.
|
|
523
|
+
|
|
524
|
+
A zero contributes nothing to any dot product, so it is not a term. Everything else is, and the
|
|
525
|
+
order is total: dimension, then entry ordinal, then layer ordinal.
|
|
526
|
+
"""
|
|
527
|
+
|
|
528
|
+
_check_dimensions(dimensions, "packed.postings.dimensions")
|
|
529
|
+
terms: list[PostingTerm] = []
|
|
530
|
+
for layer_ordinal, layer in enumerate(EMBEDDING_LAYERS):
|
|
531
|
+
vectors = vectors_by_layer.get(layer)
|
|
532
|
+
if vectors is None:
|
|
533
|
+
raise _refuse(
|
|
534
|
+
"PACKED_MEMBER_LAYER",
|
|
535
|
+
"packed.postings.layers",
|
|
536
|
+
f"postings require exactly the four layers {list(EMBEDDING_LAYERS)}",
|
|
537
|
+
)
|
|
538
|
+
for ordinal, vector in enumerate(vectors):
|
|
539
|
+
for dimension, value in enumerate(vector):
|
|
540
|
+
if value:
|
|
541
|
+
terms.append((dimension, ordinal, layer_ordinal, value))
|
|
542
|
+
terms.sort()
|
|
543
|
+
return tuple(terms)
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
def segment_posting_terms(
|
|
547
|
+
terms: Sequence[PostingTerm], *, max_terms_per_segment: int | None = None
|
|
548
|
+
) -> tuple[tuple[PostingTerm, ...], ...]:
|
|
549
|
+
"""Split canonical terms into 1-8 bounded segments, or refuse rather than widen the cap."""
|
|
550
|
+
|
|
551
|
+
limit = MAX_POSTING_SEGMENT_TERMS if max_terms_per_segment is None else max_terms_per_segment
|
|
552
|
+
if type(limit) is not int or not 1 <= limit <= MAX_POSTING_SEGMENT_TERMS:
|
|
553
|
+
raise _refuse(
|
|
554
|
+
"PACKED_POSTING_SEGMENTS",
|
|
555
|
+
"packed.postings.max_terms_per_segment",
|
|
556
|
+
f"a segment carries 1 to {MAX_POSTING_SEGMENT_TERMS} terms",
|
|
557
|
+
)
|
|
558
|
+
segments = tuple(
|
|
559
|
+
tuple(terms[start : start + limit]) for start in range(0, max(len(terms), 1), limit)
|
|
560
|
+
)
|
|
561
|
+
if not MIN_POSTING_SEGMENTS <= len(segments) <= MAX_POSTING_SEGMENTS:
|
|
562
|
+
raise _refuse(
|
|
563
|
+
"PACKED_POSTING_SEGMENTS",
|
|
564
|
+
"packed.postings.segments",
|
|
565
|
+
f"a range carries {MIN_POSTING_SEGMENTS} to {MAX_POSTING_SEGMENTS} posting segments; "
|
|
566
|
+
f"{len(segments)} would be needed",
|
|
567
|
+
)
|
|
568
|
+
return segments
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
def pack_posting_segment(
|
|
572
|
+
terms: Sequence[PostingTerm],
|
|
573
|
+
*,
|
|
574
|
+
segment_index: int,
|
|
575
|
+
dimensions: int,
|
|
576
|
+
entry_count: int,
|
|
577
|
+
range_index: int,
|
|
578
|
+
binding_sha256: str,
|
|
579
|
+
backend_coordinate: str,
|
|
580
|
+
) -> bytes:
|
|
581
|
+
"""Serialize one posting segment as a fixed-width big-endian term table."""
|
|
582
|
+
|
|
583
|
+
_check_dimensions(dimensions, "packed.postings.dimensions")
|
|
584
|
+
_check_entry_count(entry_count, "packed.postings.entry_count")
|
|
585
|
+
_validate_posting_terms(terms, dimensions=dimensions, entry_count=entry_count)
|
|
586
|
+
payload = bytearray()
|
|
587
|
+
for dimension, ordinal, layer_ordinal, value in terms:
|
|
588
|
+
payload += struct.pack(POSTING_TERM_FORMAT, dimension, ordinal, layer_ordinal, value)
|
|
589
|
+
if MEMBER_HEADER_BYTES + len(payload) > MAX_POSTING_SEGMENT_BYTES:
|
|
590
|
+
raise _refuse(
|
|
591
|
+
"PACKED_POSTING_LIMIT",
|
|
592
|
+
"packed.postings.segment",
|
|
593
|
+
f"one posting segment holds at most {MAX_POSTING_SEGMENT_BYTES} bytes",
|
|
594
|
+
)
|
|
595
|
+
header = pack_member_header(
|
|
596
|
+
kind="posting",
|
|
597
|
+
subindex=segment_index,
|
|
598
|
+
dimensions=dimensions,
|
|
599
|
+
entry_count=entry_count,
|
|
600
|
+
term_count=len(terms),
|
|
601
|
+
range_index=range_index,
|
|
602
|
+
payload_bytes=len(payload),
|
|
603
|
+
binding_sha256=binding_sha256,
|
|
604
|
+
backend_coordinate=backend_coordinate,
|
|
605
|
+
)
|
|
606
|
+
return header + bytes(payload)
|
|
607
|
+
|
|
608
|
+
|
|
609
|
+
def _validate_posting_terms(
|
|
610
|
+
terms: Sequence[PostingTerm], *, dimensions: int, entry_count: int
|
|
611
|
+
) -> None:
|
|
612
|
+
previous: PostingTerm | None = None
|
|
613
|
+
for position, term in enumerate(terms):
|
|
614
|
+
if len(term) != 4:
|
|
615
|
+
raise _refuse(
|
|
616
|
+
"PACKED_POSTING_TERM", f"packed.postings[{position}]", "a term is exactly four ints"
|
|
617
|
+
)
|
|
618
|
+
dimension, ordinal, layer_ordinal, value = term
|
|
619
|
+
if (
|
|
620
|
+
type(dimension) is not int
|
|
621
|
+
or not 0 <= dimension < dimensions
|
|
622
|
+
or type(ordinal) is not int
|
|
623
|
+
or not 0 <= ordinal < entry_count
|
|
624
|
+
or type(layer_ordinal) is not int
|
|
625
|
+
or not 0 <= layer_ordinal < len(EMBEDDING_LAYERS)
|
|
626
|
+
or type(value) is not int
|
|
627
|
+
or not MIN_INT32 <= value <= MAX_INT32
|
|
628
|
+
or value == 0
|
|
629
|
+
):
|
|
630
|
+
raise _refuse(
|
|
631
|
+
"PACKED_POSTING_TERM",
|
|
632
|
+
f"packed.postings[{position}]",
|
|
633
|
+
"a term names an in-range dimension, entry ordinal, layer ordinal and int32",
|
|
634
|
+
)
|
|
635
|
+
if previous is not None and term <= previous:
|
|
636
|
+
raise _refuse(
|
|
637
|
+
"PACKED_POSTING_ORDER",
|
|
638
|
+
f"packed.postings[{position}]",
|
|
639
|
+
"posting terms ascend by dimension, entry ordinal, then layer ordinal",
|
|
640
|
+
)
|
|
641
|
+
previous = term
|
|
642
|
+
|
|
643
|
+
|
|
644
|
+
def parse_posting_segment(
|
|
645
|
+
raw: bytes, *, descriptor: PackedMemberDescriptor | None = None
|
|
646
|
+
) -> PackedPostingSegment:
|
|
647
|
+
"""Read one posting segment exactly, refusing a term outside the range's own closure."""
|
|
648
|
+
|
|
649
|
+
header = parse_member_header(raw)
|
|
650
|
+
_check_binding(header, descriptor, kind="posting")
|
|
651
|
+
if len(raw) > MAX_POSTING_SEGMENT_BYTES:
|
|
652
|
+
raise _refuse(
|
|
653
|
+
"PACKED_POSTING_LIMIT", "packed.postings.segment", "segment exceeds its byte ceiling"
|
|
654
|
+
)
|
|
655
|
+
if header.term_count > MAX_POSTING_SEGMENT_TERMS:
|
|
656
|
+
raise _refuse(
|
|
657
|
+
"PACKED_POSTING_LIMIT", "packed.postings.term_count", "segment exceeds its term ceiling"
|
|
658
|
+
)
|
|
659
|
+
expected = header.term_count * POSTING_TERM_BYTES
|
|
660
|
+
payload = _payload(bytes(raw), header, expected_payload=expected)
|
|
661
|
+
terms = tuple(
|
|
662
|
+
struct.unpack_from(POSTING_TERM_FORMAT, payload, position * POSTING_TERM_BYTES)
|
|
663
|
+
for position in range(header.term_count)
|
|
664
|
+
)
|
|
665
|
+
_validate_posting_terms(terms, dimensions=header.dimensions, entry_count=header.entry_count)
|
|
666
|
+
return PackedPostingSegment(
|
|
667
|
+
segment_index=header.subindex,
|
|
668
|
+
dimensions=header.dimensions,
|
|
669
|
+
entry_count=header.entry_count,
|
|
670
|
+
range_index=header.range_index,
|
|
671
|
+
terms=terms,
|
|
672
|
+
)
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
# ------------------------------------------------------------------------------------------
|
|
676
|
+
# Bound segments and the conservative bound
|
|
677
|
+
# ------------------------------------------------------------------------------------------
|
|
678
|
+
|
|
679
|
+
|
|
680
|
+
LayerBounds = tuple[tuple[int, ...], tuple[int, ...]]
|
|
681
|
+
|
|
682
|
+
|
|
683
|
+
@dataclass(frozen=True)
|
|
684
|
+
class PackedBoundSegment:
|
|
685
|
+
"""One backend's exact per-layer, per-dimension minima and maxima for one range."""
|
|
686
|
+
|
|
687
|
+
dimensions: int
|
|
688
|
+
entry_count: int
|
|
689
|
+
range_index: int
|
|
690
|
+
bounds: tuple[LayerBounds, ...]
|
|
691
|
+
|
|
692
|
+
|
|
693
|
+
def compute_layer_bounds(
|
|
694
|
+
vectors_by_layer: Mapping[str, Sequence[Sequence[int]]], *, dimensions: int
|
|
695
|
+
) -> tuple[LayerBounds, ...]:
|
|
696
|
+
"""Derive the exact per-dimension minima and maxima of each layer in this range."""
|
|
697
|
+
|
|
698
|
+
_check_dimensions(dimensions, "packed.bounds.dimensions")
|
|
699
|
+
bounds: list[LayerBounds] = []
|
|
700
|
+
for layer in EMBEDDING_LAYERS:
|
|
701
|
+
vectors = vectors_by_layer.get(layer)
|
|
702
|
+
if not vectors:
|
|
703
|
+
raise _refuse(
|
|
704
|
+
"PACKED_MEMBER_LAYER",
|
|
705
|
+
"packed.bounds.layers",
|
|
706
|
+
f"bounds require exactly the four layers {list(EMBEDDING_LAYERS)}",
|
|
707
|
+
)
|
|
708
|
+
minima = [MAX_INT32] * dimensions
|
|
709
|
+
maxima = [MIN_INT32] * dimensions
|
|
710
|
+
for vector in vectors:
|
|
711
|
+
if len(vector) != dimensions:
|
|
712
|
+
raise _refuse(
|
|
713
|
+
"PACKED_MEMBER_DIMENSION",
|
|
714
|
+
"packed.bounds.vectors",
|
|
715
|
+
"every vector summarized has exactly the backend's dimension",
|
|
716
|
+
)
|
|
717
|
+
for dimension, value in enumerate(vector):
|
|
718
|
+
if value < minima[dimension]:
|
|
719
|
+
minima[dimension] = value
|
|
720
|
+
if value > maxima[dimension]:
|
|
721
|
+
maxima[dimension] = value
|
|
722
|
+
bounds.append((tuple(minima), tuple(maxima)))
|
|
723
|
+
return tuple(bounds)
|
|
724
|
+
|
|
725
|
+
|
|
726
|
+
def pack_bound_segment(
|
|
727
|
+
bounds: Sequence[LayerBounds],
|
|
728
|
+
*,
|
|
729
|
+
dimensions: int,
|
|
730
|
+
entry_count: int,
|
|
731
|
+
range_index: int,
|
|
732
|
+
binding_sha256: str,
|
|
733
|
+
backend_coordinate: str,
|
|
734
|
+
) -> bytes:
|
|
735
|
+
"""Serialize one backend's four-layer min/max summary as a fixed-width int32 table."""
|
|
736
|
+
|
|
737
|
+
_check_dimensions(dimensions, "packed.bounds.dimensions")
|
|
738
|
+
_check_entry_count(entry_count, "packed.bounds.entry_count")
|
|
739
|
+
if len(bounds) != BOUND_SEGMENT_LAYERS:
|
|
740
|
+
raise _refuse(
|
|
741
|
+
"PACKED_BOUND_SEGMENT",
|
|
742
|
+
"packed.bounds",
|
|
743
|
+
f"a bound segment summarizes exactly {BOUND_SEGMENT_LAYERS} layers",
|
|
744
|
+
)
|
|
745
|
+
payload = bytearray()
|
|
746
|
+
for layer_ordinal, layer_bounds in enumerate(bounds):
|
|
747
|
+
if len(layer_bounds) != 2:
|
|
748
|
+
raise _refuse(
|
|
749
|
+
"PACKED_BOUND_SEGMENT",
|
|
750
|
+
f"packed.bounds[{layer_ordinal}]",
|
|
751
|
+
"each layer summary is exactly one minima and one maxima table",
|
|
752
|
+
)
|
|
753
|
+
for values in layer_bounds:
|
|
754
|
+
if len(values) != dimensions:
|
|
755
|
+
raise _refuse(
|
|
756
|
+
"PACKED_BOUND_SEGMENT",
|
|
757
|
+
f"packed.bounds[{layer_ordinal}]",
|
|
758
|
+
"each summary table has exactly the backend's dimension",
|
|
759
|
+
)
|
|
760
|
+
for value in values:
|
|
761
|
+
if type(value) is not int or not MIN_INT32 <= value <= MAX_INT32:
|
|
762
|
+
raise _refuse(
|
|
763
|
+
"PACKED_BOUND_SEGMENT",
|
|
764
|
+
f"packed.bounds[{layer_ordinal}]",
|
|
765
|
+
"a bound summary holds exact int32 values only",
|
|
766
|
+
)
|
|
767
|
+
payload += struct.pack(f">{dimensions}i", *values)
|
|
768
|
+
header = pack_member_header(
|
|
769
|
+
kind="bound",
|
|
770
|
+
subindex=SUBINDEX_NONE,
|
|
771
|
+
dimensions=dimensions,
|
|
772
|
+
entry_count=entry_count,
|
|
773
|
+
term_count=0,
|
|
774
|
+
range_index=range_index,
|
|
775
|
+
payload_bytes=len(payload),
|
|
776
|
+
binding_sha256=binding_sha256,
|
|
777
|
+
backend_coordinate=backend_coordinate,
|
|
778
|
+
)
|
|
779
|
+
raw = header + bytes(payload)
|
|
780
|
+
if len(raw) != bound_segment_bytes(dimensions=dimensions):
|
|
781
|
+
raise _refuse( # pragma: no cover - the arithmetic above is the law
|
|
782
|
+
"PACKED_BOUND_SEGMENT", "packed.bounds", "bound segment violated its exact byte law"
|
|
783
|
+
)
|
|
784
|
+
return raw
|
|
785
|
+
|
|
786
|
+
|
|
787
|
+
def parse_bound_segment(
|
|
788
|
+
raw: bytes, *, descriptor: PackedMemberDescriptor | None = None
|
|
789
|
+
) -> PackedBoundSegment:
|
|
790
|
+
"""Read one bound segment exactly, refusing an inverted or mis-shaped summary."""
|
|
791
|
+
|
|
792
|
+
header = parse_member_header(raw)
|
|
793
|
+
_check_binding(header, descriptor, kind="bound")
|
|
794
|
+
if header.subindex != SUBINDEX_NONE:
|
|
795
|
+
raise _refuse(
|
|
796
|
+
"PACKED_BOUND_SEGMENT",
|
|
797
|
+
"packed.bounds.subindex",
|
|
798
|
+
"a bound segment summarizes every layer and carries no subindex",
|
|
799
|
+
)
|
|
800
|
+
expected = BOUND_SEGMENT_LAYERS * header.dimensions * 2 * 4
|
|
801
|
+
payload = _payload(bytes(raw), header, expected_payload=expected)
|
|
802
|
+
stride = header.dimensions * 4
|
|
803
|
+
bounds: list[LayerBounds] = []
|
|
804
|
+
for layer_ordinal in range(BOUND_SEGMENT_LAYERS):
|
|
805
|
+
base = layer_ordinal * stride * 2
|
|
806
|
+
minima = struct.unpack_from(f">{header.dimensions}i", payload, base)
|
|
807
|
+
maxima = struct.unpack_from(f">{header.dimensions}i", payload, base + stride)
|
|
808
|
+
for dimension, (low, high) in enumerate(zip(minima, maxima, strict=True)):
|
|
809
|
+
if low > high:
|
|
810
|
+
raise _refuse(
|
|
811
|
+
"PACKED_BOUND_ORDER",
|
|
812
|
+
f"packed.bounds[{layer_ordinal}][{dimension}]",
|
|
813
|
+
"a summarized minimum may never exceed its maximum",
|
|
814
|
+
)
|
|
815
|
+
bounds.append((minima, maxima))
|
|
816
|
+
return PackedBoundSegment(
|
|
817
|
+
dimensions=header.dimensions,
|
|
818
|
+
entry_count=header.entry_count,
|
|
819
|
+
range_index=header.range_index,
|
|
820
|
+
bounds=tuple(bounds),
|
|
821
|
+
)
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
def conservative_upper_bound(
|
|
825
|
+
query: Sequence[int], minima: Sequence[int], maxima: Sequence[int]
|
|
826
|
+
) -> int:
|
|
827
|
+
"""The largest dot product any vector inside this summary could have with ``query``.
|
|
828
|
+
|
|
829
|
+
Exact integers throughout: ``sum_i(q_i * max_i)`` where ``q_i >= 0`` and ``sum_i(q_i * min_i)``
|
|
830
|
+
otherwise. It is an upper bound on the *layer* score, never a score, and it is the only number
|
|
831
|
+
this module computes from a query.
|
|
832
|
+
"""
|
|
833
|
+
|
|
834
|
+
total = 0
|
|
835
|
+
for value, low, high in zip(query, minima, maxima, strict=True):
|
|
836
|
+
total += value * (high if value >= 0 else low)
|
|
837
|
+
return total
|
|
838
|
+
|
|
839
|
+
|
|
840
|
+
def range_upper_bound(query: Sequence[int], bounds: Sequence[LayerBounds]) -> int:
|
|
841
|
+
"""The range's bound: the maximum of its four layer bounds.
|
|
842
|
+
|
|
843
|
+
Prune only when this is *strictly* below the current kth score. Equality must exact-scan: a
|
|
844
|
+
range that ties on score may still win on entry id, and a pruned range cannot present an id.
|
|
845
|
+
"""
|
|
846
|
+
|
|
847
|
+
if len(bounds) != BOUND_SEGMENT_LAYERS:
|
|
848
|
+
raise _refuse(
|
|
849
|
+
"PACKED_BOUND_SEGMENT",
|
|
850
|
+
"packed.bounds",
|
|
851
|
+
f"a range bound maximizes exactly {BOUND_SEGMENT_LAYERS} layer bounds",
|
|
852
|
+
)
|
|
853
|
+
return max(conservative_upper_bound(query, minima, maxima) for minima, maxima in bounds)
|
|
854
|
+
|
|
855
|
+
|
|
856
|
+
# ------------------------------------------------------------------------------------------
|
|
857
|
+
# Descriptors, range manifests, and the outer head
|
|
858
|
+
# ------------------------------------------------------------------------------------------
|
|
859
|
+
|
|
860
|
+
|
|
861
|
+
@dataclass(frozen=True)
|
|
862
|
+
class PackedBackend:
|
|
863
|
+
"""One available backend's published coordinate, exactly as the vectors were produced under."""
|
|
864
|
+
|
|
865
|
+
coordinate: str
|
|
866
|
+
dimensions: int
|
|
867
|
+
quantization: str
|
|
868
|
+
query_safe: bool
|
|
869
|
+
|
|
870
|
+
def __post_init__(self) -> None:
|
|
871
|
+
if not isinstance(self.coordinate, str) or not 1 <= len(self.coordinate) <= 256:
|
|
872
|
+
raise _refuse(
|
|
873
|
+
"PACKED_BACKEND", "packed.backend.coordinate", "must be a bounded coordinate string"
|
|
874
|
+
)
|
|
875
|
+
_check_dimensions(self.dimensions, "packed.backend.dimensions")
|
|
876
|
+
if not isinstance(self.quantization, str) or not self.quantization:
|
|
877
|
+
raise _refuse(
|
|
878
|
+
"PACKED_BACKEND", "packed.backend.quantization", "must name its exact quantization"
|
|
879
|
+
)
|
|
880
|
+
if type(self.query_safe) is not bool:
|
|
881
|
+
raise _refuse("PACKED_BACKEND", "packed.backend.query_safe", "must be a boolean")
|
|
882
|
+
|
|
883
|
+
def to_dict(self) -> dict[str, Any]:
|
|
884
|
+
return {
|
|
885
|
+
"coordinate": self.coordinate,
|
|
886
|
+
"dimensions": self.dimensions,
|
|
887
|
+
"quantization": self.quantization,
|
|
888
|
+
"query_safe": self.query_safe,
|
|
889
|
+
}
|
|
890
|
+
|
|
891
|
+
|
|
892
|
+
@dataclass(frozen=True)
|
|
893
|
+
class PackedMemberDescriptor:
|
|
894
|
+
"""One member's exact identity: what it is, where it belongs, and its byte-for-byte digest."""
|
|
895
|
+
|
|
896
|
+
kind: str
|
|
897
|
+
sha256: str
|
|
898
|
+
bytes: int
|
|
899
|
+
range_index: int
|
|
900
|
+
first_key: str
|
|
901
|
+
last_key: str
|
|
902
|
+
entry_count: int
|
|
903
|
+
facts_sha256: str
|
|
904
|
+
range_binding_sha256: str
|
|
905
|
+
backend_coordinate: str | None = None
|
|
906
|
+
dimensions: int | None = None
|
|
907
|
+
layer: str | None = None
|
|
908
|
+
segment_index: int | None = None
|
|
909
|
+
term_count: int | None = None
|
|
910
|
+
|
|
911
|
+
def __post_init__(self) -> None:
|
|
912
|
+
if self.kind not in MEMBER_KINDS:
|
|
913
|
+
raise _refuse("PACKED_MEMBER_KIND", "packed.descriptor.kind", "unknown member kind")
|
|
914
|
+
if not _is_digest(self.sha256) or not _is_digest(self.range_binding_sha256):
|
|
915
|
+
raise _refuse(
|
|
916
|
+
"PACKED_MEMBER_DIGEST", "packed.descriptor", "member digests are lowercase SHA-256"
|
|
917
|
+
)
|
|
918
|
+
if not _is_digest(self.facts_sha256):
|
|
919
|
+
raise _refuse(
|
|
920
|
+
"PACKED_MEMBER_DIGEST",
|
|
921
|
+
"packed.descriptor.facts_sha256",
|
|
922
|
+
"a descriptor binds its range's exact facts order digest",
|
|
923
|
+
)
|
|
924
|
+
if type(self.bytes) is not int or self.bytes < 0:
|
|
925
|
+
raise _refuse("PACKED_MEMBER_HEADER", "packed.descriptor.bytes", "must be a byte count")
|
|
926
|
+
if self.layer is not None and self.layer not in EMBEDDING_LAYERS:
|
|
927
|
+
raise _refuse("PACKED_MEMBER_LAYER", "packed.descriptor.layer", "unknown layer")
|
|
928
|
+
|
|
929
|
+
def to_dict(self) -> dict[str, Any]:
|
|
930
|
+
return {
|
|
931
|
+
"kind": self.kind,
|
|
932
|
+
"sha256": self.sha256,
|
|
933
|
+
"bytes": self.bytes,
|
|
934
|
+
"range_index": self.range_index,
|
|
935
|
+
"first_key": self.first_key,
|
|
936
|
+
"last_key": self.last_key,
|
|
937
|
+
"entry_count": self.entry_count,
|
|
938
|
+
"facts_sha256": self.facts_sha256,
|
|
939
|
+
"range_binding_sha256": self.range_binding_sha256,
|
|
940
|
+
"backend_coordinate": self.backend_coordinate,
|
|
941
|
+
"dimensions": self.dimensions,
|
|
942
|
+
"layer": self.layer,
|
|
943
|
+
"segment_index": self.segment_index,
|
|
944
|
+
"term_count": self.term_count,
|
|
945
|
+
}
|
|
946
|
+
|
|
947
|
+
def replace(self, **changes: Any) -> PackedMemberDescriptor:
|
|
948
|
+
return replace(self, **changes)
|
|
949
|
+
|
|
950
|
+
|
|
951
|
+
_MEMBER_DESCRIPTOR_MEMBERS = frozenset(PackedMemberDescriptor.__dataclass_fields__)
|
|
952
|
+
|
|
953
|
+
|
|
954
|
+
def member_descriptor_from_dict(value: Any, *, path: str) -> PackedMemberDescriptor:
|
|
955
|
+
if not isinstance(value, dict) or set(value) != _MEMBER_DESCRIPTOR_MEMBERS:
|
|
956
|
+
raise _refuse("PACKED_MEMBER_HEADER", path, "a member descriptor contract differs")
|
|
957
|
+
if len(canonical_json_bytes(value)) > MAX_MEMBER_DESCRIPTOR_BYTES:
|
|
958
|
+
raise _refuse(
|
|
959
|
+
"PACKED_DESCRIPTOR_LIMIT",
|
|
960
|
+
path,
|
|
961
|
+
f"a member descriptor holds at most {MAX_MEMBER_DESCRIPTOR_BYTES} bytes",
|
|
962
|
+
)
|
|
963
|
+
return PackedMemberDescriptor(**value)
|
|
964
|
+
|
|
965
|
+
|
|
966
|
+
@dataclass(frozen=True)
|
|
967
|
+
class PackedRangeDescriptor:
|
|
968
|
+
"""One range's entry in the outer head: its key interval and its manifest's exact digest."""
|
|
969
|
+
|
|
970
|
+
range_index: int
|
|
971
|
+
first_key: str
|
|
972
|
+
last_key: str
|
|
973
|
+
entry_count: int
|
|
974
|
+
facts_sha256: str
|
|
975
|
+
range_binding_sha256: str
|
|
976
|
+
manifest_sha256: str
|
|
977
|
+
manifest_bytes: int
|
|
978
|
+
member_count: int
|
|
979
|
+
|
|
980
|
+
def to_dict(self) -> dict[str, Any]:
|
|
981
|
+
return {
|
|
982
|
+
"range_index": self.range_index,
|
|
983
|
+
"first_key": self.first_key,
|
|
984
|
+
"last_key": self.last_key,
|
|
985
|
+
"entry_count": self.entry_count,
|
|
986
|
+
"facts_sha256": self.facts_sha256,
|
|
987
|
+
"range_binding_sha256": self.range_binding_sha256,
|
|
988
|
+
"manifest_sha256": self.manifest_sha256,
|
|
989
|
+
"manifest_bytes": self.manifest_bytes,
|
|
990
|
+
"member_count": self.member_count,
|
|
991
|
+
}
|
|
992
|
+
|
|
993
|
+
def replace(self, **changes: Any) -> PackedRangeDescriptor:
|
|
994
|
+
return replace(self, **changes)
|
|
995
|
+
|
|
996
|
+
|
|
997
|
+
_RANGE_DESCRIPTOR_MEMBERS = frozenset(PackedRangeDescriptor.__dataclass_fields__)
|
|
998
|
+
|
|
999
|
+
|
|
1000
|
+
def range_descriptor_from_dict(value: Any, *, path: str) -> PackedRangeDescriptor:
|
|
1001
|
+
if not isinstance(value, dict) or set(value) != _RANGE_DESCRIPTOR_MEMBERS:
|
|
1002
|
+
raise _refuse("PACKED_HEAD_DESCRIPTORS", path, "a range descriptor contract differs")
|
|
1003
|
+
if len(canonical_json_bytes(value)) > MAX_RANGE_DESCRIPTOR_BYTES:
|
|
1004
|
+
raise _refuse(
|
|
1005
|
+
"PACKED_DESCRIPTOR_LIMIT",
|
|
1006
|
+
path,
|
|
1007
|
+
f"a range descriptor holds at most {MAX_RANGE_DESCRIPTOR_BYTES} bytes",
|
|
1008
|
+
)
|
|
1009
|
+
if not _is_digest(value["manifest_sha256"]) or not _is_digest(value["facts_sha256"]):
|
|
1010
|
+
raise _refuse("PACKED_HEAD_DESCRIPTORS", path, "a range descriptor binds exact digests")
|
|
1011
|
+
return PackedRangeDescriptor(**value)
|
|
1012
|
+
|
|
1013
|
+
|
|
1014
|
+
def range_binding_sha256(
|
|
1015
|
+
*,
|
|
1016
|
+
range_index: int,
|
|
1017
|
+
first_key: str,
|
|
1018
|
+
last_key: str,
|
|
1019
|
+
entry_count: int,
|
|
1020
|
+
facts_sha256: str,
|
|
1021
|
+
) -> str:
|
|
1022
|
+
"""The digest every member of one range carries: its interval, size, and exact facts order."""
|
|
1023
|
+
|
|
1024
|
+
return canonical_sha256(
|
|
1025
|
+
{
|
|
1026
|
+
"entry_count": entry_count,
|
|
1027
|
+
"facts_sha256": facts_sha256,
|
|
1028
|
+
"first_key": first_key,
|
|
1029
|
+
"last_key": last_key,
|
|
1030
|
+
"range_index": range_index,
|
|
1031
|
+
}
|
|
1032
|
+
)
|
|
1033
|
+
|
|
1034
|
+
|
|
1035
|
+
# ------------------------------------------------------------------------------------------
|
|
1036
|
+
# The one encoding seam a packed publisher may reach
|
|
1037
|
+
# ------------------------------------------------------------------------------------------
|
|
1038
|
+
|
|
1039
|
+
|
|
1040
|
+
def encode_layer_batch(
|
|
1041
|
+
backend: EmbeddingBackend,
|
|
1042
|
+
texts: Sequence[str],
|
|
1043
|
+
*,
|
|
1044
|
+
entry_count: int,
|
|
1045
|
+
dimensions: int,
|
|
1046
|
+
) -> tuple[tuple[tuple[int, ...], ...], BatchEncodeResult]:
|
|
1047
|
+
"""Encode one range/layer/backend's fresh texts through the bounded seam, and verify the result.
|
|
1048
|
+
|
|
1049
|
+
This is the only place in the packed pair that reaches an encoder, and it reaches exactly one
|
|
1050
|
+
entry point. The counter record comes back untouched: the publisher may sum ordered records but
|
|
1051
|
+
may never author one, so nothing here rewrites a field of it.
|
|
1052
|
+
"""
|
|
1053
|
+
|
|
1054
|
+
result = backend.encode_many(tuple(texts))
|
|
1055
|
+
if not isinstance(result, BatchEncodeResult):
|
|
1056
|
+
raise _refuse(
|
|
1057
|
+
"PACKED_ENCODE_RESULT",
|
|
1058
|
+
"packed.encode.result",
|
|
1059
|
+
"the bounded seam must return its own counter record",
|
|
1060
|
+
)
|
|
1061
|
+
if len(result.vectors) != entry_count or result.scalar_invocations != entry_count:
|
|
1062
|
+
raise _refuse(
|
|
1063
|
+
"PACKED_ENCODE_RESULT",
|
|
1064
|
+
"packed.encode.result",
|
|
1065
|
+
"the seam returned another count than the items this range/layer asked for",
|
|
1066
|
+
)
|
|
1067
|
+
for ordinal, vector in enumerate(result.vectors):
|
|
1068
|
+
if len(vector) != dimensions or any(
|
|
1069
|
+
type(value) is not int or not MIN_INT32 <= value <= MAX_INT32 for value in vector
|
|
1070
|
+
):
|
|
1071
|
+
raise _refuse(
|
|
1072
|
+
"PACKED_ENCODE_RESULT",
|
|
1073
|
+
f"packed.encode.vectors[{ordinal}]",
|
|
1074
|
+
"the seam returned a vector outside the backend's exact integer coordinate",
|
|
1075
|
+
)
|
|
1076
|
+
return result.vectors, result
|
|
1077
|
+
|
|
1078
|
+
|
|
1079
|
+
# ------------------------------------------------------------------------------------------
|
|
1080
|
+
# Building one range
|
|
1081
|
+
# ------------------------------------------------------------------------------------------
|
|
1082
|
+
|
|
1083
|
+
|
|
1084
|
+
@dataclass(frozen=True)
|
|
1085
|
+
class PackedRangeInput:
|
|
1086
|
+
"""One range's ordered publishable entries, exactly as classification produced them."""
|
|
1087
|
+
|
|
1088
|
+
range_index: int
|
|
1089
|
+
entries: tuple[Mapping[str, Any], ...]
|
|
1090
|
+
|
|
1091
|
+
def __post_init__(self) -> None:
|
|
1092
|
+
if type(self.range_index) is not int or not 0 <= self.range_index < MAX_RANGE_DESCRIPTORS:
|
|
1093
|
+
raise _refuse(
|
|
1094
|
+
"PACKED_RANGE_ORDER",
|
|
1095
|
+
"packed.range.range_index",
|
|
1096
|
+
f"a generation holds at most {MAX_RANGE_DESCRIPTORS} ranges",
|
|
1097
|
+
)
|
|
1098
|
+
_check_entry_count(len(self.entries), "packed.range.entries")
|
|
1099
|
+
previous: bytes | None = None
|
|
1100
|
+
for position, entry in enumerate(self.entries):
|
|
1101
|
+
path = f"packed.range.entries[{position}]"
|
|
1102
|
+
if not isinstance(entry, Mapping) or not {
|
|
1103
|
+
"key",
|
|
1104
|
+
"entry_id",
|
|
1105
|
+
"classification",
|
|
1106
|
+
"semantic_facts_digest",
|
|
1107
|
+
"entry",
|
|
1108
|
+
"history",
|
|
1109
|
+
} <= set(entry):
|
|
1110
|
+
raise _refuse("PACKED_RANGE_ENTRY", path, "a range entry contract differs")
|
|
1111
|
+
if entry["classification"] not in PACKED_CLASSIFICATIONS:
|
|
1112
|
+
raise _refuse(
|
|
1113
|
+
"PACKED_RANGE_ENTRY",
|
|
1114
|
+
path,
|
|
1115
|
+
f"a published entry is one of {list(PACKED_CLASSIFICATIONS)}",
|
|
1116
|
+
)
|
|
1117
|
+
if not isinstance(entry["key"], str) or not entry["key"]:
|
|
1118
|
+
raise _refuse("PACKED_RANGE_ENTRY", path, "a range entry carries its exact key")
|
|
1119
|
+
key = entry["key"].encode("utf-8")
|
|
1120
|
+
if previous is not None and key <= previous:
|
|
1121
|
+
raise _refuse(
|
|
1122
|
+
"PACKED_RANGE_ORDER", path, "a range ascends by key with no repeated key"
|
|
1123
|
+
)
|
|
1124
|
+
previous = key
|
|
1125
|
+
|
|
1126
|
+
@property
|
|
1127
|
+
def keys(self) -> tuple[str, ...]:
|
|
1128
|
+
return tuple(str(entry["key"]) for entry in self.entries)
|
|
1129
|
+
|
|
1130
|
+
@property
|
|
1131
|
+
def first_key(self) -> str:
|
|
1132
|
+
return str(self.entries[0]["key"])
|
|
1133
|
+
|
|
1134
|
+
@property
|
|
1135
|
+
def last_key(self) -> str:
|
|
1136
|
+
return str(self.entries[-1]["key"])
|
|
1137
|
+
|
|
1138
|
+
|
|
1139
|
+
@dataclass(frozen=True)
|
|
1140
|
+
class BuiltPackedMember:
|
|
1141
|
+
"""One member's exact bytes beside the descriptor that binds them."""
|
|
1142
|
+
|
|
1143
|
+
raw: bytes
|
|
1144
|
+
descriptor: PackedMemberDescriptor
|
|
1145
|
+
|
|
1146
|
+
|
|
1147
|
+
@dataclass(frozen=True)
|
|
1148
|
+
class BuiltPackedRange:
|
|
1149
|
+
"""One complete range: every member's bytes, its manifest, and its head descriptor."""
|
|
1150
|
+
|
|
1151
|
+
range_index: int
|
|
1152
|
+
first_key: str
|
|
1153
|
+
last_key: str
|
|
1154
|
+
entry_count: int
|
|
1155
|
+
facts_sha256: str
|
|
1156
|
+
history_sha256: str
|
|
1157
|
+
range_binding_sha256: str
|
|
1158
|
+
members: tuple[BuiltPackedMember, ...]
|
|
1159
|
+
manifest: dict[str, Any]
|
|
1160
|
+
manifest_raw: bytes
|
|
1161
|
+
manifest_sha256: str
|
|
1162
|
+
descriptor: PackedRangeDescriptor
|
|
1163
|
+
bounds: dict[str, tuple[LayerBounds, ...]]
|
|
1164
|
+
posting_term_count: int
|
|
1165
|
+
|
|
1166
|
+
|
|
1167
|
+
def _canonical_text(payload: Any, *, code: str, path: str, maximum: int) -> tuple[str, str]:
|
|
1168
|
+
"""Serialize one nested document to canonical text carried as one string.
|
|
1169
|
+
|
|
1170
|
+
A range holds up to a thousand v2 entries and each one is itself a deep document; nesting them
|
|
1171
|
+
as JSON objects would exceed the canonical member ceiling long before it exceeded the byte
|
|
1172
|
+
ceiling. Carrying each as its exact canonical text keeps the member's own shape flat and keeps
|
|
1173
|
+
the entry byte-for-byte reproducible.
|
|
1174
|
+
"""
|
|
1175
|
+
|
|
1176
|
+
try:
|
|
1177
|
+
raw = canonical_json_bytes(payload)
|
|
1178
|
+
except CanonicalJSONError as error:
|
|
1179
|
+
raise _refuse(code, path, "a range document is not canonically representable") from error
|
|
1180
|
+
if len(raw) > maximum:
|
|
1181
|
+
raise _refuse(code, path, "a range document exceeds its byte bound")
|
|
1182
|
+
return raw.decode("utf-8"), sha256_bytes(raw)
|
|
1183
|
+
|
|
1184
|
+
|
|
1185
|
+
def build_packed_range(
|
|
1186
|
+
range_input: PackedRangeInput,
|
|
1187
|
+
vectors: Mapping[str, Mapping[str, Sequence[Sequence[int]]]],
|
|
1188
|
+
*,
|
|
1189
|
+
backends: Sequence[PackedBackend],
|
|
1190
|
+
posting_backend_coordinate: str,
|
|
1191
|
+
max_terms_per_segment: int | None = None,
|
|
1192
|
+
) -> BuiltPackedRange:
|
|
1193
|
+
"""Assemble every member of one range from its entries and its per-backend layer vectors."""
|
|
1194
|
+
|
|
1195
|
+
if not 1 <= len(backends) <= MAX_PACKED_BACKENDS:
|
|
1196
|
+
raise _refuse(
|
|
1197
|
+
"PACKED_BACKEND",
|
|
1198
|
+
"packed.range.backends",
|
|
1199
|
+
f"a packed generation carries 1 to {MAX_PACKED_BACKENDS} backends",
|
|
1200
|
+
)
|
|
1201
|
+
coordinates = [backend.coordinate for backend in backends]
|
|
1202
|
+
if len(set(coordinates)) != len(coordinates) or coordinates != sorted(coordinates):
|
|
1203
|
+
raise _refuse(
|
|
1204
|
+
"PACKED_BACKEND",
|
|
1205
|
+
"packed.range.backends",
|
|
1206
|
+
"backends are distinct and published in ascending coordinate order",
|
|
1207
|
+
)
|
|
1208
|
+
if posting_backend_coordinate not in coordinates:
|
|
1209
|
+
raise _refuse(
|
|
1210
|
+
"PACKED_POSTING_BACKEND",
|
|
1211
|
+
"packed.range.posting_backend_coordinate",
|
|
1212
|
+
"postings are derived from one of this generation's own backends",
|
|
1213
|
+
)
|
|
1214
|
+
|
|
1215
|
+
entry_count = len(range_input.entries)
|
|
1216
|
+
first_key = range_input.first_key
|
|
1217
|
+
last_key = range_input.last_key
|
|
1218
|
+
|
|
1219
|
+
facts_items = []
|
|
1220
|
+
history_items = []
|
|
1221
|
+
for entry in range_input.entries:
|
|
1222
|
+
entry_text, entry_sha256 = _canonical_text(
|
|
1223
|
+
entry["entry"],
|
|
1224
|
+
code="PACKED_FACTS_LIMIT",
|
|
1225
|
+
path="packed.range.entries[].entry",
|
|
1226
|
+
maximum=MAX_FACTS_MEMBER_BYTES,
|
|
1227
|
+
)
|
|
1228
|
+
history_text, history_sha256 = _canonical_text(
|
|
1229
|
+
entry["history"],
|
|
1230
|
+
code="PACKED_HISTORY_LIMIT",
|
|
1231
|
+
path="packed.range.entries[].history",
|
|
1232
|
+
maximum=MAX_HISTORY_MEMBER_BYTES,
|
|
1233
|
+
)
|
|
1234
|
+
facts_items.append(
|
|
1235
|
+
{
|
|
1236
|
+
"key": entry["key"],
|
|
1237
|
+
"entry_id": entry["entry_id"],
|
|
1238
|
+
"classification": entry["classification"],
|
|
1239
|
+
"semantic_facts_digest": entry["semantic_facts_digest"],
|
|
1240
|
+
"entry_sha256": entry_sha256,
|
|
1241
|
+
"entry_json": entry_text,
|
|
1242
|
+
}
|
|
1243
|
+
)
|
|
1244
|
+
history_items.append(
|
|
1245
|
+
{
|
|
1246
|
+
"key": entry["key"],
|
|
1247
|
+
"entry_id": entry["entry_id"],
|
|
1248
|
+
"history_sha256": history_sha256,
|
|
1249
|
+
"history_json": history_text,
|
|
1250
|
+
}
|
|
1251
|
+
)
|
|
1252
|
+
|
|
1253
|
+
facts_payload = {
|
|
1254
|
+
"schema_version": PACKED_FACTS_SCHEMA,
|
|
1255
|
+
"range_index": range_input.range_index,
|
|
1256
|
+
"first_key": first_key,
|
|
1257
|
+
"last_key": last_key,
|
|
1258
|
+
"entry_count": entry_count,
|
|
1259
|
+
"entries": facts_items,
|
|
1260
|
+
}
|
|
1261
|
+
facts_raw = _member_bytes(
|
|
1262
|
+
facts_payload,
|
|
1263
|
+
code="PACKED_FACTS_LIMIT",
|
|
1264
|
+
path="packed.range.facts",
|
|
1265
|
+
maximum=MAX_FACTS_MEMBER_BYTES,
|
|
1266
|
+
)
|
|
1267
|
+
facts_sha256 = sha256_bytes(facts_raw)
|
|
1268
|
+
|
|
1269
|
+
history_payload = {
|
|
1270
|
+
"schema_version": PACKED_HISTORY_SCHEMA,
|
|
1271
|
+
"range_index": range_input.range_index,
|
|
1272
|
+
"first_key": first_key,
|
|
1273
|
+
"last_key": last_key,
|
|
1274
|
+
"entry_count": entry_count,
|
|
1275
|
+
"entries": history_items,
|
|
1276
|
+
}
|
|
1277
|
+
history_raw = _member_bytes(
|
|
1278
|
+
history_payload,
|
|
1279
|
+
code="PACKED_HISTORY_LIMIT",
|
|
1280
|
+
path="packed.range.history",
|
|
1281
|
+
maximum=MAX_HISTORY_MEMBER_BYTES,
|
|
1282
|
+
)
|
|
1283
|
+
history_sha256 = sha256_bytes(history_raw)
|
|
1284
|
+
|
|
1285
|
+
binding = range_binding_sha256(
|
|
1286
|
+
range_index=range_input.range_index,
|
|
1287
|
+
first_key=first_key,
|
|
1288
|
+
last_key=last_key,
|
|
1289
|
+
entry_count=entry_count,
|
|
1290
|
+
facts_sha256=facts_sha256,
|
|
1291
|
+
)
|
|
1292
|
+
|
|
1293
|
+
def descriptor(**changes: Any) -> PackedMemberDescriptor:
|
|
1294
|
+
return PackedMemberDescriptor(
|
|
1295
|
+
range_index=range_input.range_index,
|
|
1296
|
+
first_key=first_key,
|
|
1297
|
+
last_key=last_key,
|
|
1298
|
+
entry_count=entry_count,
|
|
1299
|
+
facts_sha256=facts_sha256,
|
|
1300
|
+
range_binding_sha256=binding,
|
|
1301
|
+
**changes,
|
|
1302
|
+
)
|
|
1303
|
+
|
|
1304
|
+
members: list[BuiltPackedMember] = [
|
|
1305
|
+
BuiltPackedMember(
|
|
1306
|
+
raw=facts_raw,
|
|
1307
|
+
descriptor=descriptor(kind="facts", sha256=facts_sha256, bytes=len(facts_raw)),
|
|
1308
|
+
),
|
|
1309
|
+
BuiltPackedMember(
|
|
1310
|
+
raw=history_raw,
|
|
1311
|
+
descriptor=descriptor(kind="history", sha256=history_sha256, bytes=len(history_raw)),
|
|
1312
|
+
),
|
|
1313
|
+
]
|
|
1314
|
+
|
|
1315
|
+
bounds_by_backend: dict[str, tuple[LayerBounds, ...]] = {}
|
|
1316
|
+
posting_term_count = 0
|
|
1317
|
+
for backend in backends:
|
|
1318
|
+
layers = vectors.get(backend.coordinate)
|
|
1319
|
+
if layers is None or set(layers) != set(EMBEDDING_LAYERS):
|
|
1320
|
+
raise _refuse(
|
|
1321
|
+
"PACKED_MEMBER_LAYER",
|
|
1322
|
+
"packed.range.vectors",
|
|
1323
|
+
f"every backend supplies exactly the four layers {list(EMBEDDING_LAYERS)}",
|
|
1324
|
+
)
|
|
1325
|
+
for layer in EMBEDDING_LAYERS:
|
|
1326
|
+
layer_vectors = layers[layer]
|
|
1327
|
+
if len(layer_vectors) != entry_count:
|
|
1328
|
+
raise _refuse(
|
|
1329
|
+
"PACKED_MEMBER_ENTRY_COUNT",
|
|
1330
|
+
"packed.range.vectors",
|
|
1331
|
+
"every layer member covers exactly this range's entries",
|
|
1332
|
+
)
|
|
1333
|
+
raw = pack_vector_member(
|
|
1334
|
+
layer_vectors,
|
|
1335
|
+
layer=layer,
|
|
1336
|
+
dimensions=backend.dimensions,
|
|
1337
|
+
range_index=range_input.range_index,
|
|
1338
|
+
binding_sha256=binding,
|
|
1339
|
+
backend_coordinate=backend.coordinate,
|
|
1340
|
+
)
|
|
1341
|
+
members.append(
|
|
1342
|
+
BuiltPackedMember(
|
|
1343
|
+
raw=raw,
|
|
1344
|
+
descriptor=descriptor(
|
|
1345
|
+
kind="vector",
|
|
1346
|
+
sha256=sha256_bytes(raw),
|
|
1347
|
+
bytes=len(raw),
|
|
1348
|
+
backend_coordinate=backend.coordinate,
|
|
1349
|
+
dimensions=backend.dimensions,
|
|
1350
|
+
layer=layer,
|
|
1351
|
+
),
|
|
1352
|
+
)
|
|
1353
|
+
)
|
|
1354
|
+
|
|
1355
|
+
bounds = compute_layer_bounds(layers, dimensions=backend.dimensions)
|
|
1356
|
+
bounds_by_backend[backend.coordinate] = bounds
|
|
1357
|
+
raw = pack_bound_segment(
|
|
1358
|
+
bounds,
|
|
1359
|
+
dimensions=backend.dimensions,
|
|
1360
|
+
entry_count=entry_count,
|
|
1361
|
+
range_index=range_input.range_index,
|
|
1362
|
+
binding_sha256=binding,
|
|
1363
|
+
backend_coordinate=backend.coordinate,
|
|
1364
|
+
)
|
|
1365
|
+
members.append(
|
|
1366
|
+
BuiltPackedMember(
|
|
1367
|
+
raw=raw,
|
|
1368
|
+
descriptor=descriptor(
|
|
1369
|
+
kind="bound",
|
|
1370
|
+
sha256=sha256_bytes(raw),
|
|
1371
|
+
bytes=len(raw),
|
|
1372
|
+
backend_coordinate=backend.coordinate,
|
|
1373
|
+
dimensions=backend.dimensions,
|
|
1374
|
+
),
|
|
1375
|
+
)
|
|
1376
|
+
)
|
|
1377
|
+
|
|
1378
|
+
if backend.coordinate == posting_backend_coordinate:
|
|
1379
|
+
terms = compute_layer_postings(layers, dimensions=backend.dimensions)
|
|
1380
|
+
posting_term_count = len(terms)
|
|
1381
|
+
for segment_index, segment in enumerate(
|
|
1382
|
+
segment_posting_terms(terms, max_terms_per_segment=max_terms_per_segment)
|
|
1383
|
+
):
|
|
1384
|
+
raw = pack_posting_segment(
|
|
1385
|
+
segment,
|
|
1386
|
+
segment_index=segment_index,
|
|
1387
|
+
dimensions=backend.dimensions,
|
|
1388
|
+
entry_count=entry_count,
|
|
1389
|
+
range_index=range_input.range_index,
|
|
1390
|
+
binding_sha256=binding,
|
|
1391
|
+
backend_coordinate=backend.coordinate,
|
|
1392
|
+
)
|
|
1393
|
+
members.append(
|
|
1394
|
+
BuiltPackedMember(
|
|
1395
|
+
raw=raw,
|
|
1396
|
+
descriptor=descriptor(
|
|
1397
|
+
kind="posting",
|
|
1398
|
+
sha256=sha256_bytes(raw),
|
|
1399
|
+
bytes=len(raw),
|
|
1400
|
+
backend_coordinate=backend.coordinate,
|
|
1401
|
+
dimensions=backend.dimensions,
|
|
1402
|
+
segment_index=segment_index,
|
|
1403
|
+
term_count=len(segment),
|
|
1404
|
+
),
|
|
1405
|
+
)
|
|
1406
|
+
)
|
|
1407
|
+
|
|
1408
|
+
if len(members) > MAX_RANGE_MEMBER_DESCRIPTORS:
|
|
1409
|
+
raise _refuse(
|
|
1410
|
+
"PACKED_RANGE_MEMBERS",
|
|
1411
|
+
"packed.range.members",
|
|
1412
|
+
f"a range manifest holds at most {MAX_RANGE_MEMBER_DESCRIPTORS} member descriptors",
|
|
1413
|
+
)
|
|
1414
|
+
for member in members:
|
|
1415
|
+
if len(canonical_json_bytes(member.descriptor.to_dict())) > MAX_MEMBER_DESCRIPTOR_BYTES:
|
|
1416
|
+
raise _refuse(
|
|
1417
|
+
"PACKED_DESCRIPTOR_LIMIT",
|
|
1418
|
+
"packed.range.members[]",
|
|
1419
|
+
f"a member descriptor holds at most {MAX_MEMBER_DESCRIPTOR_BYTES} bytes",
|
|
1420
|
+
)
|
|
1421
|
+
|
|
1422
|
+
body = {
|
|
1423
|
+
"schema_version": PACKED_RANGE_MANIFEST_SCHEMA,
|
|
1424
|
+
"range_index": range_input.range_index,
|
|
1425
|
+
"first_key": first_key,
|
|
1426
|
+
"last_key": last_key,
|
|
1427
|
+
"entry_count": entry_count,
|
|
1428
|
+
"facts_sha256": facts_sha256,
|
|
1429
|
+
"history_sha256": history_sha256,
|
|
1430
|
+
"range_binding_sha256": binding,
|
|
1431
|
+
"backends": list(coordinates),
|
|
1432
|
+
"posting_backend_coordinate": posting_backend_coordinate,
|
|
1433
|
+
"members": [member.descriptor.to_dict() for member in members],
|
|
1434
|
+
}
|
|
1435
|
+
manifest = {**body, "root_sha256": canonical_sha256(body)}
|
|
1436
|
+
manifest_raw = _member_bytes(
|
|
1437
|
+
manifest,
|
|
1438
|
+
code="PACKED_RANGE_MANIFEST",
|
|
1439
|
+
path="packed.range.manifest",
|
|
1440
|
+
maximum=MAX_RANGE_MANIFEST_BYTES,
|
|
1441
|
+
)
|
|
1442
|
+
return BuiltPackedRange(
|
|
1443
|
+
range_index=range_input.range_index,
|
|
1444
|
+
first_key=first_key,
|
|
1445
|
+
last_key=last_key,
|
|
1446
|
+
entry_count=entry_count,
|
|
1447
|
+
facts_sha256=facts_sha256,
|
|
1448
|
+
history_sha256=history_sha256,
|
|
1449
|
+
range_binding_sha256=binding,
|
|
1450
|
+
members=tuple(members),
|
|
1451
|
+
manifest=manifest,
|
|
1452
|
+
manifest_raw=manifest_raw,
|
|
1453
|
+
manifest_sha256=sha256_bytes(manifest_raw),
|
|
1454
|
+
descriptor=PackedRangeDescriptor(
|
|
1455
|
+
range_index=range_input.range_index,
|
|
1456
|
+
first_key=first_key,
|
|
1457
|
+
last_key=last_key,
|
|
1458
|
+
entry_count=entry_count,
|
|
1459
|
+
facts_sha256=facts_sha256,
|
|
1460
|
+
range_binding_sha256=binding,
|
|
1461
|
+
manifest_sha256=sha256_bytes(manifest_raw),
|
|
1462
|
+
manifest_bytes=len(manifest_raw),
|
|
1463
|
+
member_count=len(members),
|
|
1464
|
+
),
|
|
1465
|
+
bounds=bounds_by_backend,
|
|
1466
|
+
posting_term_count=posting_term_count,
|
|
1467
|
+
)
|
|
1468
|
+
|
|
1469
|
+
|
|
1470
|
+
def _member_bytes(payload: Any, *, code: str, path: str, maximum: int) -> bytes:
|
|
1471
|
+
try:
|
|
1472
|
+
raw = canonical_json_bytes(payload)
|
|
1473
|
+
except CanonicalJSONError as error:
|
|
1474
|
+
raise _refuse(code, path, "a packed document is not canonically representable") from error
|
|
1475
|
+
if len(raw) > maximum:
|
|
1476
|
+
raise _refuse(
|
|
1477
|
+
code,
|
|
1478
|
+
path,
|
|
1479
|
+
f"the member holds at most {maximum} bytes; publication refuses rather than splitting",
|
|
1480
|
+
)
|
|
1481
|
+
return raw
|
|
1482
|
+
|
|
1483
|
+
|
|
1484
|
+
# ------------------------------------------------------------------------------------------
|
|
1485
|
+
# The outer head
|
|
1486
|
+
# ------------------------------------------------------------------------------------------
|
|
1487
|
+
|
|
1488
|
+
|
|
1489
|
+
COUNTER_MEMBERS = (
|
|
1490
|
+
"batch_invocations",
|
|
1491
|
+
"encoded_layer_items",
|
|
1492
|
+
"backend_scalar_invocations",
|
|
1493
|
+
"encoded_utf8_bytes",
|
|
1494
|
+
"reused_layer_items",
|
|
1495
|
+
)
|
|
1496
|
+
|
|
1497
|
+
|
|
1498
|
+
def ranges_chain_sha256(descriptors: Sequence[PackedRangeDescriptor]) -> str:
|
|
1499
|
+
"""One chained digest over every range descriptor, in published order."""
|
|
1500
|
+
|
|
1501
|
+
chain: str | None = None
|
|
1502
|
+
for descriptor in descriptors:
|
|
1503
|
+
chain = canonical_sha256({"previous_sha256": chain, "range": descriptor.to_dict()})
|
|
1504
|
+
return chain or canonical_sha256({"previous_sha256": None, "range": None})
|
|
1505
|
+
|
|
1506
|
+
|
|
1507
|
+
def build_packed_head(
|
|
1508
|
+
*,
|
|
1509
|
+
provider_id: str,
|
|
1510
|
+
generation_sha256: str,
|
|
1511
|
+
predecessor_head_sha256: str | None,
|
|
1512
|
+
backends: Sequence[PackedBackend],
|
|
1513
|
+
posting_backend_coordinate: str,
|
|
1514
|
+
ranges: Sequence[PackedRangeDescriptor],
|
|
1515
|
+
coverage: Mapping[str, Any],
|
|
1516
|
+
counters: Mapping[str, int],
|
|
1517
|
+
invocations: Sequence[Mapping[str, Any]],
|
|
1518
|
+
terminal_sha256: str,
|
|
1519
|
+
) -> dict[str, Any]:
|
|
1520
|
+
"""Assemble the outer head. Every law it carries is enforced by the reader, not by this call."""
|
|
1521
|
+
|
|
1522
|
+
if len(ranges) > MAX_RANGE_DESCRIPTORS:
|
|
1523
|
+
raise _refuse(
|
|
1524
|
+
"PACKED_HEAD_DESCRIPTORS",
|
|
1525
|
+
"packed.head.ranges",
|
|
1526
|
+
f"an outer head holds at most {MAX_RANGE_DESCRIPTORS} range descriptors",
|
|
1527
|
+
)
|
|
1528
|
+
if not ranges:
|
|
1529
|
+
raise _refuse(
|
|
1530
|
+
"PACKED_HEAD_DESCRIPTORS", "packed.head.ranges", "a published generation has a range"
|
|
1531
|
+
)
|
|
1532
|
+
if set(counters) != set(COUNTER_MEMBERS):
|
|
1533
|
+
raise _refuse(
|
|
1534
|
+
"PACKED_COUNTER_CONTRACT",
|
|
1535
|
+
"packed.head.counters",
|
|
1536
|
+
f"the head carries exactly {list(COUNTER_MEMBERS)}",
|
|
1537
|
+
)
|
|
1538
|
+
for name in COUNTER_MEMBERS:
|
|
1539
|
+
if type(counters[name]) is not int or counters[name] < 0:
|
|
1540
|
+
raise _refuse(
|
|
1541
|
+
"PACKED_COUNTER_CONTRACT",
|
|
1542
|
+
f"packed.head.counters.{name}",
|
|
1543
|
+
"a work counter is a non-negative integer",
|
|
1544
|
+
)
|
|
1545
|
+
for descriptor in ranges:
|
|
1546
|
+
if len(canonical_json_bytes(descriptor.to_dict())) > MAX_RANGE_DESCRIPTOR_BYTES:
|
|
1547
|
+
raise _refuse(
|
|
1548
|
+
"PACKED_DESCRIPTOR_LIMIT",
|
|
1549
|
+
"packed.head.ranges[]",
|
|
1550
|
+
f"a range descriptor holds at most {MAX_RANGE_DESCRIPTOR_BYTES} bytes",
|
|
1551
|
+
)
|
|
1552
|
+
if not _is_digest(terminal_sha256):
|
|
1553
|
+
raise _refuse(
|
|
1554
|
+
"PACKED_HEAD_DIGEST",
|
|
1555
|
+
"packed.head.terminal_sha256",
|
|
1556
|
+
"the terminal publication record is named by one lowercase SHA-256",
|
|
1557
|
+
)
|
|
1558
|
+
if not coverage_is_valid(coverage):
|
|
1559
|
+
raise _refuse(
|
|
1560
|
+
"PACKED_HEAD_COVERAGE",
|
|
1561
|
+
"packed.head.coverage",
|
|
1562
|
+
"a published generation states what it covers and what it drew from",
|
|
1563
|
+
)
|
|
1564
|
+
body = {
|
|
1565
|
+
"schema_version": PACKED_HEAD_SCHEMA,
|
|
1566
|
+
"provider_id": provider_id,
|
|
1567
|
+
"generation_sha256": generation_sha256,
|
|
1568
|
+
"predecessor_head_sha256": predecessor_head_sha256,
|
|
1569
|
+
"backends": [backend.to_dict() for backend in backends],
|
|
1570
|
+
"posting_backend_coordinate": posting_backend_coordinate,
|
|
1571
|
+
"range_count": len(ranges),
|
|
1572
|
+
"entry_count": sum(descriptor.entry_count for descriptor in ranges),
|
|
1573
|
+
# What this generation covers travels with the head rather than only with the receipts
|
|
1574
|
+
# behind it, because the head is what a release archive is built from and what an
|
|
1575
|
+
# installed catalogue reads back. A consumer that holds only the published bytes can
|
|
1576
|
+
# still tell a bounded catalogue from an exhaustive one.
|
|
1577
|
+
"coverage": dict(coverage),
|
|
1578
|
+
"ranges": [descriptor.to_dict() for descriptor in ranges],
|
|
1579
|
+
"ranges_chain_sha256": ranges_chain_sha256(ranges),
|
|
1580
|
+
"counters": {name: counters[name] for name in COUNTER_MEMBERS},
|
|
1581
|
+
"invocations": [dict(record) for record in invocations],
|
|
1582
|
+
"terminal_sha256": terminal_sha256,
|
|
1583
|
+
}
|
|
1584
|
+
head = {**body, "root_sha256": canonical_sha256(body)}
|
|
1585
|
+
if len(canonical_json_bytes(head)) > MAX_PACKED_HEAD_BYTES:
|
|
1586
|
+
raise _refuse(
|
|
1587
|
+
"PACKED_HEAD_LIMIT",
|
|
1588
|
+
"packed.head",
|
|
1589
|
+
f"an outer head holds at most {MAX_PACKED_HEAD_BYTES} bytes",
|
|
1590
|
+
)
|
|
1591
|
+
return head
|
|
1592
|
+
|
|
1593
|
+
|
|
1594
|
+
# ------------------------------------------------------------------------------------------
|
|
1595
|
+
# The durable output root
|
|
1596
|
+
# ------------------------------------------------------------------------------------------
|
|
1597
|
+
|
|
1598
|
+
|
|
1599
|
+
class PackedDirectory:
|
|
1600
|
+
"""One locally confined packed root whose members install atomically and read back exactly."""
|
|
1601
|
+
|
|
1602
|
+
def __init__(self, root: Path, *, root_descriptor: int | None = None) -> None:
|
|
1603
|
+
self.root = Path(root)
|
|
1604
|
+
self._retained_root_descriptor = root_descriptor
|
|
1605
|
+
self.bytes_written = 0
|
|
1606
|
+
self._root_fd = -1
|
|
1607
|
+
self._members_fd = -1
|
|
1608
|
+
self._owned: list[int] = []
|
|
1609
|
+
|
|
1610
|
+
@contextlib.contextmanager
|
|
1611
|
+
def opened(self) -> Iterator[PackedDirectory]:
|
|
1612
|
+
try:
|
|
1613
|
+
if self._retained_root_descriptor is None:
|
|
1614
|
+
self.root.mkdir(parents=True, exist_ok=True, mode=0o700)
|
|
1615
|
+
self._root_fd = self._open_directory(self.root)
|
|
1616
|
+
else:
|
|
1617
|
+
self._root_fd = os.dup(self._retained_root_descriptor)
|
|
1618
|
+
self._owned.append(self._root_fd)
|
|
1619
|
+
self._require_private_directory(self._root_fd)
|
|
1620
|
+
try:
|
|
1621
|
+
os.mkdir(PACKED_MEMBERS_DIRNAME, mode=0o700, dir_fd=self._root_fd)
|
|
1622
|
+
except FileExistsError:
|
|
1623
|
+
pass
|
|
1624
|
+
self._members_fd = self._open_directory_at(self._root_fd, PACKED_MEMBERS_DIRNAME)
|
|
1625
|
+
except OSError as error:
|
|
1626
|
+
self._release()
|
|
1627
|
+
raise _refuse("PACKED_OUTPUT_IO", "packed.output", "output root is unusable") from error
|
|
1628
|
+
except SourceContractError:
|
|
1629
|
+
self._release()
|
|
1630
|
+
raise
|
|
1631
|
+
try:
|
|
1632
|
+
yield self
|
|
1633
|
+
finally:
|
|
1634
|
+
self._release()
|
|
1635
|
+
|
|
1636
|
+
def _release(self) -> None:
|
|
1637
|
+
for descriptor in reversed(self._owned):
|
|
1638
|
+
os.close(descriptor)
|
|
1639
|
+
self._owned = []
|
|
1640
|
+
self._root_fd = -1
|
|
1641
|
+
self._members_fd = -1
|
|
1642
|
+
|
|
1643
|
+
def _open_directory(self, path: Path) -> int:
|
|
1644
|
+
descriptor = os.open(path, _OPEN_DIRECTORY)
|
|
1645
|
+
self._owned.append(descriptor)
|
|
1646
|
+
self._require_private_directory(descriptor)
|
|
1647
|
+
return descriptor
|
|
1648
|
+
|
|
1649
|
+
def _open_directory_at(self, parent_descriptor: int, name: str) -> int:
|
|
1650
|
+
descriptor = os.open(name, _OPEN_DIRECTORY, dir_fd=parent_descriptor)
|
|
1651
|
+
self._owned.append(descriptor)
|
|
1652
|
+
self._require_private_directory(descriptor)
|
|
1653
|
+
return descriptor
|
|
1654
|
+
|
|
1655
|
+
def _require_private_directory(self, descriptor: int) -> None:
|
|
1656
|
+
info = os.fstat(descriptor)
|
|
1657
|
+
if not stat.S_ISDIR(info.st_mode) or stat.S_IMODE(info.st_mode) & 0o077:
|
|
1658
|
+
raise _refuse(
|
|
1659
|
+
"PACKED_OUTPUT_PATH",
|
|
1660
|
+
"packed.output",
|
|
1661
|
+
"output components must be private directories",
|
|
1662
|
+
)
|
|
1663
|
+
|
|
1664
|
+
@property
|
|
1665
|
+
def root_descriptor(self) -> int:
|
|
1666
|
+
"""The exact open root used by this directory context."""
|
|
1667
|
+
|
|
1668
|
+
if self._root_fd < 0:
|
|
1669
|
+
raise RuntimeError("packed directory is not open")
|
|
1670
|
+
return self._root_fd
|
|
1671
|
+
|
|
1672
|
+
def read(self, name: str, *, maximum: int) -> bytes | None:
|
|
1673
|
+
try:
|
|
1674
|
+
return _read_regular_at(self._root_fd, name, maximum=maximum)
|
|
1675
|
+
except FileNotFoundError:
|
|
1676
|
+
return None
|
|
1677
|
+
|
|
1678
|
+
def read_member(self, sha256: str, *, kind: str, maximum: int) -> bytes | None:
|
|
1679
|
+
try:
|
|
1680
|
+
return _read_regular_at(
|
|
1681
|
+
self._members_fd, member_filename(kind, sha256), maximum=maximum
|
|
1682
|
+
)
|
|
1683
|
+
except FileNotFoundError:
|
|
1684
|
+
return None
|
|
1685
|
+
|
|
1686
|
+
def has_member(self, sha256: str, *, kind: str) -> bool:
|
|
1687
|
+
try:
|
|
1688
|
+
os.stat(member_filename(kind, sha256), dir_fd=self._members_fd, follow_symlinks=False)
|
|
1689
|
+
except FileNotFoundError:
|
|
1690
|
+
return False
|
|
1691
|
+
except OSError as error:
|
|
1692
|
+
raise _refuse(
|
|
1693
|
+
"PACKED_OUTPUT_IO", "packed.output.member", "member lookup failed"
|
|
1694
|
+
) from error
|
|
1695
|
+
return True
|
|
1696
|
+
|
|
1697
|
+
def write_member_bytes(self, raw: bytes, *, kind: str) -> tuple[str, int]:
|
|
1698
|
+
digest = sha256_bytes(raw)
|
|
1699
|
+
name = member_filename(kind, digest)
|
|
1700
|
+
try:
|
|
1701
|
+
existing = _read_regular_at(self._members_fd, name, maximum=len(raw))
|
|
1702
|
+
except FileNotFoundError:
|
|
1703
|
+
existing = None
|
|
1704
|
+
except OSError as error:
|
|
1705
|
+
raise _refuse(
|
|
1706
|
+
"PACKED_OUTPUT_IO", "packed.output.member", "member lookup failed"
|
|
1707
|
+
) from error
|
|
1708
|
+
if existing is not None:
|
|
1709
|
+
if existing != raw:
|
|
1710
|
+
raise _refuse(
|
|
1711
|
+
"PACKED_OUTPUT_READBACK", "packed.output.member", "digest path bytes differ"
|
|
1712
|
+
)
|
|
1713
|
+
return digest, len(raw)
|
|
1714
|
+
temporary = f".publish.{os.getpid()}.{secrets.token_hex(16)}.tmp"
|
|
1715
|
+
try:
|
|
1716
|
+
_write_regular_at(self._members_fd, temporary, raw)
|
|
1717
|
+
with contextlib.suppress(FileExistsError):
|
|
1718
|
+
os.link(
|
|
1719
|
+
temporary,
|
|
1720
|
+
name,
|
|
1721
|
+
src_dir_fd=self._members_fd,
|
|
1722
|
+
dst_dir_fd=self._members_fd,
|
|
1723
|
+
follow_symlinks=False,
|
|
1724
|
+
)
|
|
1725
|
+
except OSError as error:
|
|
1726
|
+
raise _refuse(
|
|
1727
|
+
"PACKED_OUTPUT_IO", "packed.output.member", "member publication failed"
|
|
1728
|
+
) from error
|
|
1729
|
+
finally:
|
|
1730
|
+
with contextlib.suppress(FileNotFoundError):
|
|
1731
|
+
os.unlink(temporary, dir_fd=self._members_fd)
|
|
1732
|
+
os.fsync(self._members_fd)
|
|
1733
|
+
if _read_regular_at(self._members_fd, name, maximum=len(raw)) != raw:
|
|
1734
|
+
raise _refuse(
|
|
1735
|
+
"PACKED_OUTPUT_READBACK", "packed.output.member", "member readback differs"
|
|
1736
|
+
)
|
|
1737
|
+
self.bytes_written += len(raw)
|
|
1738
|
+
return digest, len(raw)
|
|
1739
|
+
|
|
1740
|
+
def publish(self, name: str, payload: Mapping[str, Any], *, maximum: int) -> str:
|
|
1741
|
+
raw = canonical_json_bytes(dict(payload))
|
|
1742
|
+
if len(raw) > maximum:
|
|
1743
|
+
raise _refuse(
|
|
1744
|
+
"PACKED_OUTPUT_LIMIT", f"packed.output.{name}", "document exceeds its byte bound"
|
|
1745
|
+
)
|
|
1746
|
+
temporary = f".{name}.{os.getpid()}.{secrets.token_hex(8)}.tmp"
|
|
1747
|
+
try:
|
|
1748
|
+
_write_regular_at(self._root_fd, temporary, raw)
|
|
1749
|
+
os.replace(temporary, name, src_dir_fd=self._root_fd, dst_dir_fd=self._root_fd)
|
|
1750
|
+
os.fsync(self._root_fd)
|
|
1751
|
+
except OSError as error:
|
|
1752
|
+
raise _refuse(
|
|
1753
|
+
"PACKED_OUTPUT_IO", f"packed.output.{name}", "document publication failed"
|
|
1754
|
+
) from error
|
|
1755
|
+
finally:
|
|
1756
|
+
with contextlib.suppress(FileNotFoundError):
|
|
1757
|
+
os.unlink(temporary, dir_fd=self._root_fd)
|
|
1758
|
+
if _read_regular_at(self._root_fd, name, maximum=maximum) != raw:
|
|
1759
|
+
raise _refuse(
|
|
1760
|
+
"PACKED_OUTPUT_READBACK", f"packed.output.{name}", "document readback differs"
|
|
1761
|
+
)
|
|
1762
|
+
self.bytes_written += len(raw)
|
|
1763
|
+
return sha256_bytes(raw)
|
|
1764
|
+
|
|
1765
|
+
|
|
1766
|
+
def _read_regular_at(parent_fd: int, name: str, *, maximum: int) -> bytes:
|
|
1767
|
+
try:
|
|
1768
|
+
raw = read_bounded_at(parent_fd, name, maximum=maximum)
|
|
1769
|
+
except BoundedReadFailure as error:
|
|
1770
|
+
raise _refuse(
|
|
1771
|
+
"PACKED_OUTPUT_PATH",
|
|
1772
|
+
"packed.output",
|
|
1773
|
+
"members must be stable bounded single-link regular files",
|
|
1774
|
+
) from error
|
|
1775
|
+
assert raw is not None
|
|
1776
|
+
return raw
|
|
1777
|
+
|
|
1778
|
+
|
|
1779
|
+
def _write_regular_at(parent_fd: int, name: str, raw: bytes) -> None:
|
|
1780
|
+
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0)
|
|
1781
|
+
descriptor = os.open(name, flags, 0o600, dir_fd=parent_fd)
|
|
1782
|
+
try:
|
|
1783
|
+
written = 0
|
|
1784
|
+
while written < len(raw):
|
|
1785
|
+
count = os.write(descriptor, raw[written:])
|
|
1786
|
+
if count < 1:
|
|
1787
|
+
raise _refuse("PACKED_OUTPUT_IO", "packed.output", "write made no progress")
|
|
1788
|
+
written += count
|
|
1789
|
+
os.fsync(descriptor)
|
|
1790
|
+
finally:
|
|
1791
|
+
os.close(descriptor)
|
|
1792
|
+
|
|
1793
|
+
|
|
1794
|
+
# ------------------------------------------------------------------------------------------
|
|
1795
|
+
# The independent reader
|
|
1796
|
+
# ------------------------------------------------------------------------------------------
|
|
1797
|
+
|
|
1798
|
+
|
|
1799
|
+
@dataclass(frozen=True)
|
|
1800
|
+
class VerifiedPackedRange:
|
|
1801
|
+
"""What one range verified to, recomputed from its own bytes rather than from a summary."""
|
|
1802
|
+
|
|
1803
|
+
range_index: int
|
|
1804
|
+
first_key: str
|
|
1805
|
+
last_key: str
|
|
1806
|
+
entry_count: int
|
|
1807
|
+
facts_sha256: str
|
|
1808
|
+
history_sha256: str
|
|
1809
|
+
range_binding_sha256: str
|
|
1810
|
+
manifest_bytes: int
|
|
1811
|
+
member_count: int
|
|
1812
|
+
member_sha256s: tuple[str, ...]
|
|
1813
|
+
vector_payload_sha256s: tuple[str, ...]
|
|
1814
|
+
posting_sha256s: tuple[str, ...]
|
|
1815
|
+
bound_sha256s: tuple[str, ...]
|
|
1816
|
+
posting_term_count: int
|
|
1817
|
+
entry_ids: tuple[str, ...]
|
|
1818
|
+
classification_counts: tuple[tuple[str, int], ...]
|
|
1819
|
+
|
|
1820
|
+
|
|
1821
|
+
@dataclass(frozen=True)
|
|
1822
|
+
class VerifiedPackedGeneration:
|
|
1823
|
+
"""One complete packed generation's verified shape."""
|
|
1824
|
+
|
|
1825
|
+
head_sha256: str
|
|
1826
|
+
head: dict[str, Any]
|
|
1827
|
+
range_count: int
|
|
1828
|
+
entry_count: int
|
|
1829
|
+
member_count: int
|
|
1830
|
+
posting_term_count: int
|
|
1831
|
+
backends: tuple[str, ...]
|
|
1832
|
+
classification_counts: tuple[tuple[str, int], ...]
|
|
1833
|
+
fresh_range_count: int
|
|
1834
|
+
vector_payload_sha256s: tuple[str, ...]
|
|
1835
|
+
posting_member_sha256s: tuple[str, ...]
|
|
1836
|
+
bound_member_sha256s: tuple[str, ...]
|
|
1837
|
+
|
|
1838
|
+
|
|
1839
|
+
def verify_packed_range(
|
|
1840
|
+
directory: PackedDirectory,
|
|
1841
|
+
manifest_sha256: str,
|
|
1842
|
+
*,
|
|
1843
|
+
backends: Sequence[PackedBackend],
|
|
1844
|
+
posting_backend_coordinate: str,
|
|
1845
|
+
) -> VerifiedPackedRange:
|
|
1846
|
+
"""Scan one emitted range once, recomputing every accelerator from the exact pack bytes."""
|
|
1847
|
+
|
|
1848
|
+
raw = directory.read_member(manifest_sha256, kind="range", maximum=MAX_RANGE_MANIFEST_BYTES)
|
|
1849
|
+
if raw is None or sha256_bytes(raw) != manifest_sha256:
|
|
1850
|
+
raise _refuse(
|
|
1851
|
+
"PACKED_RANGE_DIGEST",
|
|
1852
|
+
"packed.range.manifest",
|
|
1853
|
+
"the installed range manifest is not the manifest the head binds",
|
|
1854
|
+
)
|
|
1855
|
+
try:
|
|
1856
|
+
manifest = parse_canonical_json(raw)
|
|
1857
|
+
except CanonicalJSONError as error:
|
|
1858
|
+
raise _refuse(
|
|
1859
|
+
"PACKED_RANGE_MANIFEST", "packed.range.manifest", "manifest is not canonical"
|
|
1860
|
+
) from error
|
|
1861
|
+
expected_members = {
|
|
1862
|
+
"schema_version",
|
|
1863
|
+
"range_index",
|
|
1864
|
+
"first_key",
|
|
1865
|
+
"last_key",
|
|
1866
|
+
"entry_count",
|
|
1867
|
+
"facts_sha256",
|
|
1868
|
+
"history_sha256",
|
|
1869
|
+
"range_binding_sha256",
|
|
1870
|
+
"backends",
|
|
1871
|
+
"posting_backend_coordinate",
|
|
1872
|
+
"members",
|
|
1873
|
+
"root_sha256",
|
|
1874
|
+
}
|
|
1875
|
+
if (
|
|
1876
|
+
not isinstance(manifest, dict)
|
|
1877
|
+
or set(manifest) != expected_members
|
|
1878
|
+
or manifest["schema_version"] != PACKED_RANGE_MANIFEST_SCHEMA
|
|
1879
|
+
or not isinstance(manifest["members"], list)
|
|
1880
|
+
):
|
|
1881
|
+
raise _refuse(
|
|
1882
|
+
"PACKED_RANGE_MANIFEST", "packed.range.manifest", "range manifest contract differs"
|
|
1883
|
+
)
|
|
1884
|
+
body = {key: value for key, value in manifest.items() if key != "root_sha256"}
|
|
1885
|
+
if canonical_sha256(body) != manifest["root_sha256"]:
|
|
1886
|
+
raise _refuse(
|
|
1887
|
+
"PACKED_RANGE_MANIFEST", "packed.range.manifest", "manifest root digest differs"
|
|
1888
|
+
)
|
|
1889
|
+
if not 1 <= len(manifest["members"]) <= MAX_RANGE_MEMBER_DESCRIPTORS:
|
|
1890
|
+
raise _refuse(
|
|
1891
|
+
"PACKED_RANGE_MEMBERS",
|
|
1892
|
+
"packed.range.members",
|
|
1893
|
+
f"a range manifest holds 1 to {MAX_RANGE_MEMBER_DESCRIPTORS} member descriptors",
|
|
1894
|
+
)
|
|
1895
|
+
if manifest["posting_backend_coordinate"] != posting_backend_coordinate or manifest[
|
|
1896
|
+
"backends"
|
|
1897
|
+
] != [backend.coordinate for backend in backends]:
|
|
1898
|
+
raise _refuse(
|
|
1899
|
+
"PACKED_RANGE_MANIFEST",
|
|
1900
|
+
"packed.range.backends",
|
|
1901
|
+
"a range names exactly the generation's backends and posting backend",
|
|
1902
|
+
)
|
|
1903
|
+
|
|
1904
|
+
descriptors = [
|
|
1905
|
+
member_descriptor_from_dict(value, path=f"packed.range.members[{index}]")
|
|
1906
|
+
for index, value in enumerate(manifest["members"])
|
|
1907
|
+
]
|
|
1908
|
+
binding = range_binding_sha256(
|
|
1909
|
+
range_index=manifest["range_index"],
|
|
1910
|
+
first_key=manifest["first_key"],
|
|
1911
|
+
last_key=manifest["last_key"],
|
|
1912
|
+
entry_count=manifest["entry_count"],
|
|
1913
|
+
facts_sha256=manifest["facts_sha256"],
|
|
1914
|
+
)
|
|
1915
|
+
if binding != manifest["range_binding_sha256"]:
|
|
1916
|
+
raise _refuse(
|
|
1917
|
+
"PACKED_RANGE_MANIFEST",
|
|
1918
|
+
"packed.range.range_binding_sha256",
|
|
1919
|
+
"the range binding does not reproduce from this range's own interval and facts order",
|
|
1920
|
+
)
|
|
1921
|
+
for descriptor in descriptors:
|
|
1922
|
+
if (
|
|
1923
|
+
descriptor.range_index != manifest["range_index"]
|
|
1924
|
+
or descriptor.first_key != manifest["first_key"]
|
|
1925
|
+
or descriptor.last_key != manifest["last_key"]
|
|
1926
|
+
or descriptor.entry_count != manifest["entry_count"]
|
|
1927
|
+
or descriptor.facts_sha256 != manifest["facts_sha256"]
|
|
1928
|
+
or descriptor.range_binding_sha256 != binding
|
|
1929
|
+
):
|
|
1930
|
+
raise _refuse(
|
|
1931
|
+
"PACKED_MEMBER_BINDING",
|
|
1932
|
+
"packed.range.members[]",
|
|
1933
|
+
"a member descriptor is bound to another range",
|
|
1934
|
+
)
|
|
1935
|
+
|
|
1936
|
+
facts = _verified_member(directory, _one(descriptors, "facts"), maximum=MAX_FACTS_MEMBER_BYTES)
|
|
1937
|
+
entry_ids, classification_counts = _verify_facts_member(facts, manifest)
|
|
1938
|
+
_verified_member(directory, _one(descriptors, "history"), maximum=MAX_HISTORY_MEMBER_BYTES)
|
|
1939
|
+
|
|
1940
|
+
vectors_by_backend: dict[str, dict[str, tuple[tuple[int, ...], ...]]] = {}
|
|
1941
|
+
vector_payload_sha256s: list[str] = []
|
|
1942
|
+
for backend in backends:
|
|
1943
|
+
layers: dict[str, tuple[tuple[int, ...], ...]] = {}
|
|
1944
|
+
for descriptor in descriptors:
|
|
1945
|
+
if descriptor.kind != "vector" or descriptor.backend_coordinate != backend.coordinate:
|
|
1946
|
+
continue
|
|
1947
|
+
raw_member = _verified_member(directory, descriptor, maximum=MAX_VECTOR_MEMBER_BYTES)
|
|
1948
|
+
member = parse_vector_member(raw_member, descriptor=descriptor)
|
|
1949
|
+
vector_payload_sha256s.append(sha256_bytes(raw_member[MEMBER_HEADER_BYTES:]))
|
|
1950
|
+
if descriptor.dimensions != backend.dimensions:
|
|
1951
|
+
raise _refuse(
|
|
1952
|
+
"PACKED_MEMBER_DIMENSION",
|
|
1953
|
+
"packed.range.members[]",
|
|
1954
|
+
"a vector member does not carry its backend's dimension",
|
|
1955
|
+
)
|
|
1956
|
+
if member.layer in layers:
|
|
1957
|
+
raise _refuse(
|
|
1958
|
+
"PACKED_MEMBER_LAYER",
|
|
1959
|
+
"packed.range.members[]",
|
|
1960
|
+
"a backend carries each layer exactly once",
|
|
1961
|
+
)
|
|
1962
|
+
layers[member.layer] = member.vectors
|
|
1963
|
+
if set(layers) != set(EMBEDDING_LAYERS):
|
|
1964
|
+
raise _refuse(
|
|
1965
|
+
"PACKED_MEMBER_LAYER",
|
|
1966
|
+
"packed.range.members[]",
|
|
1967
|
+
f"every backend closes over the four layers {list(EMBEDDING_LAYERS)}",
|
|
1968
|
+
)
|
|
1969
|
+
vectors_by_backend[backend.coordinate] = layers
|
|
1970
|
+
|
|
1971
|
+
bound_sha256s: list[str] = []
|
|
1972
|
+
for backend in backends:
|
|
1973
|
+
descriptor = _one(
|
|
1974
|
+
[item for item in descriptors if item.backend_coordinate == backend.coordinate], "bound"
|
|
1975
|
+
)
|
|
1976
|
+
raw_member = _verified_member(
|
|
1977
|
+
directory, descriptor, maximum=bound_segment_bytes(dimensions=backend.dimensions)
|
|
1978
|
+
)
|
|
1979
|
+
segment = parse_bound_segment(raw_member, descriptor=descriptor)
|
|
1980
|
+
exact = compute_layer_bounds(
|
|
1981
|
+
vectors_by_backend[backend.coordinate], dimensions=backend.dimensions
|
|
1982
|
+
)
|
|
1983
|
+
_verify_bounds(segment.bounds, exact)
|
|
1984
|
+
bound_sha256s.append(descriptor.sha256)
|
|
1985
|
+
|
|
1986
|
+
posting_descriptors = [item for item in descriptors if item.kind == "posting"]
|
|
1987
|
+
if not MIN_POSTING_SEGMENTS <= len(posting_descriptors) <= MAX_POSTING_SEGMENTS:
|
|
1988
|
+
raise _refuse(
|
|
1989
|
+
"PACKED_POSTING_SEGMENTS",
|
|
1990
|
+
"packed.range.members[]",
|
|
1991
|
+
f"a range carries {MIN_POSTING_SEGMENTS} to {MAX_POSTING_SEGMENTS} posting segments",
|
|
1992
|
+
)
|
|
1993
|
+
observed: list[PostingTerm] = []
|
|
1994
|
+
for expected_index, descriptor in enumerate(posting_descriptors):
|
|
1995
|
+
if descriptor.segment_index != expected_index:
|
|
1996
|
+
raise _refuse(
|
|
1997
|
+
"PACKED_POSTING_SEGMENTS",
|
|
1998
|
+
"packed.range.members[]",
|
|
1999
|
+
"posting segments are published in ascending segment order",
|
|
2000
|
+
)
|
|
2001
|
+
raw_member = _verified_member(directory, descriptor, maximum=MAX_POSTING_SEGMENT_BYTES)
|
|
2002
|
+
segment = parse_posting_segment(raw_member, descriptor=descriptor)
|
|
2003
|
+
if descriptor.term_count != len(segment.terms):
|
|
2004
|
+
raise _refuse(
|
|
2005
|
+
"PACKED_POSTING_TERM",
|
|
2006
|
+
"packed.range.members[]",
|
|
2007
|
+
"a posting descriptor's term count differs from its segment",
|
|
2008
|
+
)
|
|
2009
|
+
if observed and segment.terms and segment.terms[0] <= observed[-1]:
|
|
2010
|
+
raise _refuse(
|
|
2011
|
+
"PACKED_POSTING_ORDER",
|
|
2012
|
+
"packed.range.members[]",
|
|
2013
|
+
"posting terms ascend across segment boundaries too",
|
|
2014
|
+
)
|
|
2015
|
+
observed.extend(segment.terms)
|
|
2016
|
+
posting_dimensions = next(
|
|
2017
|
+
backend.dimensions
|
|
2018
|
+
for backend in backends
|
|
2019
|
+
if backend.coordinate == posting_backend_coordinate
|
|
2020
|
+
)
|
|
2021
|
+
recomputed = compute_layer_postings(
|
|
2022
|
+
vectors_by_backend[posting_backend_coordinate], dimensions=posting_dimensions
|
|
2023
|
+
)
|
|
2024
|
+
if tuple(observed) != recomputed:
|
|
2025
|
+
raise _refuse(
|
|
2026
|
+
"PACKED_POSTING_MISMATCH",
|
|
2027
|
+
"packed.range.postings",
|
|
2028
|
+
"the published postings are not the postings these packs contain",
|
|
2029
|
+
)
|
|
2030
|
+
|
|
2031
|
+
return VerifiedPackedRange(
|
|
2032
|
+
range_index=manifest["range_index"],
|
|
2033
|
+
first_key=manifest["first_key"],
|
|
2034
|
+
last_key=manifest["last_key"],
|
|
2035
|
+
entry_count=manifest["entry_count"],
|
|
2036
|
+
facts_sha256=manifest["facts_sha256"],
|
|
2037
|
+
history_sha256=manifest["history_sha256"],
|
|
2038
|
+
range_binding_sha256=manifest["range_binding_sha256"],
|
|
2039
|
+
manifest_bytes=len(raw),
|
|
2040
|
+
member_count=len(descriptors),
|
|
2041
|
+
member_sha256s=tuple(descriptor.sha256 for descriptor in descriptors),
|
|
2042
|
+
vector_payload_sha256s=tuple(vector_payload_sha256s),
|
|
2043
|
+
posting_sha256s=tuple(descriptor.sha256 for descriptor in posting_descriptors),
|
|
2044
|
+
bound_sha256s=tuple(bound_sha256s),
|
|
2045
|
+
posting_term_count=len(observed),
|
|
2046
|
+
entry_ids=entry_ids,
|
|
2047
|
+
classification_counts=classification_counts,
|
|
2048
|
+
)
|
|
2049
|
+
|
|
2050
|
+
|
|
2051
|
+
def _one(descriptors: Sequence[PackedMemberDescriptor], kind: str) -> PackedMemberDescriptor:
|
|
2052
|
+
found = [descriptor for descriptor in descriptors if descriptor.kind == kind]
|
|
2053
|
+
if len(found) != 1:
|
|
2054
|
+
raise _refuse(
|
|
2055
|
+
"PACKED_RANGE_MEMBERS",
|
|
2056
|
+
"packed.range.members[]",
|
|
2057
|
+
f"a range carries exactly one {kind} member",
|
|
2058
|
+
)
|
|
2059
|
+
return found[0]
|
|
2060
|
+
|
|
2061
|
+
|
|
2062
|
+
def _verified_member(
|
|
2063
|
+
directory: PackedDirectory, descriptor: PackedMemberDescriptor, *, maximum: int
|
|
2064
|
+
) -> bytes:
|
|
2065
|
+
raw = directory.read_member(descriptor.sha256, kind=descriptor.kind, maximum=maximum)
|
|
2066
|
+
if raw is None or len(raw) != descriptor.bytes or sha256_bytes(raw) != descriptor.sha256:
|
|
2067
|
+
raise _refuse(
|
|
2068
|
+
"PACKED_MEMBER_DIGEST",
|
|
2069
|
+
"packed.range.members[]",
|
|
2070
|
+
"an installed member is not the member its descriptor binds",
|
|
2071
|
+
)
|
|
2072
|
+
return raw
|
|
2073
|
+
|
|
2074
|
+
|
|
2075
|
+
def _verify_facts_member(
|
|
2076
|
+
raw: bytes, manifest: Mapping[str, Any]
|
|
2077
|
+
) -> tuple[tuple[str, ...], tuple[tuple[str, int], ...]]:
|
|
2078
|
+
try:
|
|
2079
|
+
payload = parse_canonical_json(raw)
|
|
2080
|
+
except CanonicalJSONError as error:
|
|
2081
|
+
raise _refuse(
|
|
2082
|
+
"PACKED_FACTS_MEMBER", "packed.range.facts", "facts member is not canonical"
|
|
2083
|
+
) from error
|
|
2084
|
+
if (
|
|
2085
|
+
not isinstance(payload, dict)
|
|
2086
|
+
or payload.get("schema_version") != PACKED_FACTS_SCHEMA
|
|
2087
|
+
or payload.get("range_index") != manifest["range_index"]
|
|
2088
|
+
or payload.get("entry_count") != manifest["entry_count"]
|
|
2089
|
+
or not isinstance(payload.get("entries"), list)
|
|
2090
|
+
or len(payload["entries"]) != manifest["entry_count"]
|
|
2091
|
+
):
|
|
2092
|
+
raise _refuse("PACKED_FACTS_MEMBER", "packed.range.facts", "facts member contract differs")
|
|
2093
|
+
entry_ids: list[str] = []
|
|
2094
|
+
classification_counts = dict.fromkeys(PACKED_CLASSIFICATIONS, 0)
|
|
2095
|
+
keys: list[str] = []
|
|
2096
|
+
previous: bytes | None = None
|
|
2097
|
+
for position, item in enumerate(payload["entries"]):
|
|
2098
|
+
path = f"packed.range.facts.entries[{position}]"
|
|
2099
|
+
if not isinstance(item, dict) or set(item) != {
|
|
2100
|
+
"key",
|
|
2101
|
+
"entry_id",
|
|
2102
|
+
"classification",
|
|
2103
|
+
"semantic_facts_digest",
|
|
2104
|
+
"entry_sha256",
|
|
2105
|
+
"entry_json",
|
|
2106
|
+
}:
|
|
2107
|
+
raise _refuse("PACKED_FACTS_MEMBER", path, "a facts entry contract differs")
|
|
2108
|
+
if sha256_bytes(item["entry_json"].encode("utf-8")) != item["entry_sha256"]:
|
|
2109
|
+
raise _refuse("PACKED_FACTS_MEMBER", path, "an entry is not the entry it digests to")
|
|
2110
|
+
if item["classification"] not in PACKED_CLASSIFICATIONS:
|
|
2111
|
+
raise _refuse(
|
|
2112
|
+
"PACKED_FACTS_MEMBER", path, "an entry carries a publishable classification"
|
|
2113
|
+
)
|
|
2114
|
+
key = item["key"].encode("utf-8")
|
|
2115
|
+
if previous is not None and key <= previous:
|
|
2116
|
+
raise _refuse("PACKED_RANGE_ORDER", path, "a facts member ascends by key")
|
|
2117
|
+
previous = key
|
|
2118
|
+
entry_ids.append(item["entry_id"])
|
|
2119
|
+
classification_counts[item["classification"]] += 1
|
|
2120
|
+
keys.append(item["key"])
|
|
2121
|
+
if keys[0] != manifest["first_key"] or keys[-1] != manifest["last_key"]:
|
|
2122
|
+
raise _refuse(
|
|
2123
|
+
"PACKED_RANGE_ORDER",
|
|
2124
|
+
"packed.range.facts",
|
|
2125
|
+
"the facts member does not span the interval its range claims",
|
|
2126
|
+
)
|
|
2127
|
+
return tuple(entry_ids), tuple(classification_counts.items())
|
|
2128
|
+
|
|
2129
|
+
|
|
2130
|
+
def _verify_bounds(published: Sequence[LayerBounds], exact: Sequence[LayerBounds]) -> None:
|
|
2131
|
+
for layer_ordinal, (summary, truth) in enumerate(zip(published, exact, strict=True)):
|
|
2132
|
+
published_minima, published_maxima = summary
|
|
2133
|
+
exact_minima, exact_maxima = truth
|
|
2134
|
+
for dimension in range(len(exact_minima)):
|
|
2135
|
+
if (
|
|
2136
|
+
published_minima[dimension] > exact_minima[dimension]
|
|
2137
|
+
or published_maxima[dimension] < exact_maxima[dimension]
|
|
2138
|
+
):
|
|
2139
|
+
raise _refuse(
|
|
2140
|
+
"PACKED_BOUND_NOT_CONSERVATIVE",
|
|
2141
|
+
f"packed.range.bounds[{layer_ordinal}][{dimension}]",
|
|
2142
|
+
"a published bound excludes a vector this range actually contains",
|
|
2143
|
+
)
|
|
2144
|
+
if published_minima != exact_minima or published_maxima != exact_maxima:
|
|
2145
|
+
raise _refuse(
|
|
2146
|
+
"PACKED_BOUND_MISMATCH",
|
|
2147
|
+
f"packed.range.bounds[{layer_ordinal}]",
|
|
2148
|
+
"a published bound is not the exact publisher-derived summary of these packs",
|
|
2149
|
+
)
|
|
2150
|
+
|
|
2151
|
+
|
|
2152
|
+
def verify_packed_generation(
|
|
2153
|
+
root: Path, *, expected_head_sha256: str, root_descriptor: int | None = None
|
|
2154
|
+
) -> VerifiedPackedGeneration:
|
|
2155
|
+
"""Read one published generation end to end and refuse anything that is not exactly it."""
|
|
2156
|
+
|
|
2157
|
+
directory = PackedDirectory(Path(root), root_descriptor=root_descriptor)
|
|
2158
|
+
with directory.opened():
|
|
2159
|
+
raw = directory.read(PACKED_HEAD_FILENAME, maximum=MAX_PACKED_HEAD_BYTES)
|
|
2160
|
+
if raw is None:
|
|
2161
|
+
raise _refuse(
|
|
2162
|
+
"PACKED_HEAD_MISSING", "packed.head", "no readable packed head at this root"
|
|
2163
|
+
)
|
|
2164
|
+
digest = sha256_bytes(raw)
|
|
2165
|
+
if digest != expected_head_sha256:
|
|
2166
|
+
raise _refuse(
|
|
2167
|
+
"PACKED_HEAD_MISMATCH",
|
|
2168
|
+
"packed.head",
|
|
2169
|
+
"the installed head is not the caller's exact generation",
|
|
2170
|
+
)
|
|
2171
|
+
head = _parse_head(raw)
|
|
2172
|
+
backends = tuple(
|
|
2173
|
+
PackedBackend(
|
|
2174
|
+
coordinate=item["coordinate"],
|
|
2175
|
+
dimensions=item["dimensions"],
|
|
2176
|
+
quantization=item["quantization"],
|
|
2177
|
+
query_safe=item["query_safe"],
|
|
2178
|
+
)
|
|
2179
|
+
for item in head["backends"]
|
|
2180
|
+
)
|
|
2181
|
+
posting_backend = head["posting_backend_coordinate"]
|
|
2182
|
+
descriptors = [
|
|
2183
|
+
range_descriptor_from_dict(value, path=f"packed.head.ranges[{index}]")
|
|
2184
|
+
for index, value in enumerate(head["ranges"])
|
|
2185
|
+
]
|
|
2186
|
+
if ranges_chain_sha256(descriptors) != head["ranges_chain_sha256"]:
|
|
2187
|
+
raise _refuse(
|
|
2188
|
+
"PACKED_HEAD_EXPECTED",
|
|
2189
|
+
"packed.head.ranges_chain_sha256",
|
|
2190
|
+
"the head's expected range chain is not the chain of its own descriptors",
|
|
2191
|
+
)
|
|
2192
|
+
member_count = 0
|
|
2193
|
+
posting_term_count = 0
|
|
2194
|
+
vector_payload_sha256s: list[str] = []
|
|
2195
|
+
posting_member_sha256s: list[str] = []
|
|
2196
|
+
bound_member_sha256s: list[str] = []
|
|
2197
|
+
classification_counts = dict.fromkeys(PACKED_CLASSIFICATIONS, 0)
|
|
2198
|
+
fresh_range_count = 0
|
|
2199
|
+
previous: VerifiedPackedRange | None = None
|
|
2200
|
+
for position, descriptor in enumerate(descriptors):
|
|
2201
|
+
if descriptor.range_index != position:
|
|
2202
|
+
raise _refuse(
|
|
2203
|
+
"PACKED_RANGE_ORDER",
|
|
2204
|
+
f"packed.head.ranges[{position}]",
|
|
2205
|
+
"ranges are consecutive from zero",
|
|
2206
|
+
)
|
|
2207
|
+
verified = verify_packed_range(
|
|
2208
|
+
directory,
|
|
2209
|
+
descriptor.manifest_sha256,
|
|
2210
|
+
backends=backends,
|
|
2211
|
+
posting_backend_coordinate=posting_backend,
|
|
2212
|
+
)
|
|
2213
|
+
if (
|
|
2214
|
+
verified.range_index != descriptor.range_index
|
|
2215
|
+
or verified.first_key != descriptor.first_key
|
|
2216
|
+
or verified.last_key != descriptor.last_key
|
|
2217
|
+
or verified.entry_count != descriptor.entry_count
|
|
2218
|
+
or verified.facts_sha256 != descriptor.facts_sha256
|
|
2219
|
+
or verified.range_binding_sha256 != descriptor.range_binding_sha256
|
|
2220
|
+
or verified.manifest_bytes != descriptor.manifest_bytes
|
|
2221
|
+
or verified.member_count != descriptor.member_count
|
|
2222
|
+
):
|
|
2223
|
+
raise _refuse(
|
|
2224
|
+
"PACKED_RANGE_DIGEST",
|
|
2225
|
+
f"packed.head.ranges[{position}]",
|
|
2226
|
+
"a range descriptor does not describe the range it names",
|
|
2227
|
+
)
|
|
2228
|
+
is_final = position == len(descriptors) - 1
|
|
2229
|
+
if not is_final and verified.entry_count != RANGE_ENTRIES:
|
|
2230
|
+
raise _refuse(
|
|
2231
|
+
"PACKED_RANGE_SIZE",
|
|
2232
|
+
f"packed.head.ranges[{position}]",
|
|
2233
|
+
f"every range but the final one holds exactly {RANGE_ENTRIES} entries",
|
|
2234
|
+
)
|
|
2235
|
+
if previous is not None and not (
|
|
2236
|
+
previous.last_key.encode("utf-8") < verified.first_key.encode("utf-8")
|
|
2237
|
+
):
|
|
2238
|
+
raise _refuse(
|
|
2239
|
+
"PACKED_RANGE_ORDER",
|
|
2240
|
+
f"packed.head.ranges[{position}]",
|
|
2241
|
+
"ranges are consecutive intervals that ascend by key",
|
|
2242
|
+
)
|
|
2243
|
+
previous = verified
|
|
2244
|
+
member_count += verified.member_count
|
|
2245
|
+
posting_term_count += verified.posting_term_count
|
|
2246
|
+
vector_payload_sha256s.extend(verified.vector_payload_sha256s)
|
|
2247
|
+
posting_member_sha256s.extend(verified.posting_sha256s)
|
|
2248
|
+
bound_member_sha256s.extend(verified.bound_sha256s)
|
|
2249
|
+
range_counts = dict(verified.classification_counts)
|
|
2250
|
+
for classification in PACKED_CLASSIFICATIONS:
|
|
2251
|
+
classification_counts[classification] += range_counts[classification]
|
|
2252
|
+
if any(range_counts[name] for name in ("changed", "new")):
|
|
2253
|
+
fresh_range_count += 1
|
|
2254
|
+
entry_count = sum(descriptor.entry_count for descriptor in descriptors)
|
|
2255
|
+
if entry_count != head["entry_count"] or len(descriptors) != head["range_count"]:
|
|
2256
|
+
raise _refuse(
|
|
2257
|
+
"PACKED_HEAD_EXPECTED",
|
|
2258
|
+
"packed.head.counts",
|
|
2259
|
+
"the head's own counts disagree with its range descriptors",
|
|
2260
|
+
)
|
|
2261
|
+
return VerifiedPackedGeneration(
|
|
2262
|
+
head_sha256=digest,
|
|
2263
|
+
head=head,
|
|
2264
|
+
range_count=len(descriptors),
|
|
2265
|
+
entry_count=entry_count,
|
|
2266
|
+
member_count=member_count,
|
|
2267
|
+
posting_term_count=posting_term_count,
|
|
2268
|
+
backends=tuple(backend.coordinate for backend in backends),
|
|
2269
|
+
classification_counts=tuple(classification_counts.items()),
|
|
2270
|
+
fresh_range_count=fresh_range_count,
|
|
2271
|
+
vector_payload_sha256s=tuple(vector_payload_sha256s),
|
|
2272
|
+
posting_member_sha256s=tuple(posting_member_sha256s),
|
|
2273
|
+
bound_member_sha256s=tuple(bound_member_sha256s),
|
|
2274
|
+
)
|
|
2275
|
+
|
|
2276
|
+
|
|
2277
|
+
_HEAD_MEMBERS = frozenset(
|
|
2278
|
+
{
|
|
2279
|
+
"schema_version",
|
|
2280
|
+
"provider_id",
|
|
2281
|
+
"generation_sha256",
|
|
2282
|
+
"predecessor_head_sha256",
|
|
2283
|
+
"backends",
|
|
2284
|
+
"posting_backend_coordinate",
|
|
2285
|
+
"range_count",
|
|
2286
|
+
"entry_count",
|
|
2287
|
+
"coverage",
|
|
2288
|
+
"ranges",
|
|
2289
|
+
"ranges_chain_sha256",
|
|
2290
|
+
"counters",
|
|
2291
|
+
"invocations",
|
|
2292
|
+
"terminal_sha256",
|
|
2293
|
+
"root_sha256",
|
|
2294
|
+
}
|
|
2295
|
+
)
|
|
2296
|
+
|
|
2297
|
+
|
|
2298
|
+
def _parse_head(raw: bytes) -> dict[str, Any]:
|
|
2299
|
+
if len(raw) > MAX_PACKED_HEAD_BYTES:
|
|
2300
|
+
raise _refuse("PACKED_HEAD_LIMIT", "packed.head", "head exceeds its byte bound")
|
|
2301
|
+
try:
|
|
2302
|
+
head = parse_canonical_json(raw)
|
|
2303
|
+
except CanonicalJSONError as error:
|
|
2304
|
+
raise _refuse("PACKED_HEAD_SCHEMA", "packed.head", "head is not canonical") from error
|
|
2305
|
+
if (
|
|
2306
|
+
not isinstance(head, dict)
|
|
2307
|
+
or set(head) != _HEAD_MEMBERS
|
|
2308
|
+
or head["schema_version"] != PACKED_HEAD_SCHEMA
|
|
2309
|
+
or not isinstance(head["ranges"], list)
|
|
2310
|
+
or not isinstance(head["backends"], list)
|
|
2311
|
+
or not isinstance(head["counters"], dict)
|
|
2312
|
+
):
|
|
2313
|
+
raise _refuse("PACKED_HEAD_SCHEMA", "packed.head", "head contract differs")
|
|
2314
|
+
body = {key: value for key, value in head.items() if key != "root_sha256"}
|
|
2315
|
+
if canonical_sha256(body) != head["root_sha256"]:
|
|
2316
|
+
raise _refuse(
|
|
2317
|
+
"PACKED_HEAD_DIGEST",
|
|
2318
|
+
"packed.head.root_sha256",
|
|
2319
|
+
"the head's own digest does not reproduce from the head it signs",
|
|
2320
|
+
)
|
|
2321
|
+
if not 1 <= len(head["ranges"]) <= MAX_RANGE_DESCRIPTORS:
|
|
2322
|
+
raise _refuse(
|
|
2323
|
+
"PACKED_HEAD_DESCRIPTORS",
|
|
2324
|
+
"packed.head.ranges",
|
|
2325
|
+
f"an outer head holds 1 to {MAX_RANGE_DESCRIPTORS} range descriptors",
|
|
2326
|
+
)
|
|
2327
|
+
if not 1 <= len(head["backends"]) <= MAX_PACKED_BACKENDS:
|
|
2328
|
+
raise _refuse(
|
|
2329
|
+
"PACKED_BACKEND",
|
|
2330
|
+
"packed.head.backends",
|
|
2331
|
+
f"a packed generation carries 1 to {MAX_PACKED_BACKENDS} backends",
|
|
2332
|
+
)
|
|
2333
|
+
if set(head["counters"]) != set(COUNTER_MEMBERS):
|
|
2334
|
+
raise _refuse(
|
|
2335
|
+
"PACKED_COUNTER_CONTRACT",
|
|
2336
|
+
"packed.head.counters",
|
|
2337
|
+
f"the head carries exactly {list(COUNTER_MEMBERS)}",
|
|
2338
|
+
)
|
|
2339
|
+
if not coverage_is_valid(head["coverage"]):
|
|
2340
|
+
raise _refuse(
|
|
2341
|
+
"PACKED_HEAD_COVERAGE",
|
|
2342
|
+
"packed.head.coverage",
|
|
2343
|
+
"a published generation states what it covers and what it drew from",
|
|
2344
|
+
)
|
|
2345
|
+
return head
|