mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,1517 @@
|
|
|
1
|
+
"""Bounded exact retrieval over a range-local packed generation.
|
|
2
|
+
|
|
3
|
+
:mod:`packed_catalog` defines what a packed generation *is* and :mod:`packed_writer` publishes one.
|
|
4
|
+
This module is the only thing that reads one to answer a question, and it is deliberately the
|
|
5
|
+
smallest surface that can do so: identities out, numbers never.
|
|
6
|
+
|
|
7
|
+
Four properties are load-bearing.
|
|
8
|
+
|
|
9
|
+
*Nothing is searched that the caller did not name.* A search takes the caller's exact outer head
|
|
10
|
+
digest **and** the publication receipt whose coordinate re-derives from its own body, and it refuses
|
|
11
|
+
unless that receipt describes the head installed at this root -- the same range descriptors, the
|
|
12
|
+
same range chain, the same counters, the same backends. Every range manifest, bound segment,
|
|
13
|
+
posting segment, vector pack and facts member is then read back and compared against the digest the
|
|
14
|
+
authenticated chain binds it to. An ambient head, an unbound summary, a substituted pack and a
|
|
15
|
+
re-minted receipt that quietly drops an accelerator out of its own binding all refuse rather than
|
|
16
|
+
answer.
|
|
17
|
+
|
|
18
|
+
*The answer is exact whenever it is complete.* An entry's score is the maximum
|
|
19
|
+
over the four layers of the exact integer dot product of the query with that layer's vector. When
|
|
20
|
+
the selected backend is the generation's posting backend the four layer dot products are accumulated
|
|
21
|
+
from the exact posting segments; otherwise the four layer packs are opened and scored directly. The
|
|
22
|
+
two agree term for term, because the postings *are* the nonzero terms of those packs.
|
|
23
|
+
|
|
24
|
+
*Pruning is strict, and the bound is the publisher's.* A range is skipped only when the sign-aware
|
|
25
|
+
conservative bound of :func:`packed_catalog.range_upper_bound`, computed against the independently
|
|
26
|
+
readback-validated bound segment published with the generation, is **strictly** below the current
|
|
27
|
+
kth score. An
|
|
28
|
+
equal bound still exact-scans: a tie is broken by entry id ascending, and a pruned range cannot
|
|
29
|
+
present an id. Nothing recomputes a bound from the packs at query time; that is the publisher's job
|
|
30
|
+
and it was already done by an independent reader before the head was installed.
|
|
31
|
+
|
|
32
|
+
*Work is bounded before it is spent.* :class:`CatalogSearchWorkLimits` is closed and has no
|
|
33
|
+
defaults, so no caller can accidentally ask for unbounded work. Ranges, packs, member bytes,
|
|
34
|
+
posting terms, candidates, open descriptors, query encodes and elapsed time are each checked before
|
|
35
|
+
the work that would consume them. A cap that closes before every unpruned range has closed returns
|
|
36
|
+
a typed *incomplete* carrying the work actuals and **no candidate at all** -- a partial ranking that
|
|
37
|
+
reads like a complete one is the one answer this module may never give.
|
|
38
|
+
|
|
39
|
+
There is no threshold, no approximate neighbour search, no persisted score and no query cache. A
|
|
40
|
+
threshold would be an admission decision taken by similarity; the other three would each make the
|
|
41
|
+
answer a function of something other than the exact bytes the caller authenticated.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
from __future__ import annotations
|
|
45
|
+
|
|
46
|
+
import contextlib
|
|
47
|
+
import os
|
|
48
|
+
import time
|
|
49
|
+
from collections.abc import Iterator, Mapping, Sequence
|
|
50
|
+
from dataclasses import dataclass
|
|
51
|
+
from pathlib import Path
|
|
52
|
+
from typing import Any
|
|
53
|
+
|
|
54
|
+
from mostlyright.data_harness.canonical import (
|
|
55
|
+
CanonicalJSONError,
|
|
56
|
+
canonical_sha256,
|
|
57
|
+
parse_canonical_json,
|
|
58
|
+
sha256_bytes,
|
|
59
|
+
)
|
|
60
|
+
from mostlyright.data_harness.sources.catalog.bounded_io import (
|
|
61
|
+
BoundedReadFailure,
|
|
62
|
+
read_bounded_descriptor,
|
|
63
|
+
)
|
|
64
|
+
from mostlyright.data_harness.sources.catalog.contracts import EMBEDDING_LAYERS
|
|
65
|
+
from mostlyright.data_harness.sources.catalog.coverage import coverage_is_valid
|
|
66
|
+
from mostlyright.data_harness.sources.catalog.embedding import (
|
|
67
|
+
BOUNDED_SCALAR_ADAPTER,
|
|
68
|
+
EmbeddingBackend,
|
|
69
|
+
)
|
|
70
|
+
from mostlyright.data_harness.sources.catalog.entry_v2 import (
|
|
71
|
+
CatalogEntryV2,
|
|
72
|
+
catalog_entry_v2_from_dict,
|
|
73
|
+
)
|
|
74
|
+
from mostlyright.data_harness.sources.catalog.packed_catalog import (
|
|
75
|
+
MAX_FACTS_MEMBER_BYTES,
|
|
76
|
+
MAX_PACKED_HEAD_BYTES,
|
|
77
|
+
MAX_POSTING_SEGMENT_BYTES,
|
|
78
|
+
MAX_POSTING_SEGMENTS,
|
|
79
|
+
MAX_RANGE_DESCRIPTORS,
|
|
80
|
+
MAX_RANGE_MANIFEST_BYTES,
|
|
81
|
+
MAX_RANGE_MEMBER_DESCRIPTORS,
|
|
82
|
+
MAX_VECTOR_MEMBER_BYTES,
|
|
83
|
+
MEMBER_HEADER_BYTES,
|
|
84
|
+
MIN_POSTING_SEGMENTS,
|
|
85
|
+
PACKED_FACTS_SCHEMA,
|
|
86
|
+
PACKED_HEAD_FILENAME,
|
|
87
|
+
PACKED_HEAD_SCHEMA,
|
|
88
|
+
PACKED_MEMBERS_DIRNAME,
|
|
89
|
+
PACKED_RANGE_MANIFEST_SCHEMA,
|
|
90
|
+
RANGE_ENTRIES,
|
|
91
|
+
CatalogPackedRefused,
|
|
92
|
+
PackedMemberDescriptor,
|
|
93
|
+
PackedRangeDescriptor,
|
|
94
|
+
bound_segment_bytes,
|
|
95
|
+
member_descriptor_from_dict,
|
|
96
|
+
member_filename,
|
|
97
|
+
parse_bound_segment,
|
|
98
|
+
parse_posting_segment,
|
|
99
|
+
parse_vector_member,
|
|
100
|
+
range_binding_sha256,
|
|
101
|
+
range_descriptor_from_dict,
|
|
102
|
+
range_upper_bound,
|
|
103
|
+
ranges_chain_sha256,
|
|
104
|
+
)
|
|
105
|
+
from mostlyright.data_harness.sources.catalog.retrieval import (
|
|
106
|
+
MAX_QUERY_CHARS,
|
|
107
|
+
MAX_RETRIEVAL_LIMIT,
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
#: The publication receipt schema this reader admits. It is stated here rather than imported from
|
|
111
|
+
#: :mod:`packed_writer`, because a read path that imports the publisher would put a writer on the
|
|
112
|
+
#: query call graph; ``test_catalog_packed_retrieval`` asserts the two constants are one string.
|
|
113
|
+
PACKED_RECEIPT_SCHEMA = "mr-data-catalog-packed.v1"
|
|
114
|
+
PACKED_SEARCH_RESULT_SCHEMA = "harness-catalog-packed-search.v1"
|
|
115
|
+
PACKED_WORK_LIMITS_SCHEMA = "harness-catalog-packed-work-limits.v1"
|
|
116
|
+
|
|
117
|
+
RETRIEVAL_V2_FILENAME = "retrieval.json"
|
|
118
|
+
SEALED_V1_FILENAME = "catalog.json"
|
|
119
|
+
|
|
120
|
+
#: The three public-source stores this repository can answer from, newest first. Dispatch is by
|
|
121
|
+
#: exact filename, and the packed head is additionally admitted only under its exact schema.
|
|
122
|
+
PUBLIC_STORE_KINDS = ("packed", "retrieval_v2", "sealed_v1")
|
|
123
|
+
|
|
124
|
+
MAX_WORK_RANGES = MAX_RANGE_DESCRIPTORS
|
|
125
|
+
MAX_WORK_PACKS = MAX_RANGE_DESCRIPTORS * MAX_RANGE_MEMBER_DESCRIPTORS
|
|
126
|
+
MAX_WORK_MEMBER_BYTES = 64 * 1024 * 1024 * 1024
|
|
127
|
+
MAX_WORK_POSTING_TERMS = 1 << 33
|
|
128
|
+
MAX_WORK_CANDIDATES = MAX_RANGE_DESCRIPTORS * RANGE_ENTRIES
|
|
129
|
+
|
|
130
|
+
#: Three descriptors is the whole peak: the root, its members directory, and one member.
|
|
131
|
+
MIN_WORK_OPEN_FILES = 3
|
|
132
|
+
MAX_WORK_OPEN_FILES = 16
|
|
133
|
+
MAX_WORK_QUERY_ENCODES = 8
|
|
134
|
+
MAX_WORK_ELAPSED_SECONDS = 24 * 60 * 60
|
|
135
|
+
|
|
136
|
+
_NANOSECONDS = 1_000_000_000
|
|
137
|
+
|
|
138
|
+
PACKED_INCOMPLETE_CODES = (
|
|
139
|
+
"PACKED_SEARCH_RANGE_BUDGET",
|
|
140
|
+
"PACKED_SEARCH_PACK_BUDGET",
|
|
141
|
+
"PACKED_SEARCH_MEMBER_BYTE_BUDGET",
|
|
142
|
+
"PACKED_SEARCH_POSTING_TERM_BUDGET",
|
|
143
|
+
"PACKED_SEARCH_CANDIDATE_BUDGET",
|
|
144
|
+
"PACKED_SEARCH_OPEN_FILE_BUDGET",
|
|
145
|
+
"PACKED_SEARCH_QUERY_ENCODE_BUDGET",
|
|
146
|
+
"PACKED_SEARCH_TIME_BUDGET",
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
_OPEN_DIRECTORY = (
|
|
150
|
+
os.O_RDONLY
|
|
151
|
+
| getattr(os, "O_DIRECTORY", 0)
|
|
152
|
+
| getattr(os, "O_NOFOLLOW", 0)
|
|
153
|
+
| getattr(os, "O_CLOEXEC", 0)
|
|
154
|
+
)
|
|
155
|
+
_OPEN_MEMBER = (
|
|
156
|
+
os.O_RDONLY
|
|
157
|
+
| getattr(os, "O_NOFOLLOW", 0)
|
|
158
|
+
| getattr(os, "O_CLOEXEC", 0)
|
|
159
|
+
| getattr(os, "O_NONBLOCK", 0)
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
_READ_CHUNK = 1 << 20
|
|
163
|
+
|
|
164
|
+
_RECEIPT_OUTPUT_MEMBERS = frozenset(
|
|
165
|
+
{
|
|
166
|
+
"head_sha256",
|
|
167
|
+
"range_count",
|
|
168
|
+
"entry_count",
|
|
169
|
+
"ranges",
|
|
170
|
+
"ranges_chain_sha256",
|
|
171
|
+
"vector_member_sha256s",
|
|
172
|
+
"posting_member_sha256s",
|
|
173
|
+
"bound_member_sha256s",
|
|
174
|
+
}
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
_FACTS_ENTRY_MEMBERS = frozenset(
|
|
178
|
+
{
|
|
179
|
+
"key",
|
|
180
|
+
"entry_id",
|
|
181
|
+
"classification",
|
|
182
|
+
"semantic_facts_digest",
|
|
183
|
+
"entry_sha256",
|
|
184
|
+
"entry_json",
|
|
185
|
+
}
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
_RANGE_MANIFEST_MEMBERS = frozenset(
|
|
189
|
+
{
|
|
190
|
+
"schema_version",
|
|
191
|
+
"range_index",
|
|
192
|
+
"first_key",
|
|
193
|
+
"last_key",
|
|
194
|
+
"entry_count",
|
|
195
|
+
"facts_sha256",
|
|
196
|
+
"history_sha256",
|
|
197
|
+
"range_binding_sha256",
|
|
198
|
+
"backends",
|
|
199
|
+
"posting_backend_coordinate",
|
|
200
|
+
"members",
|
|
201
|
+
"root_sha256",
|
|
202
|
+
}
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _refuse(code: str, path: str, detail: str) -> CatalogPackedRefused:
|
|
207
|
+
return CatalogPackedRefused(code, path, detail)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _is_digest(value: Any) -> bool:
|
|
211
|
+
return (
|
|
212
|
+
isinstance(value, str)
|
|
213
|
+
and len(value) == 64
|
|
214
|
+
and all(character in "0123456789abcdef" for character in value)
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _now_ns() -> int:
|
|
219
|
+
"""One monotonic clock reading. Named so a test can hold it still."""
|
|
220
|
+
|
|
221
|
+
return time.monotonic_ns()
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
# ------------------------------------------------------------------------------------------
|
|
225
|
+
# The closed work contract
|
|
226
|
+
# ------------------------------------------------------------------------------------------
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
_WORK_LIMIT_CEILINGS = {
|
|
230
|
+
"max_ranges": MAX_WORK_RANGES,
|
|
231
|
+
"max_packs": MAX_WORK_PACKS,
|
|
232
|
+
"max_member_bytes": MAX_WORK_MEMBER_BYTES,
|
|
233
|
+
"max_posting_terms": MAX_WORK_POSTING_TERMS,
|
|
234
|
+
"max_candidates": MAX_WORK_CANDIDATES,
|
|
235
|
+
"max_open_files": MAX_WORK_OPEN_FILES,
|
|
236
|
+
"max_query_encodes": MAX_WORK_QUERY_ENCODES,
|
|
237
|
+
"max_elapsed_seconds": MAX_WORK_ELAPSED_SECONDS,
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
@dataclass(frozen=True)
|
|
242
|
+
class CatalogSearchWorkLimits:
|
|
243
|
+
"""Every bound one packed search may consume, stated before it consumes any of them.
|
|
244
|
+
|
|
245
|
+
There is no default anywhere on this contract. A default would be a hidden answer to "how much
|
|
246
|
+
work may this question cost", supplied by whoever forgot to ask rather than by the caller who
|
|
247
|
+
has to live with it.
|
|
248
|
+
"""
|
|
249
|
+
|
|
250
|
+
max_ranges: int
|
|
251
|
+
max_packs: int
|
|
252
|
+
max_member_bytes: int
|
|
253
|
+
max_posting_terms: int
|
|
254
|
+
max_candidates: int
|
|
255
|
+
max_open_files: int
|
|
256
|
+
max_query_encodes: int
|
|
257
|
+
max_elapsed_seconds: int
|
|
258
|
+
|
|
259
|
+
def __post_init__(self) -> None:
|
|
260
|
+
for name, ceiling in _WORK_LIMIT_CEILINGS.items():
|
|
261
|
+
value = getattr(self, name)
|
|
262
|
+
if type(value) is not int or not 1 <= value <= ceiling:
|
|
263
|
+
raise _refuse(
|
|
264
|
+
"PACKED_SEARCH_LIMIT",
|
|
265
|
+
f"work_limits.{name}",
|
|
266
|
+
f"must be an integer in [1, {ceiling}]",
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
def to_dict(self) -> dict[str, Any]:
|
|
270
|
+
return {
|
|
271
|
+
"schema_version": PACKED_WORK_LIMITS_SCHEMA,
|
|
272
|
+
**{name: getattr(self, name) for name in _WORK_LIMIT_CEILINGS},
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
@property
|
|
276
|
+
def digest(self) -> str:
|
|
277
|
+
return canonical_sha256(self.to_dict())
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
@dataclass(frozen=True)
|
|
281
|
+
class PackedSearchWork:
|
|
282
|
+
"""What one search actually did. Every number here is deterministic; elapsed time is not one."""
|
|
283
|
+
|
|
284
|
+
ranges_bounded: int
|
|
285
|
+
ranges_scanned: int
|
|
286
|
+
ranges_pruned: int
|
|
287
|
+
packs_opened: int
|
|
288
|
+
member_bytes_read: int
|
|
289
|
+
posting_terms_read: int
|
|
290
|
+
candidates_examined: int
|
|
291
|
+
open_files_peak: int
|
|
292
|
+
query_encodes: int
|
|
293
|
+
|
|
294
|
+
def to_dict(self) -> dict[str, int]:
|
|
295
|
+
return {name: getattr(self, name) for name in self.__dataclass_fields__}
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
class _Exhausted(Exception):
|
|
299
|
+
"""One work cap closed.
|
|
300
|
+
|
|
301
|
+
It is private control flow, not a typed refusal: it never leaves this module as an exception,
|
|
302
|
+
and the budget it names becomes ``PackedRetrievalResult.incomplete_code`` instead. So it
|
|
303
|
+
deliberately carries a ``budget`` rather than a ``code`` -- a workbench reader never sees this
|
|
304
|
+
class, and a class that looks like a typed error while never reaching a person is exactly the
|
|
305
|
+
thing the plain-words gates exist to notice.
|
|
306
|
+
"""
|
|
307
|
+
|
|
308
|
+
def __init__(self, budget: str) -> None:
|
|
309
|
+
super().__init__(budget)
|
|
310
|
+
self.budget = budget
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
class _Budget:
|
|
314
|
+
"""Every cap, spent before the work it pays for rather than measured after it."""
|
|
315
|
+
|
|
316
|
+
def __init__(self, limits: CatalogSearchWorkLimits) -> None:
|
|
317
|
+
self._limits = limits
|
|
318
|
+
self._deadline_ns = _now_ns() + limits.max_elapsed_seconds * _NANOSECONDS
|
|
319
|
+
self._open_files = 0
|
|
320
|
+
self.ranges_bounded = 0
|
|
321
|
+
self.ranges_scanned = 0
|
|
322
|
+
self.ranges_pruned = 0
|
|
323
|
+
self.packs_opened = 0
|
|
324
|
+
self.member_bytes_read = 0
|
|
325
|
+
self.posting_terms_read = 0
|
|
326
|
+
self.candidates_examined = 0
|
|
327
|
+
self.open_files_peak = 0
|
|
328
|
+
self.query_encodes = 0
|
|
329
|
+
|
|
330
|
+
def work(self) -> PackedSearchWork:
|
|
331
|
+
return PackedSearchWork(
|
|
332
|
+
ranges_bounded=self.ranges_bounded,
|
|
333
|
+
ranges_scanned=self.ranges_scanned,
|
|
334
|
+
ranges_pruned=self.ranges_pruned,
|
|
335
|
+
packs_opened=self.packs_opened,
|
|
336
|
+
member_bytes_read=self.member_bytes_read,
|
|
337
|
+
posting_terms_read=self.posting_terms_read,
|
|
338
|
+
candidates_examined=self.candidates_examined,
|
|
339
|
+
open_files_peak=self.open_files_peak,
|
|
340
|
+
query_encodes=self.query_encodes,
|
|
341
|
+
)
|
|
342
|
+
|
|
343
|
+
def check_time(self) -> None:
|
|
344
|
+
if _now_ns() > self._deadline_ns:
|
|
345
|
+
raise _Exhausted("PACKED_SEARCH_TIME_BUDGET")
|
|
346
|
+
|
|
347
|
+
def spend_range_bounded(self) -> None:
|
|
348
|
+
if self.ranges_bounded + 1 > self._limits.max_ranges:
|
|
349
|
+
raise _Exhausted("PACKED_SEARCH_RANGE_BUDGET")
|
|
350
|
+
self.ranges_bounded += 1
|
|
351
|
+
|
|
352
|
+
def spend_range_scanned(self) -> None:
|
|
353
|
+
if self.ranges_scanned + 1 > self._limits.max_ranges:
|
|
354
|
+
raise _Exhausted("PACKED_SEARCH_RANGE_BUDGET")
|
|
355
|
+
self.ranges_scanned += 1
|
|
356
|
+
|
|
357
|
+
def note_pruned(self) -> None:
|
|
358
|
+
self.ranges_pruned += 1
|
|
359
|
+
|
|
360
|
+
def spend_member_bytes(self, count: int) -> None:
|
|
361
|
+
if self.member_bytes_read + count > self._limits.max_member_bytes:
|
|
362
|
+
raise _Exhausted("PACKED_SEARCH_MEMBER_BYTE_BUDGET")
|
|
363
|
+
self.member_bytes_read += count
|
|
364
|
+
|
|
365
|
+
def spend_pack(self) -> None:
|
|
366
|
+
if self.packs_opened + 1 > self._limits.max_packs:
|
|
367
|
+
raise _Exhausted("PACKED_SEARCH_PACK_BUDGET")
|
|
368
|
+
self.packs_opened += 1
|
|
369
|
+
|
|
370
|
+
def spend_posting_terms(self, count: int) -> None:
|
|
371
|
+
if self.posting_terms_read + count > self._limits.max_posting_terms:
|
|
372
|
+
raise _Exhausted("PACKED_SEARCH_POSTING_TERM_BUDGET")
|
|
373
|
+
self.posting_terms_read += count
|
|
374
|
+
|
|
375
|
+
def spend_candidates(self, count: int) -> None:
|
|
376
|
+
if self.candidates_examined + count > self._limits.max_candidates:
|
|
377
|
+
raise _Exhausted("PACKED_SEARCH_CANDIDATE_BUDGET")
|
|
378
|
+
self.candidates_examined += count
|
|
379
|
+
|
|
380
|
+
def spend_query_encode(self) -> None:
|
|
381
|
+
if self.query_encodes + 1 > self._limits.max_query_encodes:
|
|
382
|
+
raise _Exhausted("PACKED_SEARCH_QUERY_ENCODE_BUDGET")
|
|
383
|
+
self.query_encodes += 1
|
|
384
|
+
|
|
385
|
+
def open_file(self) -> None:
|
|
386
|
+
if self._open_files + 1 > self._limits.max_open_files:
|
|
387
|
+
raise _Exhausted("PACKED_SEARCH_OPEN_FILE_BUDGET")
|
|
388
|
+
self._open_files += 1
|
|
389
|
+
if self._open_files > self.open_files_peak:
|
|
390
|
+
self.open_files_peak = self._open_files
|
|
391
|
+
|
|
392
|
+
def close_file(self) -> None:
|
|
393
|
+
self._open_files -= 1
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
# ------------------------------------------------------------------------------------------
|
|
397
|
+
# The read-only packed root
|
|
398
|
+
# ------------------------------------------------------------------------------------------
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def _read_regular(parent_fd: int, name: str, *, maximum: int, budget: _Budget | None) -> bytes:
|
|
402
|
+
"""Read one bounded single-link regular file relative to a retained directory."""
|
|
403
|
+
|
|
404
|
+
if budget is not None:
|
|
405
|
+
budget.open_file()
|
|
406
|
+
try:
|
|
407
|
+
descriptor = os.open(name, _OPEN_MEMBER, dir_fd=parent_fd)
|
|
408
|
+
except FileNotFoundError:
|
|
409
|
+
if budget is not None:
|
|
410
|
+
budget.close_file()
|
|
411
|
+
raise
|
|
412
|
+
except OSError as error:
|
|
413
|
+
if budget is not None:
|
|
414
|
+
budget.close_file()
|
|
415
|
+
raise _refuse("PACKED_SEARCH_MEMBER", name, "cannot open a plain packed member") from error
|
|
416
|
+
try:
|
|
417
|
+
info = os.fstat(descriptor)
|
|
418
|
+
if budget is not None:
|
|
419
|
+
budget.spend_member_bytes(info.st_size)
|
|
420
|
+
try:
|
|
421
|
+
return read_bounded_descriptor(descriptor, maximum=maximum)
|
|
422
|
+
except BoundedReadFailure as error:
|
|
423
|
+
raise _refuse(
|
|
424
|
+
"PACKED_SEARCH_MEMBER",
|
|
425
|
+
name,
|
|
426
|
+
"a packed member must be stable, bounded, and single-link regular",
|
|
427
|
+
) from error
|
|
428
|
+
finally:
|
|
429
|
+
os.close(descriptor)
|
|
430
|
+
if budget is not None:
|
|
431
|
+
budget.close_file()
|
|
432
|
+
|
|
433
|
+
|
|
434
|
+
class _PackedRoot:
|
|
435
|
+
"""One locally confined packed root, opened read-only and never by pathname twice."""
|
|
436
|
+
|
|
437
|
+
def __init__(self, root: Path, budget: _Budget) -> None:
|
|
438
|
+
self.root = Path(root)
|
|
439
|
+
self._budget = budget
|
|
440
|
+
self._root_fd = -1
|
|
441
|
+
self._members_fd = -1
|
|
442
|
+
self._stack: contextlib.ExitStack | None = None
|
|
443
|
+
|
|
444
|
+
@contextlib.contextmanager
|
|
445
|
+
def opened(self) -> Iterator[_PackedRoot]:
|
|
446
|
+
"""Retain the root for the whole search. The members directory opens on first use.
|
|
447
|
+
|
|
448
|
+
The head is a document of the root, so a root with no members directory still refuses on
|
|
449
|
+
its missing head rather than on a directory a caller never asked about.
|
|
450
|
+
"""
|
|
451
|
+
|
|
452
|
+
with contextlib.ExitStack() as stack:
|
|
453
|
+
self._budget.open_file()
|
|
454
|
+
stack.callback(self._budget.close_file)
|
|
455
|
+
try:
|
|
456
|
+
root_fd = os.open(self.root, _OPEN_DIRECTORY)
|
|
457
|
+
except OSError as error:
|
|
458
|
+
raise _refuse(
|
|
459
|
+
"PACKED_SEARCH_ROOT", str(self.root), "cannot open a plain packed root"
|
|
460
|
+
) from error
|
|
461
|
+
stack.callback(os.close, root_fd)
|
|
462
|
+
self._root_fd = root_fd
|
|
463
|
+
self._stack = stack
|
|
464
|
+
try:
|
|
465
|
+
yield self
|
|
466
|
+
finally:
|
|
467
|
+
self._root_fd = -1
|
|
468
|
+
self._members_fd = -1
|
|
469
|
+
self._stack = None
|
|
470
|
+
|
|
471
|
+
def _members(self) -> int:
|
|
472
|
+
if self._members_fd != -1:
|
|
473
|
+
return self._members_fd
|
|
474
|
+
stack = self._stack
|
|
475
|
+
if stack is None: # pragma: no cover - only reachable outside `opened`
|
|
476
|
+
raise _refuse("PACKED_SEARCH_ROOT", str(self.root), "the packed root is not open")
|
|
477
|
+
self._budget.open_file()
|
|
478
|
+
stack.callback(self._budget.close_file)
|
|
479
|
+
try:
|
|
480
|
+
members_fd = os.open(PACKED_MEMBERS_DIRNAME, _OPEN_DIRECTORY, dir_fd=self._root_fd)
|
|
481
|
+
except OSError as error:
|
|
482
|
+
raise _refuse(
|
|
483
|
+
"PACKED_SEARCH_ROOT",
|
|
484
|
+
PACKED_MEMBERS_DIRNAME,
|
|
485
|
+
"cannot open the packed members directory",
|
|
486
|
+
) from error
|
|
487
|
+
stack.callback(os.close, members_fd)
|
|
488
|
+
self._members_fd = members_fd
|
|
489
|
+
return members_fd
|
|
490
|
+
|
|
491
|
+
def read_document(self, name: str, *, maximum: int) -> bytes | None:
|
|
492
|
+
try:
|
|
493
|
+
return _read_regular(self._root_fd, name, maximum=maximum, budget=self._budget)
|
|
494
|
+
except FileNotFoundError:
|
|
495
|
+
return None
|
|
496
|
+
|
|
497
|
+
def read_member(self, sha256: str, *, kind: str, maximum: int) -> bytes | None:
|
|
498
|
+
members_fd = self._members()
|
|
499
|
+
self._budget.spend_pack()
|
|
500
|
+
try:
|
|
501
|
+
return _read_regular(
|
|
502
|
+
members_fd,
|
|
503
|
+
member_filename(kind, sha256),
|
|
504
|
+
maximum=maximum,
|
|
505
|
+
budget=self._budget,
|
|
506
|
+
)
|
|
507
|
+
except FileNotFoundError:
|
|
508
|
+
return None
|
|
509
|
+
|
|
510
|
+
|
|
511
|
+
def detect_public_store_kind(root: Path) -> str:
|
|
512
|
+
"""Name the public store installed at ``root`` by exact filename, never by discovery.
|
|
513
|
+
|
|
514
|
+
A packed head is admitted only under its exact schema: a file with that name that is not a
|
|
515
|
+
packed head refuses rather than silently falling through to the legacy v2 reader.
|
|
516
|
+
"""
|
|
517
|
+
|
|
518
|
+
if not isinstance(root, Path):
|
|
519
|
+
raise _refuse("PACKED_SEARCH_STORE", "root", "must be a pathlib.Path")
|
|
520
|
+
try:
|
|
521
|
+
root_fd = os.open(root, _OPEN_DIRECTORY)
|
|
522
|
+
except OSError as error:
|
|
523
|
+
raise _refuse(
|
|
524
|
+
"PACKED_SEARCH_STORE", str(root), "no readable public store at this root"
|
|
525
|
+
) from error
|
|
526
|
+
try:
|
|
527
|
+
try:
|
|
528
|
+
raw = _read_regular(
|
|
529
|
+
root_fd, PACKED_HEAD_FILENAME, maximum=MAX_PACKED_HEAD_BYTES, budget=None
|
|
530
|
+
)
|
|
531
|
+
except FileNotFoundError:
|
|
532
|
+
raw = None
|
|
533
|
+
if raw is not None:
|
|
534
|
+
try:
|
|
535
|
+
head = parse_canonical_json(raw)
|
|
536
|
+
except CanonicalJSONError as error:
|
|
537
|
+
raise _refuse(
|
|
538
|
+
"PACKED_SEARCH_STORE",
|
|
539
|
+
PACKED_HEAD_FILENAME,
|
|
540
|
+
"a packed head is canonical JSON or it is not a packed head",
|
|
541
|
+
) from error
|
|
542
|
+
if not isinstance(head, dict) or head.get("schema_version") != PACKED_HEAD_SCHEMA:
|
|
543
|
+
raise _refuse(
|
|
544
|
+
"PACKED_SEARCH_STORE",
|
|
545
|
+
PACKED_HEAD_FILENAME,
|
|
546
|
+
"this root installs a head that is not a packed head",
|
|
547
|
+
)
|
|
548
|
+
return "packed"
|
|
549
|
+
for name, kind in (
|
|
550
|
+
(RETRIEVAL_V2_FILENAME, "retrieval_v2"),
|
|
551
|
+
(SEALED_V1_FILENAME, "sealed_v1"),
|
|
552
|
+
):
|
|
553
|
+
try:
|
|
554
|
+
os.stat(name, dir_fd=root_fd, follow_symlinks=False)
|
|
555
|
+
except FileNotFoundError:
|
|
556
|
+
continue
|
|
557
|
+
except OSError as error:
|
|
558
|
+
raise _refuse(
|
|
559
|
+
"PACKED_SEARCH_STORE", name, "cannot inspect a public store member"
|
|
560
|
+
) from error
|
|
561
|
+
return kind
|
|
562
|
+
raise _refuse(
|
|
563
|
+
"PACKED_SEARCH_STORE",
|
|
564
|
+
str(root),
|
|
565
|
+
"this root holds no packed head, v2 retrieval manifest or sealed catalog",
|
|
566
|
+
)
|
|
567
|
+
finally:
|
|
568
|
+
os.close(root_fd)
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
# ------------------------------------------------------------------------------------------
|
|
572
|
+
# What one search returns
|
|
573
|
+
# ------------------------------------------------------------------------------------------
|
|
574
|
+
|
|
575
|
+
|
|
576
|
+
@dataclass(frozen=True)
|
|
577
|
+
class PackedRankedEntry:
|
|
578
|
+
"""One ordered identity and the exact facts it was carried by. No ranking number, ever."""
|
|
579
|
+
|
|
580
|
+
rank: int
|
|
581
|
+
entry_id: str
|
|
582
|
+
entry_coordinate: str
|
|
583
|
+
entry_digest: str
|
|
584
|
+
semantic_facts_digest: str
|
|
585
|
+
range_index: int
|
|
586
|
+
entry: CatalogEntryV2
|
|
587
|
+
|
|
588
|
+
def to_dict(self) -> dict[str, Any]:
|
|
589
|
+
return {
|
|
590
|
+
"rank": self.rank,
|
|
591
|
+
"entry_id": self.entry_id,
|
|
592
|
+
"entry_coordinate": self.entry_coordinate,
|
|
593
|
+
"entry_digest": self.entry_digest,
|
|
594
|
+
"semantic_facts_digest": self.semantic_facts_digest,
|
|
595
|
+
}
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
@dataclass(frozen=True)
|
|
599
|
+
class PackedRetrievalResult:
|
|
600
|
+
"""One search's complete outcome, or a typed incomplete carrying only what it did.
|
|
601
|
+
|
|
602
|
+
An incomplete result carries no entry. That is the whole point of the type: a caller that
|
|
603
|
+
cannot tell a budget-truncated ranking from a finished one would present the first as the
|
|
604
|
+
second, and the difference is exactly the entries the search never looked at.
|
|
605
|
+
"""
|
|
606
|
+
|
|
607
|
+
status: str
|
|
608
|
+
incomplete_code: str | None
|
|
609
|
+
head_sha256: str
|
|
610
|
+
receipt_coordinate_sha256: str
|
|
611
|
+
backend_coordinate: str
|
|
612
|
+
question_digest: str
|
|
613
|
+
limit: int
|
|
614
|
+
range_manifest_sha256s: tuple[str, ...]
|
|
615
|
+
vector_member_sha256s: tuple[str, ...]
|
|
616
|
+
posting_member_sha256s: tuple[str, ...]
|
|
617
|
+
bound_member_sha256s: tuple[str, ...]
|
|
618
|
+
entries: tuple[PackedRankedEntry, ...]
|
|
619
|
+
limits: CatalogSearchWorkLimits
|
|
620
|
+
work: PackedSearchWork
|
|
621
|
+
|
|
622
|
+
def __post_init__(self) -> None:
|
|
623
|
+
if self.status not in {"complete", "incomplete"}:
|
|
624
|
+
raise _refuse(
|
|
625
|
+
"PACKED_SEARCH_STATUS", "result.status", "a search is complete or incomplete"
|
|
626
|
+
)
|
|
627
|
+
if self.status == "complete":
|
|
628
|
+
if self.incomplete_code is not None:
|
|
629
|
+
raise _refuse(
|
|
630
|
+
"PACKED_SEARCH_STATUS",
|
|
631
|
+
"result.incomplete_code",
|
|
632
|
+
"a complete search names no exhausted budget",
|
|
633
|
+
)
|
|
634
|
+
elif self.incomplete_code not in PACKED_INCOMPLETE_CODES or self.entries:
|
|
635
|
+
raise _refuse(
|
|
636
|
+
"PACKED_SEARCH_STATUS",
|
|
637
|
+
"result.entries",
|
|
638
|
+
"an incomplete search names the budget that closed and offers no ranked identity",
|
|
639
|
+
)
|
|
640
|
+
if tuple(item.rank for item in self.entries) != tuple(range(1, len(self.entries) + 1)):
|
|
641
|
+
raise _refuse(
|
|
642
|
+
"PACKED_SEARCH_ORDER",
|
|
643
|
+
"result.entries",
|
|
644
|
+
"ranks must be contiguous and ordered from one",
|
|
645
|
+
)
|
|
646
|
+
|
|
647
|
+
def to_dict(self) -> dict[str, Any]:
|
|
648
|
+
return {
|
|
649
|
+
"schema_version": PACKED_SEARCH_RESULT_SCHEMA,
|
|
650
|
+
"status": self.status,
|
|
651
|
+
"incomplete_code": self.incomplete_code,
|
|
652
|
+
"head_sha256": self.head_sha256,
|
|
653
|
+
"receipt_coordinate_sha256": self.receipt_coordinate_sha256,
|
|
654
|
+
"backend_coordinate": self.backend_coordinate,
|
|
655
|
+
"layers": list(EMBEDDING_LAYERS),
|
|
656
|
+
"question_digest": self.question_digest,
|
|
657
|
+
"limit": self.limit,
|
|
658
|
+
"range_manifest_sha256s": list(self.range_manifest_sha256s),
|
|
659
|
+
"vector_member_sha256s": list(self.vector_member_sha256s),
|
|
660
|
+
"posting_member_sha256s": list(self.posting_member_sha256s),
|
|
661
|
+
"bound_member_sha256s": list(self.bound_member_sha256s),
|
|
662
|
+
"entries": [item.to_dict() for item in self.entries],
|
|
663
|
+
"work_limits": self.limits.to_dict(),
|
|
664
|
+
"work_actuals": self.work.to_dict(),
|
|
665
|
+
}
|
|
666
|
+
|
|
667
|
+
@property
|
|
668
|
+
def identity_sha256(self) -> str:
|
|
669
|
+
return canonical_sha256(self.to_dict())
|
|
670
|
+
|
|
671
|
+
|
|
672
|
+
# ------------------------------------------------------------------------------------------
|
|
673
|
+
# Authentication
|
|
674
|
+
# ------------------------------------------------------------------------------------------
|
|
675
|
+
|
|
676
|
+
|
|
677
|
+
@dataclass(frozen=True)
|
|
678
|
+
class _Authenticated:
|
|
679
|
+
"""The head and the receipt, agreed with each other before a single range is opened."""
|
|
680
|
+
|
|
681
|
+
head: dict[str, Any]
|
|
682
|
+
head_sha256: str
|
|
683
|
+
receipt_coordinate_sha256: str
|
|
684
|
+
descriptors: tuple[PackedRangeDescriptor, ...]
|
|
685
|
+
backends: tuple[Mapping[str, Any], ...]
|
|
686
|
+
posting_backend_coordinate: str
|
|
687
|
+
vector_payload_sha256s: frozenset[str]
|
|
688
|
+
posting_member_sha256s: frozenset[str]
|
|
689
|
+
bound_member_sha256s: frozenset[str]
|
|
690
|
+
|
|
691
|
+
|
|
692
|
+
def _receipt_coordinate(receipt: Any) -> str:
|
|
693
|
+
"""Check the receipt against itself, before a byte of the generation has been read.
|
|
694
|
+
|
|
695
|
+
This half needs no head: a receipt that is not a complete bounded-scalar publication, or that
|
|
696
|
+
does not reproduce its own coordinate, is refused whatever is installed at the root.
|
|
697
|
+
"""
|
|
698
|
+
|
|
699
|
+
if not isinstance(receipt, Mapping) or receipt.get("schema_version") != PACKED_RECEIPT_SCHEMA:
|
|
700
|
+
raise _refuse("PACKED_RECEIPT_SCHEMA", "packed.receipt", "receipt contract differs")
|
|
701
|
+
if receipt.get("adapter_mode") != BOUNDED_SCALAR_ADAPTER:
|
|
702
|
+
raise _refuse(
|
|
703
|
+
"PACKED_RECEIPT_ADAPTER",
|
|
704
|
+
"packed.receipt.adapter_mode",
|
|
705
|
+
f"a packed generation may claim only the {BOUNDED_SCALAR_ADAPTER!r} adapter",
|
|
706
|
+
)
|
|
707
|
+
if receipt.get("status") != "complete":
|
|
708
|
+
raise _refuse(
|
|
709
|
+
"PACKED_RECEIPT_STATUS",
|
|
710
|
+
"packed.receipt.status",
|
|
711
|
+
"only a complete publication may be searched",
|
|
712
|
+
)
|
|
713
|
+
body = {key: value for key, value in receipt.items() if key != "coordinate_sha256"}
|
|
714
|
+
try:
|
|
715
|
+
coordinate = canonical_sha256(body)
|
|
716
|
+
except CanonicalJSONError as error:
|
|
717
|
+
raise _refuse(
|
|
718
|
+
"PACKED_RECEIPT_COORDINATE", "packed.receipt", "receipt is not canonical"
|
|
719
|
+
) from error
|
|
720
|
+
if coordinate != receipt.get("coordinate_sha256"):
|
|
721
|
+
raise _refuse(
|
|
722
|
+
"PACKED_RECEIPT_COORDINATE",
|
|
723
|
+
"packed.receipt.coordinate_sha256",
|
|
724
|
+
"the receipt does not reproduce its own coordinate",
|
|
725
|
+
)
|
|
726
|
+
return coordinate
|
|
727
|
+
|
|
728
|
+
|
|
729
|
+
def _authenticate(
|
|
730
|
+
root: _PackedRoot, *, expected_head_sha256: str, receipt: Mapping[str, Any], coordinate: str
|
|
731
|
+
) -> _Authenticated:
|
|
732
|
+
raw = root.read_document(PACKED_HEAD_FILENAME, maximum=MAX_PACKED_HEAD_BYTES)
|
|
733
|
+
if raw is None:
|
|
734
|
+
raise _refuse("PACKED_HEAD_MISSING", "packed.head", "no readable packed head at this root")
|
|
735
|
+
observed = sha256_bytes(raw)
|
|
736
|
+
if observed != expected_head_sha256:
|
|
737
|
+
raise _refuse(
|
|
738
|
+
"PACKED_HEAD_MISMATCH",
|
|
739
|
+
"packed.head",
|
|
740
|
+
"the installed head is not the caller's exact generation",
|
|
741
|
+
)
|
|
742
|
+
head = _parse_head(raw)
|
|
743
|
+
descriptors = tuple(
|
|
744
|
+
range_descriptor_from_dict(value, path=f"packed.head.ranges[{index}]")
|
|
745
|
+
for index, value in enumerate(head["ranges"])
|
|
746
|
+
)
|
|
747
|
+
if ranges_chain_sha256(descriptors) != head["ranges_chain_sha256"]:
|
|
748
|
+
raise _refuse(
|
|
749
|
+
"PACKED_HEAD_EXPECTED",
|
|
750
|
+
"packed.head.ranges_chain_sha256",
|
|
751
|
+
"the head's stated range chain is not the chain of its own descriptors",
|
|
752
|
+
)
|
|
753
|
+
for position, descriptor in enumerate(descriptors):
|
|
754
|
+
if descriptor.range_index != position:
|
|
755
|
+
raise _refuse(
|
|
756
|
+
"PACKED_RANGE_ORDER",
|
|
757
|
+
f"packed.head.ranges[{position}]",
|
|
758
|
+
"ranges are consecutive from zero",
|
|
759
|
+
)
|
|
760
|
+
return _bound_receipt(
|
|
761
|
+
receipt,
|
|
762
|
+
head=head,
|
|
763
|
+
head_sha256=observed,
|
|
764
|
+
descriptors=descriptors,
|
|
765
|
+
coordinate=coordinate,
|
|
766
|
+
)
|
|
767
|
+
|
|
768
|
+
|
|
769
|
+
def _parse_head(raw: bytes) -> dict[str, Any]:
|
|
770
|
+
"""Read the outer head exactly. The publisher's own laws, enforced by this reader."""
|
|
771
|
+
|
|
772
|
+
if len(raw) > MAX_PACKED_HEAD_BYTES:
|
|
773
|
+
raise _refuse("PACKED_HEAD_LIMIT", "packed.head", "head exceeds its byte bound")
|
|
774
|
+
try:
|
|
775
|
+
head = parse_canonical_json(raw)
|
|
776
|
+
except CanonicalJSONError as error:
|
|
777
|
+
raise _refuse("PACKED_HEAD_SCHEMA", "packed.head", "head is not canonical") from error
|
|
778
|
+
if (
|
|
779
|
+
not isinstance(head, dict)
|
|
780
|
+
or head.get("schema_version") != PACKED_HEAD_SCHEMA
|
|
781
|
+
or not isinstance(head.get("ranges"), list)
|
|
782
|
+
or not isinstance(head.get("backends"), list)
|
|
783
|
+
or not isinstance(head.get("counters"), dict)
|
|
784
|
+
or not isinstance(head.get("posting_backend_coordinate"), str)
|
|
785
|
+
):
|
|
786
|
+
raise _refuse("PACKED_HEAD_SCHEMA", "packed.head", "head contract differs")
|
|
787
|
+
body = {key: value for key, value in head.items() if key != "root_sha256"}
|
|
788
|
+
if canonical_sha256(body) != head.get("root_sha256"):
|
|
789
|
+
raise _refuse(
|
|
790
|
+
"PACKED_HEAD_DIGEST",
|
|
791
|
+
"packed.head.root_sha256",
|
|
792
|
+
"the head's own digest does not reproduce from the head it signs",
|
|
793
|
+
)
|
|
794
|
+
if not 1 <= len(head["ranges"]) <= MAX_RANGE_DESCRIPTORS:
|
|
795
|
+
raise _refuse(
|
|
796
|
+
"PACKED_HEAD_DESCRIPTORS",
|
|
797
|
+
"packed.head.ranges",
|
|
798
|
+
f"an outer head holds 1 to {MAX_RANGE_DESCRIPTORS} range descriptors",
|
|
799
|
+
)
|
|
800
|
+
# This is the reader an *installed* catalogue is read back through, which makes it the one
|
|
801
|
+
# place the coverage statement has to be required rather than merely carried. A head that
|
|
802
|
+
# omitted it, or stated one that is not a coverage, would otherwise be readable here while
|
|
803
|
+
# the publisher's own reader refused it -- and a catalogue that cannot say whether it holds
|
|
804
|
+
# the whole sweep is exactly the thing the member exists to prevent.
|
|
805
|
+
if not coverage_is_valid(head.get("coverage")):
|
|
806
|
+
raise _refuse(
|
|
807
|
+
"PACKED_HEAD_COVERAGE",
|
|
808
|
+
"packed.head.coverage",
|
|
809
|
+
"an installed generation states what it covers and what it drew from",
|
|
810
|
+
)
|
|
811
|
+
return head
|
|
812
|
+
|
|
813
|
+
|
|
814
|
+
def _bound_receipt(
|
|
815
|
+
receipt: Mapping[str, Any],
|
|
816
|
+
*,
|
|
817
|
+
head: dict[str, Any],
|
|
818
|
+
head_sha256: str,
|
|
819
|
+
descriptors: tuple[PackedRangeDescriptor, ...],
|
|
820
|
+
coordinate: str,
|
|
821
|
+
) -> _Authenticated:
|
|
822
|
+
outputs = receipt.get("outputs")
|
|
823
|
+
if not isinstance(outputs, Mapping) or not _RECEIPT_OUTPUT_MEMBERS <= set(outputs):
|
|
824
|
+
raise _refuse("PACKED_RECEIPT_SCHEMA", "packed.receipt.outputs", "outputs contract differs")
|
|
825
|
+
if (
|
|
826
|
+
outputs["head_sha256"] != head_sha256
|
|
827
|
+
or outputs["range_count"] != head["range_count"]
|
|
828
|
+
or outputs["entry_count"] != head["entry_count"]
|
|
829
|
+
or outputs["ranges"] != head["ranges"]
|
|
830
|
+
or outputs["ranges_chain_sha256"] != head["ranges_chain_sha256"]
|
|
831
|
+
or receipt.get("counters") != head["counters"]
|
|
832
|
+
or receipt.get("backends") != head["backends"]
|
|
833
|
+
or receipt.get("posting_backend_coordinate") != head["posting_backend_coordinate"]
|
|
834
|
+
or receipt.get("coverage") != head["coverage"]
|
|
835
|
+
):
|
|
836
|
+
raise _refuse(
|
|
837
|
+
"PACKED_RECEIPT_MISMATCH",
|
|
838
|
+
"packed.receipt.outputs",
|
|
839
|
+
"the receipt does not describe the generation installed at this root",
|
|
840
|
+
)
|
|
841
|
+
summaries: dict[str, frozenset[str]] = {}
|
|
842
|
+
backend_count = len(head["backends"])
|
|
843
|
+
range_count = len(descriptors)
|
|
844
|
+
expected = {
|
|
845
|
+
"vector_member_sha256s": (
|
|
846
|
+
range_count * len(EMBEDDING_LAYERS) * backend_count,
|
|
847
|
+
range_count * len(EMBEDDING_LAYERS) * backend_count,
|
|
848
|
+
),
|
|
849
|
+
"bound_member_sha256s": (range_count * backend_count, range_count * backend_count),
|
|
850
|
+
"posting_member_sha256s": (
|
|
851
|
+
range_count * MIN_POSTING_SEGMENTS,
|
|
852
|
+
range_count * MAX_POSTING_SEGMENTS,
|
|
853
|
+
),
|
|
854
|
+
}
|
|
855
|
+
for name, (low, high) in expected.items():
|
|
856
|
+
values = outputs[name]
|
|
857
|
+
if not isinstance(values, list) or any(not _is_digest(value) for value in values):
|
|
858
|
+
raise _refuse(
|
|
859
|
+
"PACKED_RECEIPT_SCHEMA",
|
|
860
|
+
f"packed.receipt.outputs.{name}",
|
|
861
|
+
"a receipt binds each summary by a lowercase SHA-256",
|
|
862
|
+
)
|
|
863
|
+
if not low <= len(values) <= high:
|
|
864
|
+
raise _refuse(
|
|
865
|
+
"PACKED_SEARCH_UNBOUND_MEMBER",
|
|
866
|
+
f"packed.receipt.outputs.{name}",
|
|
867
|
+
"the receipt does not bind one summary per published range, backend and layer",
|
|
868
|
+
)
|
|
869
|
+
summaries[name] = frozenset(values)
|
|
870
|
+
return _Authenticated(
|
|
871
|
+
head=head,
|
|
872
|
+
head_sha256=head_sha256,
|
|
873
|
+
receipt_coordinate_sha256=coordinate,
|
|
874
|
+
descriptors=descriptors,
|
|
875
|
+
backends=tuple(head["backends"]),
|
|
876
|
+
posting_backend_coordinate=str(head["posting_backend_coordinate"]),
|
|
877
|
+
vector_payload_sha256s=summaries["vector_member_sha256s"],
|
|
878
|
+
posting_member_sha256s=summaries["posting_member_sha256s"],
|
|
879
|
+
bound_member_sha256s=summaries["bound_member_sha256s"],
|
|
880
|
+
)
|
|
881
|
+
|
|
882
|
+
|
|
883
|
+
# ------------------------------------------------------------------------------------------
|
|
884
|
+
# One range, read back against the chain that binds it
|
|
885
|
+
# ------------------------------------------------------------------------------------------
|
|
886
|
+
|
|
887
|
+
|
|
888
|
+
@dataclass(frozen=True)
|
|
889
|
+
class _VerifiedRange:
|
|
890
|
+
descriptor: PackedRangeDescriptor
|
|
891
|
+
members: tuple[PackedMemberDescriptor, ...]
|
|
892
|
+
|
|
893
|
+
|
|
894
|
+
def _verified_range(
|
|
895
|
+
root: _PackedRoot, authenticated: _Authenticated, descriptor: PackedRangeDescriptor
|
|
896
|
+
) -> _VerifiedRange:
|
|
897
|
+
raw = root.read_member(
|
|
898
|
+
descriptor.manifest_sha256, kind="range", maximum=MAX_RANGE_MANIFEST_BYTES
|
|
899
|
+
)
|
|
900
|
+
if (
|
|
901
|
+
raw is None
|
|
902
|
+
or len(raw) != descriptor.manifest_bytes
|
|
903
|
+
or sha256_bytes(raw) != descriptor.manifest_sha256
|
|
904
|
+
):
|
|
905
|
+
raise _refuse(
|
|
906
|
+
"PACKED_RANGE_DIGEST",
|
|
907
|
+
"packed.range.manifest",
|
|
908
|
+
"the installed range manifest is not the manifest the head binds",
|
|
909
|
+
)
|
|
910
|
+
try:
|
|
911
|
+
manifest = parse_canonical_json(raw)
|
|
912
|
+
except CanonicalJSONError as error:
|
|
913
|
+
raise _refuse(
|
|
914
|
+
"PACKED_RANGE_MANIFEST", "packed.range.manifest", "manifest is not canonical"
|
|
915
|
+
) from error
|
|
916
|
+
if (
|
|
917
|
+
not isinstance(manifest, dict)
|
|
918
|
+
or set(manifest) != _RANGE_MANIFEST_MEMBERS
|
|
919
|
+
or manifest["schema_version"] != PACKED_RANGE_MANIFEST_SCHEMA
|
|
920
|
+
or not isinstance(manifest["members"], list)
|
|
921
|
+
):
|
|
922
|
+
raise _refuse(
|
|
923
|
+
"PACKED_RANGE_MANIFEST", "packed.range.manifest", "range manifest contract differs"
|
|
924
|
+
)
|
|
925
|
+
body = {key: value for key, value in manifest.items() if key != "root_sha256"}
|
|
926
|
+
if canonical_sha256(body) != manifest["root_sha256"]:
|
|
927
|
+
raise _refuse(
|
|
928
|
+
"PACKED_RANGE_MANIFEST", "packed.range.manifest", "manifest root digest differs"
|
|
929
|
+
)
|
|
930
|
+
if not 1 <= len(manifest["members"]) <= MAX_RANGE_MEMBER_DESCRIPTORS:
|
|
931
|
+
raise _refuse(
|
|
932
|
+
"PACKED_RANGE_MEMBERS",
|
|
933
|
+
"packed.range.members",
|
|
934
|
+
f"a range manifest holds 1 to {MAX_RANGE_MEMBER_DESCRIPTORS} member descriptors",
|
|
935
|
+
)
|
|
936
|
+
if (
|
|
937
|
+
manifest["backends"] != [backend["coordinate"] for backend in authenticated.backends]
|
|
938
|
+
or manifest["posting_backend_coordinate"] != authenticated.posting_backend_coordinate
|
|
939
|
+
):
|
|
940
|
+
raise _refuse(
|
|
941
|
+
"PACKED_RANGE_MANIFEST",
|
|
942
|
+
"packed.range.backends",
|
|
943
|
+
"a range names exactly the generation's backends and posting backend",
|
|
944
|
+
)
|
|
945
|
+
binding = range_binding_sha256(
|
|
946
|
+
range_index=manifest["range_index"],
|
|
947
|
+
first_key=manifest["first_key"],
|
|
948
|
+
last_key=manifest["last_key"],
|
|
949
|
+
entry_count=manifest["entry_count"],
|
|
950
|
+
facts_sha256=manifest["facts_sha256"],
|
|
951
|
+
)
|
|
952
|
+
if (
|
|
953
|
+
binding != manifest["range_binding_sha256"]
|
|
954
|
+
or manifest["range_index"] != descriptor.range_index
|
|
955
|
+
or manifest["first_key"] != descriptor.first_key
|
|
956
|
+
or manifest["last_key"] != descriptor.last_key
|
|
957
|
+
or manifest["entry_count"] != descriptor.entry_count
|
|
958
|
+
or manifest["facts_sha256"] != descriptor.facts_sha256
|
|
959
|
+
or binding != descriptor.range_binding_sha256
|
|
960
|
+
or len(manifest["members"]) != descriptor.member_count
|
|
961
|
+
):
|
|
962
|
+
raise _refuse(
|
|
963
|
+
"PACKED_RANGE_MANIFEST",
|
|
964
|
+
"packed.range.range_binding_sha256",
|
|
965
|
+
"the range manifest does not describe the range its head descriptor names",
|
|
966
|
+
)
|
|
967
|
+
members = tuple(
|
|
968
|
+
member_descriptor_from_dict(value, path=f"packed.range.members[{index}]")
|
|
969
|
+
for index, value in enumerate(manifest["members"])
|
|
970
|
+
)
|
|
971
|
+
for member in members:
|
|
972
|
+
if (
|
|
973
|
+
member.range_index != descriptor.range_index
|
|
974
|
+
or member.first_key != descriptor.first_key
|
|
975
|
+
or member.last_key != descriptor.last_key
|
|
976
|
+
or member.entry_count != descriptor.entry_count
|
|
977
|
+
or member.facts_sha256 != descriptor.facts_sha256
|
|
978
|
+
or member.range_binding_sha256 != binding
|
|
979
|
+
):
|
|
980
|
+
raise _refuse(
|
|
981
|
+
"PACKED_MEMBER_BINDING",
|
|
982
|
+
"packed.range.members[]",
|
|
983
|
+
"a member descriptor is bound to another range",
|
|
984
|
+
)
|
|
985
|
+
return _VerifiedRange(descriptor=descriptor, members=members)
|
|
986
|
+
|
|
987
|
+
|
|
988
|
+
def _one_member(members: Sequence[PackedMemberDescriptor], kind: str) -> PackedMemberDescriptor:
|
|
989
|
+
found = [member for member in members if member.kind == kind]
|
|
990
|
+
if len(found) != 1:
|
|
991
|
+
raise _refuse(
|
|
992
|
+
"PACKED_RANGE_MEMBERS",
|
|
993
|
+
"packed.range.members[]",
|
|
994
|
+
f"a range carries exactly one {kind} member",
|
|
995
|
+
)
|
|
996
|
+
return found[0]
|
|
997
|
+
|
|
998
|
+
|
|
999
|
+
def _member_bytes(root: _PackedRoot, member: PackedMemberDescriptor, *, maximum: int) -> bytes:
|
|
1000
|
+
raw = root.read_member(member.sha256, kind=member.kind, maximum=maximum)
|
|
1001
|
+
if raw is None or len(raw) != member.bytes or sha256_bytes(raw) != member.sha256:
|
|
1002
|
+
raise _refuse(
|
|
1003
|
+
"PACKED_MEMBER_DIGEST",
|
|
1004
|
+
"packed.range.members[]",
|
|
1005
|
+
"an installed member is not the member its descriptor binds",
|
|
1006
|
+
)
|
|
1007
|
+
return raw
|
|
1008
|
+
|
|
1009
|
+
|
|
1010
|
+
def _require_bound_summary(digest: str, bound: frozenset[str], *, path: str) -> None:
|
|
1011
|
+
if digest not in bound:
|
|
1012
|
+
raise _refuse(
|
|
1013
|
+
"PACKED_SEARCH_UNBOUND_MEMBER",
|
|
1014
|
+
path,
|
|
1015
|
+
"this summary is not one the authenticated receipt binds",
|
|
1016
|
+
)
|
|
1017
|
+
|
|
1018
|
+
|
|
1019
|
+
# ------------------------------------------------------------------------------------------
|
|
1020
|
+
# The facts member
|
|
1021
|
+
# ------------------------------------------------------------------------------------------
|
|
1022
|
+
|
|
1023
|
+
|
|
1024
|
+
@dataclass(frozen=True)
|
|
1025
|
+
class _FactsItem:
|
|
1026
|
+
entry_id: str
|
|
1027
|
+
semantic_facts_digest: str
|
|
1028
|
+
entry_sha256: str
|
|
1029
|
+
entry_json: str
|
|
1030
|
+
|
|
1031
|
+
|
|
1032
|
+
def _verified_facts(raw: bytes, descriptor: PackedRangeDescriptor) -> tuple[_FactsItem, ...]:
|
|
1033
|
+
"""Read one facts member under exactly the law ``packed_catalog`` publishes it by.
|
|
1034
|
+
|
|
1035
|
+
``packed_catalog._verify_facts_member`` enforces the same law and returns only the entry ids;
|
|
1036
|
+
this reader needs each entry's exact canonical text as well, and parsing an eight-mebibyte
|
|
1037
|
+
member twice per scanned range is a cost a provider-scale search cannot pay. The two are
|
|
1038
|
+
proved to agree, and to refuse the same corruptions, in ``test_catalog_packed_retrieval``.
|
|
1039
|
+
"""
|
|
1040
|
+
|
|
1041
|
+
try:
|
|
1042
|
+
payload = parse_canonical_json(raw)
|
|
1043
|
+
except CanonicalJSONError as error:
|
|
1044
|
+
raise _refuse(
|
|
1045
|
+
"PACKED_FACTS_MEMBER", "packed.range.facts", "facts member is not canonical"
|
|
1046
|
+
) from error
|
|
1047
|
+
if (
|
|
1048
|
+
not isinstance(payload, dict)
|
|
1049
|
+
or payload.get("schema_version") != PACKED_FACTS_SCHEMA
|
|
1050
|
+
or payload.get("range_index") != descriptor.range_index
|
|
1051
|
+
or payload.get("entry_count") != descriptor.entry_count
|
|
1052
|
+
or not isinstance(payload.get("entries"), list)
|
|
1053
|
+
or len(payload["entries"]) != descriptor.entry_count
|
|
1054
|
+
):
|
|
1055
|
+
raise _refuse("PACKED_FACTS_MEMBER", "packed.range.facts", "facts member contract differs")
|
|
1056
|
+
items: list[_FactsItem] = []
|
|
1057
|
+
keys: list[str] = []
|
|
1058
|
+
previous: bytes | None = None
|
|
1059
|
+
for position, item in enumerate(payload["entries"]):
|
|
1060
|
+
path = f"packed.range.facts.entries[{position}]"
|
|
1061
|
+
if not isinstance(item, dict) or set(item) != _FACTS_ENTRY_MEMBERS:
|
|
1062
|
+
raise _refuse("PACKED_FACTS_MEMBER", path, "a facts entry contract differs")
|
|
1063
|
+
if sha256_bytes(item["entry_json"].encode("utf-8")) != item["entry_sha256"]:
|
|
1064
|
+
raise _refuse("PACKED_FACTS_MEMBER", path, "an entry is not the entry it digests to")
|
|
1065
|
+
key = item["key"].encode("utf-8")
|
|
1066
|
+
if previous is not None and key <= previous:
|
|
1067
|
+
raise _refuse("PACKED_RANGE_ORDER", path, "a facts member ascends by key")
|
|
1068
|
+
previous = key
|
|
1069
|
+
keys.append(item["key"])
|
|
1070
|
+
items.append(
|
|
1071
|
+
_FactsItem(
|
|
1072
|
+
entry_id=item["entry_id"],
|
|
1073
|
+
semantic_facts_digest=item["semantic_facts_digest"],
|
|
1074
|
+
entry_sha256=item["entry_sha256"],
|
|
1075
|
+
entry_json=item["entry_json"],
|
|
1076
|
+
)
|
|
1077
|
+
)
|
|
1078
|
+
if keys[0] != descriptor.first_key or keys[-1] != descriptor.last_key:
|
|
1079
|
+
raise _refuse(
|
|
1080
|
+
"PACKED_RANGE_ORDER",
|
|
1081
|
+
"packed.range.facts",
|
|
1082
|
+
"the facts member does not span the interval its range claims",
|
|
1083
|
+
)
|
|
1084
|
+
return tuple(items)
|
|
1085
|
+
|
|
1086
|
+
|
|
1087
|
+
def _rebuilt_entry(item: _FactsItem) -> CatalogEntryV2:
|
|
1088
|
+
"""Rebuild one carried entry under the exact v2 contract, and check it is the one named.
|
|
1089
|
+
|
|
1090
|
+
The digest beside it in the facts member is the delta's *classification* digest -- the one
|
|
1091
|
+
``streaming_delta`` computes over identity, facts and provenance while deliberately excluding
|
|
1092
|
+
the version link, the observations and the vector instances. It is therefore not
|
|
1093
|
+
``CatalogEntryV2.semantic_facts_digest`` and is carried onward as the publisher stated it,
|
|
1094
|
+
rather than recomputed here under a definition this module does not own.
|
|
1095
|
+
"""
|
|
1096
|
+
|
|
1097
|
+
try:
|
|
1098
|
+
document = parse_canonical_json(item.entry_json.encode("utf-8"))
|
|
1099
|
+
except CanonicalJSONError as error:
|
|
1100
|
+
raise _refuse(
|
|
1101
|
+
"PACKED_SEARCH_FACTS_ENTRY",
|
|
1102
|
+
"packed.range.facts.entries[].entry_json",
|
|
1103
|
+
"a carried entry is not canonical",
|
|
1104
|
+
) from error
|
|
1105
|
+
if not isinstance(document, dict):
|
|
1106
|
+
raise _refuse(
|
|
1107
|
+
"PACKED_SEARCH_FACTS_ENTRY",
|
|
1108
|
+
"packed.range.facts.entries[].entry_json",
|
|
1109
|
+
"a carried entry is an object",
|
|
1110
|
+
)
|
|
1111
|
+
entry = catalog_entry_v2_from_dict(document)
|
|
1112
|
+
if entry.entry_id != item.entry_id:
|
|
1113
|
+
raise _refuse(
|
|
1114
|
+
"PACKED_SEARCH_FACTS_ENTRY",
|
|
1115
|
+
"packed.range.facts.entries[]",
|
|
1116
|
+
"a carried entry is not the entry its facts member names",
|
|
1117
|
+
)
|
|
1118
|
+
return entry
|
|
1119
|
+
|
|
1120
|
+
|
|
1121
|
+
# ------------------------------------------------------------------------------------------
|
|
1122
|
+
# Exact scoring: postings when they are this backend's, packs otherwise
|
|
1123
|
+
# ------------------------------------------------------------------------------------------
|
|
1124
|
+
|
|
1125
|
+
|
|
1126
|
+
def _scores_from_postings(
|
|
1127
|
+
root: _PackedRoot,
|
|
1128
|
+
verified: _VerifiedRange,
|
|
1129
|
+
authenticated: _Authenticated,
|
|
1130
|
+
*,
|
|
1131
|
+
query: Sequence[int],
|
|
1132
|
+
budget: _Budget,
|
|
1133
|
+
opened: list[str],
|
|
1134
|
+
) -> list[int]:
|
|
1135
|
+
postings = [member for member in verified.members if member.kind == "posting"]
|
|
1136
|
+
if not MIN_POSTING_SEGMENTS <= len(postings) <= MAX_POSTING_SEGMENTS:
|
|
1137
|
+
raise _refuse(
|
|
1138
|
+
"PACKED_POSTING_SEGMENTS",
|
|
1139
|
+
"packed.range.members[]",
|
|
1140
|
+
f"a range carries {MIN_POSTING_SEGMENTS} to {MAX_POSTING_SEGMENTS} posting segments",
|
|
1141
|
+
)
|
|
1142
|
+
entry_count = verified.descriptor.entry_count
|
|
1143
|
+
layers = len(EMBEDDING_LAYERS)
|
|
1144
|
+
accumulated = [[0] * layers for _ in range(entry_count)]
|
|
1145
|
+
previous: tuple[int, int, int, int] | None = None
|
|
1146
|
+
for expected_index, member in enumerate(postings):
|
|
1147
|
+
if member.segment_index != expected_index:
|
|
1148
|
+
raise _refuse(
|
|
1149
|
+
"PACKED_POSTING_SEGMENTS",
|
|
1150
|
+
"packed.range.members[]",
|
|
1151
|
+
"posting segments are published in ascending segment order",
|
|
1152
|
+
)
|
|
1153
|
+
_require_bound_summary(
|
|
1154
|
+
member.sha256,
|
|
1155
|
+
authenticated.posting_member_sha256s,
|
|
1156
|
+
path="packed.range.postings",
|
|
1157
|
+
)
|
|
1158
|
+
budget.spend_posting_terms(int(member.term_count or 0))
|
|
1159
|
+
raw = _member_bytes(root, member, maximum=MAX_POSTING_SEGMENT_BYTES)
|
|
1160
|
+
segment = parse_posting_segment(raw, descriptor=member)
|
|
1161
|
+
if member.term_count != len(segment.terms):
|
|
1162
|
+
raise _refuse(
|
|
1163
|
+
"PACKED_POSTING_TERM",
|
|
1164
|
+
"packed.range.members[]",
|
|
1165
|
+
"a posting descriptor's term count differs from its segment",
|
|
1166
|
+
)
|
|
1167
|
+
if previous is not None and segment.terms and segment.terms[0] <= previous:
|
|
1168
|
+
raise _refuse(
|
|
1169
|
+
"PACKED_POSTING_ORDER",
|
|
1170
|
+
"packed.range.members[]",
|
|
1171
|
+
"posting terms ascend across segment boundaries too",
|
|
1172
|
+
)
|
|
1173
|
+
for dimension, ordinal, layer_ordinal, value in segment.terms:
|
|
1174
|
+
weight = query[dimension]
|
|
1175
|
+
if weight:
|
|
1176
|
+
accumulated[ordinal][layer_ordinal] += weight * value
|
|
1177
|
+
if segment.terms:
|
|
1178
|
+
previous = segment.terms[-1]
|
|
1179
|
+
opened.append(member.sha256)
|
|
1180
|
+
return [max(row) for row in accumulated]
|
|
1181
|
+
|
|
1182
|
+
|
|
1183
|
+
def _scores_from_packs(
|
|
1184
|
+
root: _PackedRoot,
|
|
1185
|
+
verified: _VerifiedRange,
|
|
1186
|
+
authenticated: _Authenticated,
|
|
1187
|
+
*,
|
|
1188
|
+
query: Sequence[int],
|
|
1189
|
+
backend_coordinate: str,
|
|
1190
|
+
opened: list[str],
|
|
1191
|
+
) -> list[int]:
|
|
1192
|
+
entry_count = verified.descriptor.entry_count
|
|
1193
|
+
best: list[int | None] = [None] * entry_count
|
|
1194
|
+
seen: set[str] = set()
|
|
1195
|
+
for layer in EMBEDDING_LAYERS:
|
|
1196
|
+
member = _one_layer_member(verified.members, layer=layer, coordinate=backend_coordinate)
|
|
1197
|
+
raw = _member_bytes(root, member, maximum=MAX_VECTOR_MEMBER_BYTES)
|
|
1198
|
+
payload_sha256 = sha256_bytes(raw[MEMBER_HEADER_BYTES:])
|
|
1199
|
+
_require_bound_summary(
|
|
1200
|
+
payload_sha256,
|
|
1201
|
+
authenticated.vector_payload_sha256s,
|
|
1202
|
+
path="packed.range.vectors",
|
|
1203
|
+
)
|
|
1204
|
+
parsed = parse_vector_member(raw, descriptor=member)
|
|
1205
|
+
if parsed.entry_count != entry_count or layer in seen:
|
|
1206
|
+
raise _refuse(
|
|
1207
|
+
"PACKED_MEMBER_LAYER",
|
|
1208
|
+
"packed.range.members[]",
|
|
1209
|
+
"a backend carries each layer exactly once, over this range's own entries",
|
|
1210
|
+
)
|
|
1211
|
+
seen.add(layer)
|
|
1212
|
+
for ordinal, vector in enumerate(parsed.vectors):
|
|
1213
|
+
total = 0
|
|
1214
|
+
for weight, value in zip(query, vector, strict=True):
|
|
1215
|
+
total += weight * value
|
|
1216
|
+
current = best[ordinal]
|
|
1217
|
+
if current is None or total > current:
|
|
1218
|
+
best[ordinal] = total
|
|
1219
|
+
opened.append(payload_sha256)
|
|
1220
|
+
return [0 if value is None else value for value in best]
|
|
1221
|
+
|
|
1222
|
+
|
|
1223
|
+
def _one_layer_member(
|
|
1224
|
+
members: Sequence[PackedMemberDescriptor], *, layer: str, coordinate: str
|
|
1225
|
+
) -> PackedMemberDescriptor:
|
|
1226
|
+
found = [
|
|
1227
|
+
member
|
|
1228
|
+
for member in members
|
|
1229
|
+
if member.kind == "vector"
|
|
1230
|
+
and member.layer == layer
|
|
1231
|
+
and member.backend_coordinate == coordinate
|
|
1232
|
+
]
|
|
1233
|
+
if len(found) != 1:
|
|
1234
|
+
raise _refuse(
|
|
1235
|
+
"PACKED_MEMBER_LAYER",
|
|
1236
|
+
"packed.range.members[]",
|
|
1237
|
+
f"this range carries no single {layer!r} pack for the selected backend",
|
|
1238
|
+
)
|
|
1239
|
+
return found[0]
|
|
1240
|
+
|
|
1241
|
+
|
|
1242
|
+
# ------------------------------------------------------------------------------------------
|
|
1243
|
+
# The search
|
|
1244
|
+
# ------------------------------------------------------------------------------------------
|
|
1245
|
+
|
|
1246
|
+
|
|
1247
|
+
@dataclass(frozen=True)
|
|
1248
|
+
class _Candidate:
|
|
1249
|
+
"""One entry that scored above zero, ordered by score descending then entry id ascending."""
|
|
1250
|
+
|
|
1251
|
+
negated: int
|
|
1252
|
+
entry_id: str
|
|
1253
|
+
range_index: int
|
|
1254
|
+
item: _FactsItem
|
|
1255
|
+
|
|
1256
|
+
|
|
1257
|
+
def rank_packed_entries(
|
|
1258
|
+
*,
|
|
1259
|
+
root: Path,
|
|
1260
|
+
expected_head_sha256: str,
|
|
1261
|
+
receipt: Mapping[str, Any],
|
|
1262
|
+
question: str,
|
|
1263
|
+
backend: EmbeddingBackend,
|
|
1264
|
+
limit: int,
|
|
1265
|
+
work_limits: CatalogSearchWorkLimits,
|
|
1266
|
+
) -> PackedRetrievalResult:
|
|
1267
|
+
"""Answer one question from one authenticated packed generation, or say what it could not do."""
|
|
1268
|
+
|
|
1269
|
+
if not isinstance(root, Path):
|
|
1270
|
+
raise _refuse("PACKED_SEARCH_ROOT", "root", "must be a pathlib.Path")
|
|
1271
|
+
if not isinstance(work_limits, CatalogSearchWorkLimits):
|
|
1272
|
+
raise _refuse("PACKED_SEARCH_LIMIT", "work_limits", "must be a CatalogSearchWorkLimits")
|
|
1273
|
+
if type(limit) is not int or not 1 <= limit <= MAX_RETRIEVAL_LIMIT:
|
|
1274
|
+
raise _refuse(
|
|
1275
|
+
"RETRIEVAL_LIMIT", "limit", f"must be an integer in [1, {MAX_RETRIEVAL_LIMIT}]"
|
|
1276
|
+
)
|
|
1277
|
+
if not isinstance(question, str) or len(question) > MAX_QUERY_CHARS:
|
|
1278
|
+
raise _refuse(
|
|
1279
|
+
"RETRIEVAL_QUERY_LIMIT",
|
|
1280
|
+
"question",
|
|
1281
|
+
f"a question is a string of at most {MAX_QUERY_CHARS} characters",
|
|
1282
|
+
)
|
|
1283
|
+
|
|
1284
|
+
if not _is_digest(expected_head_sha256):
|
|
1285
|
+
raise _refuse("PACKED_HEAD_EXPECTED", "expected_head_sha256", "must be a lowercase SHA-256")
|
|
1286
|
+
receipt_coordinate = _receipt_coordinate(receipt)
|
|
1287
|
+
coordinate = backend.descriptor.coordinate
|
|
1288
|
+
|
|
1289
|
+
budget = _Budget(work_limits)
|
|
1290
|
+
question_digest = canonical_sha256({"question": question})
|
|
1291
|
+
reader = _PackedRoot(root, budget)
|
|
1292
|
+
manifests: list[str] = []
|
|
1293
|
+
vectors: list[str] = []
|
|
1294
|
+
postings: list[str] = []
|
|
1295
|
+
bounds: list[str] = []
|
|
1296
|
+
entries: tuple[PackedRankedEntry, ...] = ()
|
|
1297
|
+
incomplete: str | None = None
|
|
1298
|
+
try:
|
|
1299
|
+
with reader.opened() as opened_root:
|
|
1300
|
+
authenticated = _authenticate(
|
|
1301
|
+
opened_root,
|
|
1302
|
+
expected_head_sha256=expected_head_sha256,
|
|
1303
|
+
receipt=receipt,
|
|
1304
|
+
coordinate=receipt_coordinate,
|
|
1305
|
+
)
|
|
1306
|
+
coordinate = _selected_backend_coordinate(authenticated, backend)
|
|
1307
|
+
query = _encoded_query(backend, question, budget=budget)
|
|
1308
|
+
if any(query):
|
|
1309
|
+
entries = _search(
|
|
1310
|
+
opened_root,
|
|
1311
|
+
authenticated,
|
|
1312
|
+
query=query,
|
|
1313
|
+
coordinate=coordinate,
|
|
1314
|
+
limit=limit,
|
|
1315
|
+
budget=budget,
|
|
1316
|
+
manifests=manifests,
|
|
1317
|
+
vectors=vectors,
|
|
1318
|
+
postings=postings,
|
|
1319
|
+
bounds=bounds,
|
|
1320
|
+
)
|
|
1321
|
+
except _Exhausted as exhausted:
|
|
1322
|
+
incomplete = exhausted.budget
|
|
1323
|
+
entries = ()
|
|
1324
|
+
return PackedRetrievalResult(
|
|
1325
|
+
status="incomplete" if incomplete is not None else "complete",
|
|
1326
|
+
incomplete_code=incomplete,
|
|
1327
|
+
head_sha256=expected_head_sha256,
|
|
1328
|
+
receipt_coordinate_sha256=receipt_coordinate,
|
|
1329
|
+
backend_coordinate=coordinate,
|
|
1330
|
+
question_digest=question_digest,
|
|
1331
|
+
limit=limit,
|
|
1332
|
+
range_manifest_sha256s=tuple(manifests),
|
|
1333
|
+
vector_member_sha256s=tuple(vectors),
|
|
1334
|
+
posting_member_sha256s=tuple(postings),
|
|
1335
|
+
bound_member_sha256s=tuple(bounds),
|
|
1336
|
+
entries=entries,
|
|
1337
|
+
limits=work_limits,
|
|
1338
|
+
work=budget.work(),
|
|
1339
|
+
)
|
|
1340
|
+
|
|
1341
|
+
|
|
1342
|
+
def _selected_backend_coordinate(authenticated: _Authenticated, backend: EmbeddingBackend) -> str:
|
|
1343
|
+
descriptor = backend.descriptor
|
|
1344
|
+
coordinate = descriptor.coordinate
|
|
1345
|
+
for published in authenticated.backends:
|
|
1346
|
+
if published["coordinate"] != coordinate:
|
|
1347
|
+
continue
|
|
1348
|
+
if (
|
|
1349
|
+
published["dimensions"] != descriptor.dimensions
|
|
1350
|
+
or published["quantization"] != descriptor.quantization
|
|
1351
|
+
):
|
|
1352
|
+
raise _refuse(
|
|
1353
|
+
"PACKED_SEARCH_BACKEND",
|
|
1354
|
+
"backend",
|
|
1355
|
+
"the selected encoder is not the published backend of that coordinate",
|
|
1356
|
+
)
|
|
1357
|
+
return coordinate
|
|
1358
|
+
raise _refuse(
|
|
1359
|
+
"PACKED_SEARCH_BACKEND",
|
|
1360
|
+
"backend",
|
|
1361
|
+
"this generation publishes no packs for the selected backend coordinate",
|
|
1362
|
+
)
|
|
1363
|
+
|
|
1364
|
+
|
|
1365
|
+
def _encoded_query(backend: EmbeddingBackend, question: str, *, budget: _Budget) -> tuple[int, ...]:
|
|
1366
|
+
budget.spend_query_encode()
|
|
1367
|
+
asked = backend.encode(question)
|
|
1368
|
+
dimensions = backend.descriptor.dimensions
|
|
1369
|
+
if (
|
|
1370
|
+
not isinstance(asked, tuple)
|
|
1371
|
+
or len(asked) != dimensions
|
|
1372
|
+
or any(type(value) is not int for value in asked)
|
|
1373
|
+
):
|
|
1374
|
+
raise _refuse(
|
|
1375
|
+
"RETRIEVAL_QUERY_VECTOR",
|
|
1376
|
+
"question",
|
|
1377
|
+
"the selected backend returned a vector outside its exact integer coordinate",
|
|
1378
|
+
)
|
|
1379
|
+
return asked
|
|
1380
|
+
|
|
1381
|
+
|
|
1382
|
+
def _search(
|
|
1383
|
+
root: _PackedRoot,
|
|
1384
|
+
authenticated: _Authenticated,
|
|
1385
|
+
*,
|
|
1386
|
+
query: tuple[int, ...],
|
|
1387
|
+
coordinate: str,
|
|
1388
|
+
limit: int,
|
|
1389
|
+
budget: _Budget,
|
|
1390
|
+
manifests: list[str],
|
|
1391
|
+
vectors: list[str],
|
|
1392
|
+
postings: list[str],
|
|
1393
|
+
bounds: list[str],
|
|
1394
|
+
) -> tuple[PackedRankedEntry, ...]:
|
|
1395
|
+
"""One bound pass to order the work, one exact pass over whatever survives pruning."""
|
|
1396
|
+
|
|
1397
|
+
range_bounds: list[int] = []
|
|
1398
|
+
for descriptor in authenticated.descriptors:
|
|
1399
|
+
budget.check_time()
|
|
1400
|
+
budget.spend_range_bounded()
|
|
1401
|
+
verified = _verified_range(root, authenticated, descriptor)
|
|
1402
|
+
manifests.append(descriptor.manifest_sha256)
|
|
1403
|
+
member = _one_backend_member(verified.members, kind="bound", coordinate=coordinate)
|
|
1404
|
+
_require_bound_summary(
|
|
1405
|
+
member.sha256, authenticated.bound_member_sha256s, path="packed.range.bounds"
|
|
1406
|
+
)
|
|
1407
|
+
raw = _member_bytes(root, member, maximum=bound_segment_bytes(dimensions=len(query)))
|
|
1408
|
+
segment = parse_bound_segment(raw, descriptor=member)
|
|
1409
|
+
range_bounds.append(range_upper_bound(query, segment.bounds))
|
|
1410
|
+
bounds.append(member.sha256)
|
|
1411
|
+
|
|
1412
|
+
order = sorted(
|
|
1413
|
+
range(len(authenticated.descriptors)),
|
|
1414
|
+
key=lambda index: (-range_bounds[index], index),
|
|
1415
|
+
)
|
|
1416
|
+
selected: list[_Candidate] = []
|
|
1417
|
+
for index in order:
|
|
1418
|
+
if len(selected) == limit and range_bounds[index] < -selected[limit - 1].negated:
|
|
1419
|
+
budget.note_pruned()
|
|
1420
|
+
continue
|
|
1421
|
+
descriptor = authenticated.descriptors[index]
|
|
1422
|
+
budget.check_time()
|
|
1423
|
+
budget.spend_range_scanned()
|
|
1424
|
+
budget.spend_candidates(descriptor.entry_count)
|
|
1425
|
+
verified = _verified_range(root, authenticated, descriptor)
|
|
1426
|
+
facts = _one_member(verified.members, "facts")
|
|
1427
|
+
items = _verified_facts(
|
|
1428
|
+
_member_bytes(root, facts, maximum=MAX_FACTS_MEMBER_BYTES), descriptor
|
|
1429
|
+
)
|
|
1430
|
+
if coordinate == authenticated.posting_backend_coordinate:
|
|
1431
|
+
scores = _scores_from_postings(
|
|
1432
|
+
root, verified, authenticated, query=query, budget=budget, opened=postings
|
|
1433
|
+
)
|
|
1434
|
+
else:
|
|
1435
|
+
scores = _scores_from_packs(
|
|
1436
|
+
root,
|
|
1437
|
+
verified,
|
|
1438
|
+
authenticated,
|
|
1439
|
+
query=query,
|
|
1440
|
+
backend_coordinate=coordinate,
|
|
1441
|
+
opened=vectors,
|
|
1442
|
+
)
|
|
1443
|
+
if len(scores) != len(items):
|
|
1444
|
+
raise _refuse(
|
|
1445
|
+
"PACKED_SEARCH_FACTS_ENTRY",
|
|
1446
|
+
"packed.range.facts",
|
|
1447
|
+
"this range's packs and its facts member describe different entry counts",
|
|
1448
|
+
)
|
|
1449
|
+
found = [
|
|
1450
|
+
_Candidate(
|
|
1451
|
+
negated=-score,
|
|
1452
|
+
entry_id=item.entry_id,
|
|
1453
|
+
range_index=descriptor.range_index,
|
|
1454
|
+
item=item,
|
|
1455
|
+
)
|
|
1456
|
+
for score, item in zip(scores, items, strict=True)
|
|
1457
|
+
if score > 0
|
|
1458
|
+
]
|
|
1459
|
+
found.sort(key=lambda candidate: (candidate.negated, candidate.entry_id))
|
|
1460
|
+
selected = sorted(
|
|
1461
|
+
[*selected, *found[:limit]],
|
|
1462
|
+
key=lambda candidate: (candidate.negated, candidate.entry_id),
|
|
1463
|
+
)[:limit]
|
|
1464
|
+
return tuple(
|
|
1465
|
+
PackedRankedEntry(
|
|
1466
|
+
rank=rank,
|
|
1467
|
+
entry_id=candidate.entry_id,
|
|
1468
|
+
entry_coordinate=entry.coordinate,
|
|
1469
|
+
entry_digest=entry.digest,
|
|
1470
|
+
semantic_facts_digest=candidate.item.semantic_facts_digest,
|
|
1471
|
+
range_index=candidate.range_index,
|
|
1472
|
+
entry=entry,
|
|
1473
|
+
)
|
|
1474
|
+
for rank, (candidate, entry) in enumerate(
|
|
1475
|
+
((candidate, _rebuilt_entry(candidate.item)) for candidate in selected), start=1
|
|
1476
|
+
)
|
|
1477
|
+
)
|
|
1478
|
+
|
|
1479
|
+
|
|
1480
|
+
def _one_backend_member(
|
|
1481
|
+
members: Sequence[PackedMemberDescriptor], *, kind: str, coordinate: str
|
|
1482
|
+
) -> PackedMemberDescriptor:
|
|
1483
|
+
found = [
|
|
1484
|
+
member
|
|
1485
|
+
for member in members
|
|
1486
|
+
if member.kind == kind and member.backend_coordinate == coordinate
|
|
1487
|
+
]
|
|
1488
|
+
if len(found) != 1:
|
|
1489
|
+
raise _refuse(
|
|
1490
|
+
"PACKED_RANGE_MEMBERS",
|
|
1491
|
+
"packed.range.members[]",
|
|
1492
|
+
f"a range carries exactly one {kind} member per backend",
|
|
1493
|
+
)
|
|
1494
|
+
return found[0]
|
|
1495
|
+
|
|
1496
|
+
|
|
1497
|
+
__all__ = [
|
|
1498
|
+
"MAX_WORK_CANDIDATES",
|
|
1499
|
+
"MAX_WORK_ELAPSED_SECONDS",
|
|
1500
|
+
"MAX_WORK_MEMBER_BYTES",
|
|
1501
|
+
"MAX_WORK_OPEN_FILES",
|
|
1502
|
+
"MAX_WORK_PACKS",
|
|
1503
|
+
"MAX_WORK_POSTING_TERMS",
|
|
1504
|
+
"MAX_WORK_QUERY_ENCODES",
|
|
1505
|
+
"MAX_WORK_RANGES",
|
|
1506
|
+
"MIN_WORK_OPEN_FILES",
|
|
1507
|
+
"PACKED_INCOMPLETE_CODES",
|
|
1508
|
+
"PACKED_RECEIPT_SCHEMA",
|
|
1509
|
+
"PACKED_SEARCH_RESULT_SCHEMA",
|
|
1510
|
+
"PUBLIC_STORE_KINDS",
|
|
1511
|
+
"CatalogSearchWorkLimits",
|
|
1512
|
+
"PackedRankedEntry",
|
|
1513
|
+
"PackedRetrievalResult",
|
|
1514
|
+
"PackedSearchWork",
|
|
1515
|
+
"detect_public_store_kind",
|
|
1516
|
+
"rank_packed_entries",
|
|
1517
|
+
]
|