supertable 2.3.8__tar.gz → 2.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {supertable-2.3.8/supertable.egg-info → supertable-2.4.0}/PKG-INFO +1 -1
- {supertable-2.3.8 → supertable-2.4.0}/pyproject.toml +1 -1
- {supertable-2.3.8 → supertable-2.4.0}/setup.py +1 -1
- {supertable-2.3.8 → supertable-2.4.0}/supertable/__init__.py +1 -1
- {supertable-2.3.8 → supertable-2.4.0}/supertable/config/settings.py +31 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/data_writer.py +21 -2
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/data_estimator.py +205 -14
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/engine_common.py +45 -3
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/spark_thrift.py +59 -8
- {supertable-2.3.8 → supertable-2.4.0}/supertable/processing.py +164 -47
- {supertable-2.3.8 → supertable-2.4.0}/supertable/quality/checker.py +27 -2
- {supertable-2.3.8 → supertable-2.4.0}/supertable/quality/scheduler.py +7 -3
- supertable-2.4.0/supertable/tests/test_data_estimator_projection.py +201 -0
- supertable-2.4.0/supertable/tests/test_metadata_partitioning.py +221 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_processing_stats.py +75 -0
- supertable-2.4.0/supertable/tests/test_quality_checker.py +120 -0
- supertable-2.4.0/supertable/tests/test_spark_file_resolution.py +148 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_stats_cache.py +91 -1
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_stats_schema_snapshot.py +7 -0
- supertable-2.4.0/supertable/tests/test_tombstone_cache.py +162 -0
- supertable-2.4.0/supertable/utils/helper.py +63 -0
- supertable-2.4.0/supertable/utils/tests/test_hourly_partition.py +63 -0
- {supertable-2.3.8 → supertable-2.4.0/supertable.egg-info}/PKG-INFO +1 -1
- {supertable-2.3.8 → supertable-2.4.0}/supertable.egg-info/SOURCES.txt +6 -0
- supertable-2.3.8/supertable/utils/helper.py +0 -39
- {supertable-2.3.8 → supertable-2.4.0}/LICENSE +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/README.md +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/requirements.txt +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/setup.cfg +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/admin.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/chain.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/consumers.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/crypto.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/events.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/export.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/logger.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/middleware.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/reader.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/retention.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/tests/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/tests/test_chain.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/tests/test_crypto.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/tests/test_emit.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/tests/test_events.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/tests/test_retention.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/writer_parquet.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/writer_redis.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/config/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/config/defaults.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/config/homedir.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/config/tests/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/config/tests/test_defaults.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/config/tests/test_homedir.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/config/tests/test_settings.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/data_classes.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/data_reader.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/__main__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/check_filter_builder.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/controller.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/data_writer_helpers.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/defaults.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/dummy_data.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/read_parquet_header.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s01_01_01_create_super_table.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s01_01_02_enable_mirroring_formats.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s01_02_create_roles.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s01_03_create_users.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_01_write_dummy_data.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_02_write_single_data.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_03_01_write_staging.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_03_02_create_pipe.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_04_01_write_monitoring_simple.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_04_02_write_monitoring_parallel.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_05_write_tombstone.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_01_read_data_error.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_02_01_read_super_data_ok.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_02_02_read_table_data_ok.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_03_read_meta.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_04_read_staging.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_06_01_read_roles.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_06_02_read_user.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_07_01_estimate_read.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_07_02_estimate_files.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_08_read_snapshot_history.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s04_01_03_delete_pipe.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s05_01_delete_table.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s05_02_delete_super_table.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/webshop/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/webshop/core.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/webshop/defaults.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/webshop/generate.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/webshop/load.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/webshop/topup.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/duckdb_lite.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/duckdb_pro.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/engine_config.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/engine_enum.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/executor.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/plan_stats.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/tests/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/tests/conftest.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/tests/test_engine.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/tests/test_engine_config.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/tests/test_engine_routing.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/tests/test_engine_spill.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/errors.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/benchmarks/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/benchmarks/benchmark_locking.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/benchmarks/measure_lock_speed.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/benchmarks/measure_lock_time.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/file_lock.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/redis_lock.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/tests/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/tests/test_file_lock.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/tests/test_redis_lock.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/logging.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/meta_reader.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/mirroring/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/mirroring/mirror_delta.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/mirroring/mirror_formats.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/mirroring/mirror_iceberg.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/mirroring/mirror_parquet.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/monitoring/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/monitoring/partitions.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/monitoring_writer.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/plan_extender.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/quality/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/quality/anomaly.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/quality/config.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/quality/history.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/query_plan_manager.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/access_control.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/filter_builder.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/permissions.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/role_manager.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/row_column_security.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/tests/test_filter_builder.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/tests/test_rbac.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/tests/test_rbac_per_table.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/user_manager.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/redis_catalog.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/redis_connector.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/redis_infra.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/redis_keys.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/simple_table.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/staging_area.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/azure_storage.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/gcp_storage.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/local_storage.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/minio_storage.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/s3_storage.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/storage_factory.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/storage_interface.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/tests/test_storage.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/super_pipe.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/super_table.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/system_query.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_align_to_schema_fix.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_compaction_selection.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_create_if_missing.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_data_reader.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_data_reader_preflight.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_data_writer.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_data_writer_compact.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_data_writer_comprehensive.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_errors.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_meta_reader.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_monitoring_partitions.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_monitoring_sink_guard.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_newer_than.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_parquet_statistics.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_processing.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_processing_compact_resources.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_query_sql.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_read_pruning_differential.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_read_pruning_integration.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_redis_key_prefix.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_resolve_overwrite_writes.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_simple_table.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_stats_pruning.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_super_table.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_supertable_all.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_system_query.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_write_probe_gate.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/utils/__init__.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/utils/profiler.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/utils/sql_parser.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/utils/tests/test_sql_parser_columns.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable/utils/timer.py +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable.egg-info/dependency_links.txt +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable.egg-info/entry_points.txt +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable.egg-info/requires.txt +0 -0
- {supertable-2.3.8 → supertable-2.4.0}/supertable.egg-info/top_level.txt +0 -0
|
@@ -19,7 +19,7 @@ long_description = readme.read_text(encoding="utf-8") if readme.exists() else ""
|
|
|
19
19
|
|
|
20
20
|
setup(
|
|
21
21
|
name="supertable",
|
|
22
|
-
version="2.
|
|
22
|
+
version="2.4.0",
|
|
23
23
|
description="SuperTable — versioned data lake library for SQL analytics on Parquet + Redis.",
|
|
24
24
|
long_description=long_description,
|
|
25
25
|
long_description_content_type="text/markdown",
|
|
@@ -25,7 +25,7 @@ See the ``supertable.demo`` package for runnable end-to-end demos and the
|
|
|
25
25
|
project documentation for the full API surface.
|
|
26
26
|
"""
|
|
27
27
|
|
|
28
|
-
__version__ = "2.
|
|
28
|
+
__version__ = "2.4.0"
|
|
29
29
|
|
|
30
30
|
# Re-export the core public surface so users can do ``from supertable import …``
|
|
31
31
|
# instead of remembering submodule paths.
|
|
@@ -157,6 +157,14 @@ class Settings:
|
|
|
157
157
|
SUPERTABLE_DUCKDB_MATERIALIZE: str = "view" # SUPERTABLE_DUCKDB_MATERIALIZE
|
|
158
158
|
SUPERTABLE_DUCKDB_PRESIGNED: bool = False # SUPERTABLE_DUCKDB_PRESIGNED
|
|
159
159
|
SUPERTABLE_DUCKDB_USE_HTTPFS: bool = False # SUPERTABLE_DUCKDB_USE_HTTPFS
|
|
160
|
+
# Allow DuckDB to DOWNLOAD a missing extension (httpfs) from
|
|
161
|
+
# extensions.duckdb.org at query time. Default OFF: httpfs is baked into
|
|
162
|
+
# the image and seeded into the extension dir, and an implicit network
|
|
163
|
+
# install on an offline/firewalled node can hang for minutes instead of
|
|
164
|
+
# failing. When OFF, a genuinely missing extension raises immediately with
|
|
165
|
+
# an actionable message; set ON only on a networked box that must
|
|
166
|
+
# self-install.
|
|
167
|
+
SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD: bool = False # SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD
|
|
160
168
|
# Write-path overwrite/delete resolution via the DuckDB pushdown probe.
|
|
161
169
|
# Disabled by default: the polars fallback reads only the projected key
|
|
162
170
|
# columns through the storage SDK and needs no httpfs extension, so it works
|
|
@@ -196,6 +204,12 @@ class Settings:
|
|
|
196
204
|
SUPERTABLE_SPARK_STATEMENT_TIMEOUT: int = 120 # SUPERTABLE_SPARK_STATEMENT_TIMEOUT
|
|
197
205
|
SUPERTABLE_SPARK_CONNECT_TIMEOUT: int = 30 # SUPERTABLE_SPARK_CONNECT_TIMEOUT
|
|
198
206
|
SUPERTABLE_SPARK_BATCH_SIZE: int = 50 # SUPERTABLE_SPARK_BATCH_SIZE
|
|
207
|
+
# Spark file access. Default False → Spark scans direct s3a://bucket/key
|
|
208
|
+
# paths using its own fs.s3a.* credentials, independent of the DuckDB
|
|
209
|
+
# presign setting that shapes the shared reflection file list. True →
|
|
210
|
+
# Spark scans presigned http(s):// URLs minted per request (use when the
|
|
211
|
+
# cluster cannot reach the object store with its own credentials).
|
|
212
|
+
SUPERTABLE_SPARK_PRESIGNED: bool = False # SUPERTABLE_SPARK_PRESIGNED
|
|
199
213
|
|
|
200
214
|
# ── Redis ────────────────────────────────────────────────────────
|
|
201
215
|
SUPERTABLE_REDIS_URL: str = "" # SUPERTABLE_REDIS_URL
|
|
@@ -291,11 +305,24 @@ class Settings:
|
|
|
291
305
|
# always read fresh and never cached). 0 disables caching.
|
|
292
306
|
SUPERTABLE_STATS_CACHE_MAX_TABLES: int = 64 # SUPERTABLE_STATS_CACHE_MAX_TABLES
|
|
293
307
|
|
|
308
|
+
# Max number of tables whose *latest* tombstone (deletion-vector) artifact is
|
|
309
|
+
# held in the in-process tombstone cache (one DataFrame per table; older
|
|
310
|
+
# versions are always read fresh and never cached). Mirrors the stats cache
|
|
311
|
+
# so a process writing in a loop skips the carry-forward read. 0 disables.
|
|
312
|
+
SUPERTABLE_TOMBSTONE_CACHE_MAX_TABLES: int = 64 # SUPERTABLE_TOMBSTONE_CACHE_MAX_TABLES
|
|
313
|
+
|
|
294
314
|
# Read-path file pruning: when True the estimator uses the stats artifact to
|
|
295
315
|
# drop parquet files that provably can't satisfy a query's WHERE predicates.
|
|
296
316
|
# Conservative (never drops a file that could match); set False to disable.
|
|
297
317
|
SUPERTABLE_READ_PRUNING_ENABLED: bool = True # SUPERTABLE_READ_PRUNING_ENABLED
|
|
298
318
|
|
|
319
|
+
# Projection-aware read sizing: when True the estimator charges a query only
|
|
320
|
+
# the on-disk (compressed) bytes of the columns it selects — via per-column
|
|
321
|
+
# ``compressed_bytes`` in the stats artifact, falling back to a type-width
|
|
322
|
+
# ratio of the whole file. Drives AUTO engine routing (lite/pro/spark). Set
|
|
323
|
+
# False to fall back to whole-file sizing (every query charged all columns).
|
|
324
|
+
SUPERTABLE_READ_PROJECTION_SIZING_ENABLED: bool = True
|
|
325
|
+
|
|
299
326
|
# ── Audit ────────────────────────────────────────────────────────
|
|
300
327
|
# Audit is OFF by default. Enable per-organization in the WebUI
|
|
301
328
|
# /ui/audit → Compliance tab (persisted at supertable:{org}:system:audit:config),
|
|
@@ -444,6 +471,7 @@ def _build_settings() -> Settings:
|
|
|
444
471
|
SUPERTABLE_DUCKDB_MATERIALIZE=_env_str("SUPERTABLE_DUCKDB_MATERIALIZE", "view"),
|
|
445
472
|
SUPERTABLE_DUCKDB_PRESIGNED=_env_bool("SUPERTABLE_DUCKDB_PRESIGNED", False),
|
|
446
473
|
SUPERTABLE_DUCKDB_USE_HTTPFS=_env_bool("SUPERTABLE_DUCKDB_USE_HTTPFS", False),
|
|
474
|
+
SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD=_env_bool("SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD", False),
|
|
447
475
|
SUPERTABLE_DUCKDB_WRITE_PROBE=_env_bool("SUPERTABLE_DUCKDB_WRITE_PROBE", False),
|
|
448
476
|
SUPERTABLE_DUCKDB_TOMBSTONE_CACHE_MAX_PER_TABLE=_env_int("SUPERTABLE_DUCKDB_TOMBSTONE_CACHE_MAX_PER_TABLE", 8),
|
|
449
477
|
SUPERTABLE_DUCKDB_TOMBSTONE_CACHE_TTL_SEC=_env_int("SUPERTABLE_DUCKDB_TOMBSTONE_CACHE_TTL_SEC", 300),
|
|
@@ -460,6 +488,7 @@ def _build_settings() -> Settings:
|
|
|
460
488
|
SUPERTABLE_SPARK_STATEMENT_TIMEOUT=_env_int("SUPERTABLE_SPARK_STATEMENT_TIMEOUT", 120),
|
|
461
489
|
SUPERTABLE_SPARK_CONNECT_TIMEOUT=_env_int("SUPERTABLE_SPARK_CONNECT_TIMEOUT", 30),
|
|
462
490
|
SUPERTABLE_SPARK_BATCH_SIZE=_env_int("SUPERTABLE_SPARK_BATCH_SIZE", 50),
|
|
491
|
+
SUPERTABLE_SPARK_PRESIGNED=_env_bool("SUPERTABLE_SPARK_PRESIGNED", False),
|
|
463
492
|
|
|
464
493
|
# ── Redis ────────────────────────────────────────────────────
|
|
465
494
|
SUPERTABLE_REDIS_URL=_env_str("SUPERTABLE_REDIS_URL"),
|
|
@@ -551,7 +580,9 @@ def _build_settings() -> Settings:
|
|
|
551
580
|
# ── Meta Reader / Caching ────────────────────────────────────
|
|
552
581
|
SUPERTABLE_SUPER_META_CACHE_TTL_S=meta_ttl,
|
|
553
582
|
SUPERTABLE_STATS_CACHE_MAX_TABLES=_env_int("SUPERTABLE_STATS_CACHE_MAX_TABLES", 64),
|
|
583
|
+
SUPERTABLE_TOMBSTONE_CACHE_MAX_TABLES=_env_int("SUPERTABLE_TOMBSTONE_CACHE_MAX_TABLES", 64),
|
|
554
584
|
SUPERTABLE_READ_PRUNING_ENABLED=_env_bool("SUPERTABLE_READ_PRUNING_ENABLED", True),
|
|
585
|
+
SUPERTABLE_READ_PROJECTION_SIZING_ENABLED=_env_bool("SUPERTABLE_READ_PROJECTION_SIZING_ENABLED", True),
|
|
555
586
|
|
|
556
587
|
# ── Audit ────────────────────────────────────────────────────
|
|
557
588
|
SUPERTABLE_AUDIT_ENABLED=_env_bool("SUPERTABLE_AUDIT_ENABLED", False),
|
|
@@ -34,6 +34,8 @@ from supertable.processing import (
|
|
|
34
34
|
prune_overlapping_files_by_stats,
|
|
35
35
|
load_stats,
|
|
36
36
|
cache_stats,
|
|
37
|
+
load_tombstone,
|
|
38
|
+
cache_tombstone,
|
|
37
39
|
write_parquet_and_collect_resources,
|
|
38
40
|
compact_resources,
|
|
39
41
|
compact_tombstones,
|
|
@@ -525,8 +527,13 @@ class DataWriter:
|
|
|
525
527
|
# required=True: a DV that exists but cannot be read must abort
|
|
526
528
|
# the write, never be treated as empty — silently dropping the
|
|
527
529
|
# carried-forward vector would resurrect previously deleted rows.
|
|
530
|
+
# Cache-first: if this process wrote the table on a prior loop
|
|
531
|
+
# iteration, the current deletion-vector is already in memory
|
|
532
|
+
# (seeded below after the pointer is pinned), so this is a hit
|
|
533
|
+
# with no storage round-trip. required=True preserves the
|
|
534
|
+
# carry-forward safety on a genuine miss (abort, never truncate).
|
|
528
535
|
prev_dv_df = (
|
|
529
|
-
|
|
536
|
+
load_tombstone(prev_tombstone_path, allow_cache=True, required=True, profiler=profiler)
|
|
530
537
|
if prev_tombstone_path else None
|
|
531
538
|
)
|
|
532
539
|
# The rowid set is consumed only by the idempotency filter below,
|
|
@@ -790,6 +797,15 @@ class DataWriter:
|
|
|
790
797
|
# and its row count.
|
|
791
798
|
last_simple_table["tombstone"] = tombstone_path
|
|
792
799
|
last_simple_table["tombstone_rows"] = tombstone_rows
|
|
800
|
+
# Seed the in-process cache so the NEXT write's carry-forward read
|
|
801
|
+
# (prev_dv_df above) is a pure memory hit. The frame is the fresh
|
|
802
|
+
# build/reclaim result when this write changed the vector, else the
|
|
803
|
+
# already-loaded prev frame for a pure carry-forward. No-op when
|
|
804
|
+
# the vector was fully consumed this write (tombstone_path is None).
|
|
805
|
+
cache_tombstone(
|
|
806
|
+
tombstone_path,
|
|
807
|
+
combined_tombstone_df if combined_tombstone_df is not None else prev_dv_df,
|
|
808
|
+
)
|
|
793
809
|
mark("compact_tombstones")
|
|
794
810
|
|
|
795
811
|
# Phase B — auto small-file compaction. Merge the accumulated
|
|
@@ -1328,7 +1344,10 @@ class DataWriter:
|
|
|
1328
1344
|
# new file while the vector kept pointing at the sunset __file__ —
|
|
1329
1345
|
# leaving them permanently unreclaimable. Failing loud leaves the
|
|
1330
1346
|
# prior snapshot + vector intact for a retry, and matches the
|
|
1331
|
-
# write-path carry-forward read (required=True) above.
|
|
1347
|
+
# write-path carry-forward read (required=True) above. Read directly
|
|
1348
|
+
# (not via the in-process cache): compact() always drains the vector,
|
|
1349
|
+
# so there is no carry-forward hit to gain and it never re-seeds the
|
|
1350
|
+
# cache after draining — the loop-caching win lives on the write path.
|
|
1332
1351
|
tombstone_df = (
|
|
1333
1352
|
_read_parquet_safe(tombstone_path, required=True)
|
|
1334
1353
|
if tombstone_path else None
|
|
@@ -7,6 +7,8 @@ from collections import defaultdict
|
|
|
7
7
|
from typing import Iterable, Set, List, Dict, Optional, Tuple
|
|
8
8
|
from urllib.parse import urlparse
|
|
9
9
|
|
|
10
|
+
import polars
|
|
11
|
+
|
|
10
12
|
from supertable.config.defaults import logger
|
|
11
13
|
from supertable.config.settings import settings
|
|
12
14
|
from supertable.data_classes import Reflection, SuperSnapshot
|
|
@@ -18,7 +20,12 @@ from supertable.utils.profiler import Profiler
|
|
|
18
20
|
from supertable.redis_catalog import RedisCatalog # Redis leaf pointers for snapshots
|
|
19
21
|
|
|
20
22
|
from supertable.utils.sql_parser import TableDefinition
|
|
21
|
-
from supertable.processing import
|
|
23
|
+
from supertable.processing import (
|
|
24
|
+
load_stats,
|
|
25
|
+
prune_files_by_predicates,
|
|
26
|
+
ROWID_COL,
|
|
27
|
+
TIMESTAMP_COL,
|
|
28
|
+
)
|
|
22
29
|
|
|
23
30
|
|
|
24
31
|
from typing import Dict, List, Optional, Set, Tuple
|
|
@@ -306,22 +313,25 @@ class DataEstimator:
|
|
|
306
313
|
super_name: str,
|
|
307
314
|
simple_name: str,
|
|
308
315
|
raw_keys: List[str],
|
|
309
|
-
|
|
316
|
+
stats_df: Optional["polars.DataFrame"],
|
|
310
317
|
profiler: Optional[Profiler] = None,
|
|
311
318
|
) -> List[str]:
|
|
312
319
|
"""Narrow *raw_keys* to those that could satisfy the query predicates.
|
|
313
320
|
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
321
|
+
Takes the already-loaded *stats_df* (loaded once per table by
|
|
322
|
+
:meth:`estimate` and shared with projection sizing) rather than a path,
|
|
323
|
+
so the artifact is read at most once. Returns *raw_keys* unchanged
|
|
324
|
+
whenever pruning is disabled, there's no stats artifact, or the query
|
|
325
|
+
carries no usable constraint for this table — and never raises (a
|
|
326
|
+
pruning failure must not break a read).
|
|
317
327
|
|
|
318
328
|
*profiler*, when supplied, accumulates the same IO/pruning counters the
|
|
319
|
-
write path emits (``
|
|
320
|
-
|
|
329
|
+
write path emits (``read_pruned_files``) so the read monitoring payload
|
|
330
|
+
can surface them.
|
|
321
331
|
"""
|
|
322
332
|
if not settings.SUPERTABLE_READ_PRUNING_ENABLED:
|
|
323
333
|
return raw_keys
|
|
324
|
-
if
|
|
334
|
+
if stats_df is None or not raw_keys:
|
|
325
335
|
return raw_keys
|
|
326
336
|
occurrences = self.predicate_constraints.get(
|
|
327
337
|
(super_name.lower(), simple_name.lower())
|
|
@@ -329,7 +339,6 @@ class DataEstimator:
|
|
|
329
339
|
if not occurrences:
|
|
330
340
|
return raw_keys
|
|
331
341
|
try:
|
|
332
|
-
stats_df = load_stats(stats_file, allow_cache=True, profiler=profiler)
|
|
333
342
|
return prune_files_by_predicates(
|
|
334
343
|
raw_keys, stats_df, occurrences, profiler=profiler,
|
|
335
344
|
)
|
|
@@ -337,6 +346,138 @@ class DataEstimator:
|
|
|
337
346
|
logger.warning(f"[estimate.prune] pruning skipped for {super_name}.{simple_name}: {e}")
|
|
338
347
|
return raw_keys
|
|
339
348
|
|
|
349
|
+
# ----------------------- projection-aware sizing -----------------------
|
|
350
|
+
|
|
351
|
+
# Rough per-value byte widths for the type-width *fallback* (used only when
|
|
352
|
+
# per-column ``compressed_bytes`` are unavailable — e.g. a stats file that
|
|
353
|
+
# predates the column, or no stats artifact at all). Substring-matched
|
|
354
|
+
# against the stored (DuckDB/polars) type name, most-specific first.
|
|
355
|
+
_STRING_AVG_WIDTH = 16
|
|
356
|
+
|
|
357
|
+
def _type_width(self, type_name: Optional[str]) -> int:
|
|
358
|
+
"""Estimated on-disk bytes-per-value for a column type (fallback only)."""
|
|
359
|
+
t = (type_name or "").strip().lower()
|
|
360
|
+
if not t:
|
|
361
|
+
return 8
|
|
362
|
+
if "bool" in t:
|
|
363
|
+
return 1
|
|
364
|
+
if "timestamp" in t or "datetime" in t:
|
|
365
|
+
return 8
|
|
366
|
+
if "date" in t:
|
|
367
|
+
return 4
|
|
368
|
+
if any(s in t for s in ("varchar", "char", "utf8", "string", "text",
|
|
369
|
+
"json", "blob", "binary", "bytea")):
|
|
370
|
+
return self._STRING_AVG_WIDTH
|
|
371
|
+
if "double" in t or "float64" in t:
|
|
372
|
+
return 8
|
|
373
|
+
if "float" in t or "real" in t:
|
|
374
|
+
return 4
|
|
375
|
+
if "decimal" in t or "numeric" in t:
|
|
376
|
+
return 8
|
|
377
|
+
if "bigint" in t or "int64" in t or "long" in t:
|
|
378
|
+
return 8
|
|
379
|
+
if "smallint" in t or "int16" in t:
|
|
380
|
+
return 2
|
|
381
|
+
if "tinyint" in t or "int8" in t:
|
|
382
|
+
return 1
|
|
383
|
+
if "int" in t:
|
|
384
|
+
return 4
|
|
385
|
+
return 8
|
|
386
|
+
|
|
387
|
+
def _selected_columns(self, super_name: str, simple_name: str) -> Optional[Set[str]]:
|
|
388
|
+
"""Lowercased set of user columns this query projects from the table.
|
|
389
|
+
|
|
390
|
+
Returns ``None`` to mean "every column" — i.e. ``SELECT *`` / ``t.*``
|
|
391
|
+
(an empty ``TableDefinition.columns``), an unrecognised table, or a
|
|
392
|
+
projection that resolves to only system columns. In every ``None`` case
|
|
393
|
+
the caller uses the whole-file size (no projection savings), which is
|
|
394
|
+
the safe over-estimate. System columns (``__rowid__`` / ``__timestamp__``)
|
|
395
|
+
are stripped: the read view hides them, so they're never scanned.
|
|
396
|
+
"""
|
|
397
|
+
key = (super_name.lower(), simple_name.lower())
|
|
398
|
+
selected: Set[str] = set()
|
|
399
|
+
matched = False
|
|
400
|
+
for t in self.tables:
|
|
401
|
+
if (t.super_name.lower(), t.simple_name.lower()) != key:
|
|
402
|
+
continue
|
|
403
|
+
matched = True
|
|
404
|
+
if not t.columns: # [] => SELECT * / t.* => whole table
|
|
405
|
+
return None
|
|
406
|
+
for c in t.columns:
|
|
407
|
+
cl = c.lower()
|
|
408
|
+
if cl not in (ROWID_COL, TIMESTAMP_COL):
|
|
409
|
+
selected.add(cl)
|
|
410
|
+
if not matched or not selected:
|
|
411
|
+
return None
|
|
412
|
+
return selected
|
|
413
|
+
|
|
414
|
+
def _projected_bytes_index(
|
|
415
|
+
self,
|
|
416
|
+
stats_df: Optional["polars.DataFrame"],
|
|
417
|
+
selected_cols: Set[str],
|
|
418
|
+
) -> Tuple[Set[str], Dict[str, int]]:
|
|
419
|
+
"""Sum per-column ``compressed_bytes`` for the selected columns per file.
|
|
420
|
+
|
|
421
|
+
Returns ``(tier3_files, proj)`` where *proj* maps a file path to the
|
|
422
|
+
summed on-disk bytes of its selected columns, and *tier3_files* is the
|
|
423
|
+
subset of files for which that sum is *trustworthy* — every matched row
|
|
424
|
+
carried a non-NULL ``compressed_bytes``. A file with any NULL (an older
|
|
425
|
+
carried-forward row) is omitted so the caller falls back to a whole-file
|
|
426
|
+
ratio rather than under-counting it as zero.
|
|
427
|
+
"""
|
|
428
|
+
if (
|
|
429
|
+
stats_df is None
|
|
430
|
+
or stats_df.height == 0
|
|
431
|
+
or "compressed_bytes" not in stats_df.columns
|
|
432
|
+
):
|
|
433
|
+
return set(), {}
|
|
434
|
+
sel = (
|
|
435
|
+
stats_df.select(["file_path", "column_name", "compressed_bytes"])
|
|
436
|
+
.with_columns(polars.col("column_name").str.to_lowercase().alias("__cn"))
|
|
437
|
+
.filter(polars.col("__cn").is_in(list(selected_cols)))
|
|
438
|
+
)
|
|
439
|
+
if sel.height == 0:
|
|
440
|
+
return set(), {}
|
|
441
|
+
agg = sel.group_by("file_path").agg(
|
|
442
|
+
[
|
|
443
|
+
polars.col("compressed_bytes").sum().alias("__b"),
|
|
444
|
+
polars.col("compressed_bytes").is_null().sum().alias("__nulls"),
|
|
445
|
+
]
|
|
446
|
+
)
|
|
447
|
+
tier3_files: Set[str] = set()
|
|
448
|
+
proj: Dict[str, int] = {}
|
|
449
|
+
for r in agg.iter_rows(named=True):
|
|
450
|
+
if int(r["__nulls"] or 0) == 0:
|
|
451
|
+
fp = r["file_path"]
|
|
452
|
+
tier3_files.add(fp)
|
|
453
|
+
proj[fp] = int(r["__b"] or 0)
|
|
454
|
+
return tier3_files, proj
|
|
455
|
+
|
|
456
|
+
def _ratio_bytes(
|
|
457
|
+
self,
|
|
458
|
+
file_key: str,
|
|
459
|
+
key_size: Dict[str, int],
|
|
460
|
+
selected_cols: Set[str],
|
|
461
|
+
schema_types: Dict[str, str],
|
|
462
|
+
) -> int:
|
|
463
|
+
"""Fallback size: scale the whole-file bytes by the selected columns'
|
|
464
|
+
type-width share of the table schema. Used only when precise
|
|
465
|
+
per-column ``compressed_bytes`` are unavailable for *file_key*."""
|
|
466
|
+
full = int(key_size.get(file_key, 0))
|
|
467
|
+
if full <= 0 or not schema_types:
|
|
468
|
+
return full
|
|
469
|
+
all_cols = {
|
|
470
|
+
c: ty for c, ty in schema_types.items()
|
|
471
|
+
if c not in (ROWID_COL, TIMESTAMP_COL)
|
|
472
|
+
}
|
|
473
|
+
total_w = sum(self._type_width(ty) for ty in all_cols.values())
|
|
474
|
+
if total_w <= 0:
|
|
475
|
+
return full
|
|
476
|
+
sel_w = sum(self._type_width(all_cols[c]) for c in selected_cols if c in all_cols)
|
|
477
|
+
if sel_w <= 0:
|
|
478
|
+
return full
|
|
479
|
+
return int(full * sel_w / total_w)
|
|
480
|
+
|
|
340
481
|
# ----------------------- main API -----------------------
|
|
341
482
|
def estimate(self) -> Reflection:
|
|
342
483
|
"""
|
|
@@ -352,7 +493,8 @@ class DataEstimator:
|
|
|
352
493
|
prune_profiler = Profiler()
|
|
353
494
|
|
|
354
495
|
supers: List[SuperSnapshot] = []
|
|
355
|
-
reflection_file_size = 0
|
|
496
|
+
reflection_file_size = 0 # projected (selected-column) bytes — routing
|
|
497
|
+
reflection_file_size_raw = 0 # whole-file bytes — physical footprint
|
|
356
498
|
max_freshness_ms = 0
|
|
357
499
|
files_before_prune = 0
|
|
358
500
|
files_pruned = 0
|
|
@@ -378,6 +520,7 @@ class DataEstimator:
|
|
|
378
520
|
)
|
|
379
521
|
|
|
380
522
|
schema: Set[str] = set()
|
|
523
|
+
schema_types: Dict[str, str] = {}
|
|
381
524
|
raw_keys: List[str] = []
|
|
382
525
|
key_size: Dict[str, int] = {}
|
|
383
526
|
stats_file: Optional[str] = None
|
|
@@ -395,7 +538,12 @@ class DataEstimator:
|
|
|
395
538
|
|
|
396
539
|
current_version = current_snapshot_data.get("snapshot_version", 0)
|
|
397
540
|
current_schema = self._schema_to_dict(current_snapshot_data.get("schema", {}))
|
|
398
|
-
|
|
541
|
+
lowered_schema = dict_keys_to_lowercase(current_schema)
|
|
542
|
+
schema.update(lowered_schema.keys())
|
|
543
|
+
# Retain name->type for the projection ratio fallback. First
|
|
544
|
+
# writer wins for a given column (schemas are stable per table).
|
|
545
|
+
for _cname, _ctype in lowered_schema.items():
|
|
546
|
+
schema_types.setdefault(_cname, _ctype)
|
|
399
547
|
sf = current_snapshot_data.get("stats_file")
|
|
400
548
|
if sf:
|
|
401
549
|
stats_file = sf
|
|
@@ -408,23 +556,62 @@ class DataEstimator:
|
|
|
408
556
|
raw_keys.append(file_key)
|
|
409
557
|
key_size[file_key] = int(resource.get("file_size", 0))
|
|
410
558
|
|
|
559
|
+
# Which columns does the query actually read? None => SELECT *
|
|
560
|
+
# (whole table, no projection savings).
|
|
561
|
+
selected_cols = self._selected_columns(super_name, simple_name)
|
|
562
|
+
need_projection = (
|
|
563
|
+
selected_cols is not None
|
|
564
|
+
and settings.SUPERTABLE_READ_PROJECTION_SIZING_ENABLED
|
|
565
|
+
)
|
|
566
|
+
has_predicate = bool(
|
|
567
|
+
self.predicate_constraints.get((super_name.lower(), simple_name.lower()))
|
|
568
|
+
)
|
|
569
|
+
|
|
570
|
+
# Load the stats artifact ONCE per table and reuse it for BOTH
|
|
571
|
+
# predicate pruning and projection sizing (cache-backed: a repeated
|
|
572
|
+
# read of the same table is a memory hit). Only load when at least
|
|
573
|
+
# one consumer needs it, so a SELECT * with no WHERE stays free.
|
|
574
|
+
stats_df: Optional["polars.DataFrame"] = None
|
|
575
|
+
if stats_file and (need_projection or has_predicate):
|
|
576
|
+
stats_df = load_stats(stats_file, allow_cache=True, profiler=prune_profiler)
|
|
577
|
+
|
|
411
578
|
# Read-path pruning: drop raw keys whose stats prove they cannot
|
|
412
579
|
# satisfy the query's WHERE before resolving them to scan URLs.
|
|
413
580
|
# The span accumulates the wall-clock of the whole pruning step
|
|
414
|
-
# (
|
|
581
|
+
# (predicate eval) across every table in the query.
|
|
415
582
|
with prune_profiler.span("read.prune"):
|
|
416
583
|
survivors = self._prune_files(
|
|
417
|
-
super_name, simple_name, raw_keys,
|
|
584
|
+
super_name, simple_name, raw_keys, stats_df,
|
|
418
585
|
profiler=prune_profiler,
|
|
419
586
|
)
|
|
420
587
|
files_before_prune += len(raw_keys)
|
|
421
588
|
files_pruned += len(raw_keys) - len(survivors)
|
|
422
589
|
files_kept += len(survivors)
|
|
423
590
|
|
|
591
|
+
# Projection-aware size: a query selecting specific columns scans
|
|
592
|
+
# only those columns' on-disk (compressed) chunks, not the whole
|
|
593
|
+
# multi-column file. Precise path sums per-column compressed_bytes
|
|
594
|
+
# from the stats artifact; files predating that column (or with no
|
|
595
|
+
# stats) fall back to a type-width ratio of the whole-file size.
|
|
596
|
+
# SELECT * keeps every column (full file).
|
|
597
|
+
tier3_files: Set[str] = set()
|
|
598
|
+
proj: Dict[str, int] = {}
|
|
599
|
+
if need_projection:
|
|
600
|
+
tier3_files, proj = self._projected_bytes_index(stats_df, selected_cols)
|
|
601
|
+
|
|
424
602
|
parquet_files: List[str] = []
|
|
425
603
|
for file_key in survivors:
|
|
426
604
|
parquet_files.append(self._to_duckdb_path(file_key))
|
|
427
|
-
|
|
605
|
+
full = int(key_size.get(file_key, 0))
|
|
606
|
+
reflection_file_size_raw += full
|
|
607
|
+
if not need_projection:
|
|
608
|
+
reflection_file_size += full
|
|
609
|
+
elif file_key in tier3_files:
|
|
610
|
+
reflection_file_size += proj.get(file_key, 0)
|
|
611
|
+
else:
|
|
612
|
+
reflection_file_size += self._ratio_bytes(
|
|
613
|
+
file_key, key_size, selected_cols, schema_types
|
|
614
|
+
)
|
|
428
615
|
|
|
429
616
|
# SuperSnapshot is created ONCE per (super_name, simple_name) after
|
|
430
617
|
# all snapshot iterations have accumulated their files and schema.
|
|
@@ -465,7 +652,11 @@ class DataEstimator:
|
|
|
465
652
|
self.timer.capture_and_reset_timing(event="ESTIMATE")
|
|
466
653
|
|
|
467
654
|
self.plan_stats.add_stat({"REFLECTIONS": total_reflections})
|
|
655
|
+
# REFLECTION_SIZE is the projected (selected-column) size that drives
|
|
656
|
+
# engine routing; REFLECTION_SIZE_RAW is the whole-file footprint, kept
|
|
657
|
+
# for observability so the two are comparable in the plans payload.
|
|
468
658
|
self.plan_stats.add_stat({"REFLECTION_SIZE": reflection_file_size})
|
|
659
|
+
self.plan_stats.add_stat({"REFLECTION_SIZE_RAW": reflection_file_size_raw})
|
|
469
660
|
|
|
470
661
|
# Read-path pruning observability — only when pruning is engaged, so a
|
|
471
662
|
# disabled-pruning read doesn't litter the payload with noise. Mirrors
|
|
@@ -215,12 +215,43 @@ def configure_httpfs_and_s3(
|
|
|
215
215
|
if not for_paths:
|
|
216
216
|
return
|
|
217
217
|
|
|
218
|
-
# Load httpfs
|
|
218
|
+
# Load httpfs. It is baked into the image and seeded into the DuckDB
|
|
219
|
+
# extension dir (see the container entrypoint), so LOAD normally succeeds
|
|
220
|
+
# with no network access.
|
|
221
|
+
#
|
|
222
|
+
# Why this is NOT a blind ``INSTALL`` fallback: ``INSTALL httpfs`` performs
|
|
223
|
+
# an HTTP GET to extensions.duckdb.org. On an offline / firewalled node
|
|
224
|
+
# that socket can stall for minutes — or hang indefinitely on a blackholed
|
|
225
|
+
# route — turning a should-be-instant failure into an unbounded query hang.
|
|
226
|
+
# So we fail fast instead:
|
|
227
|
+
# * SET autoinstall_known_extensions=false makes LOAD raise immediately
|
|
228
|
+
# when the extension is absent, rather than silently downloading it;
|
|
229
|
+
# * a network INSTALL is attempted ONLY when explicitly opted in via
|
|
230
|
+
# SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD;
|
|
231
|
+
# * otherwise we raise a clear, actionable error the caller returns.
|
|
219
232
|
try:
|
|
220
|
-
con.execute("
|
|
233
|
+
con.execute("SET autoinstall_known_extensions=false;")
|
|
221
234
|
except Exception:
|
|
222
|
-
|
|
235
|
+
pass
|
|
236
|
+
|
|
237
|
+
try:
|
|
223
238
|
con.execute("LOAD httpfs;")
|
|
239
|
+
except Exception as load_err:
|
|
240
|
+
if settings.SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD:
|
|
241
|
+
# Operator explicitly allowed reaching the network for a one-off
|
|
242
|
+
# install (e.g. an online dev box without a baked extension).
|
|
243
|
+
con.execute("INSTALL httpfs;")
|
|
244
|
+
con.execute("LOAD httpfs;")
|
|
245
|
+
else:
|
|
246
|
+
raise RuntimeError(
|
|
247
|
+
"DuckDB 'httpfs' extension is not available locally and network "
|
|
248
|
+
"auto-download is disabled, so this query cannot run. Bake/seed "
|
|
249
|
+
"httpfs into "
|
|
250
|
+
f"'{get_app_home()}/.duckdb/extensions/v<duckdb_version>/<platform>/' "
|
|
251
|
+
"(the container entrypoint restores it from /opt/duckdb-extensions), "
|
|
252
|
+
"or set SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD=true to permit a "
|
|
253
|
+
f"one-time online install. Underlying DuckDB error: {load_err}"
|
|
254
|
+
) from load_err
|
|
224
255
|
|
|
225
256
|
any_s3 = any(str(p).lower().startswith("s3://") for p in for_paths)
|
|
226
257
|
any_http = any(str(p).lower().startswith(("http://", "https://")) for p in for_paths)
|
|
@@ -634,6 +665,17 @@ def init_connection(
|
|
|
634
665
|
except Exception as e:
|
|
635
666
|
logger.warning(f"[duckdb.init] home_directory pin failed: {e}")
|
|
636
667
|
|
|
668
|
+
# Never let DuckDB auto-DOWNLOAD an extension. Everything we need (httpfs)
|
|
669
|
+
# is baked/seeded into the local extension dir; an implicit network install
|
|
670
|
+
# would reach out to extensions.duckdb.org and can hang for minutes on an
|
|
671
|
+
# offline/firewalled node — turning a should-be-instant error into an
|
|
672
|
+
# unbounded query hang. configure_httpfs_and_s3() owns the explicit,
|
|
673
|
+
# opt-in install path (SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD).
|
|
674
|
+
try:
|
|
675
|
+
con.execute("SET autoinstall_known_extensions=false;")
|
|
676
|
+
except Exception as e:
|
|
677
|
+
logger.debug(f"[duckdb.init] disabling extension auto-install failed: {e}")
|
|
678
|
+
|
|
637
679
|
# Resolve memory limit.
|
|
638
680
|
# Single env var SUPERTABLE_DUCKDB_MEMORY_LIMIT controls both executors.
|
|
639
681
|
# The `memory_limit` argument is the caller's fallback when the env var is absent.
|
|
@@ -93,6 +93,53 @@ def _to_s3a_path(file_path: str) -> str:
|
|
|
93
93
|
return file_path
|
|
94
94
|
|
|
95
95
|
|
|
96
|
+
def _resolve_spark_file(storage, file_path: str) -> str:
|
|
97
|
+
"""Resolve one snapshot file path to the form Spark should scan.
|
|
98
|
+
|
|
99
|
+
``sup.files`` are resolved once by the estimator using the *DuckDB* presign
|
|
100
|
+
setting (``SUPERTABLE_DUCKDB_PRESIGNED``) and shared by every engine, so
|
|
101
|
+
Spark re-resolves here to make its access method depend solely on
|
|
102
|
+
``SUPERTABLE_SPARK_PRESIGNED``:
|
|
103
|
+
|
|
104
|
+
* default (False) → a direct ``s3a://bucket/key`` path that Spark reads
|
|
105
|
+
with its own ``fs.s3a.*`` credentials (``_to_s3a_path`` already
|
|
106
|
+
normalises ``s3://``, presigned ``http(s)://`` and ``s3a://`` inputs);
|
|
107
|
+
* opt-in (True) → a freshly minted presigned ``http(s)://`` URL, so
|
|
108
|
+
Spark works even when the cluster has no object-store credentials — and
|
|
109
|
+
regardless of whether ``sup.files`` were presigned for DuckDB.
|
|
110
|
+
|
|
111
|
+
Presigning is best-effort: a missing ``presign``, a local path, or an
|
|
112
|
+
unparseable key falls back to the s3a form.
|
|
113
|
+
"""
|
|
114
|
+
s3a = _to_s3a_path(file_path)
|
|
115
|
+
if not settings.SUPERTABLE_SPARK_PRESIGNED or not s3a.startswith("s3a://"):
|
|
116
|
+
return s3a
|
|
117
|
+
|
|
118
|
+
presign_fn = getattr(storage, "presign", None)
|
|
119
|
+
if not callable(presign_fn):
|
|
120
|
+
return s3a
|
|
121
|
+
|
|
122
|
+
# s3a://bucket/full_key → object key. storage.presign() re-applies
|
|
123
|
+
# base_prefix (like read_bytes), so strip it here — mirroring
|
|
124
|
+
# _read_parquet_schema — to avoid doubling it.
|
|
125
|
+
full_key = s3a[len("s3a://"):].partition("/")[2]
|
|
126
|
+
if not full_key:
|
|
127
|
+
return s3a
|
|
128
|
+
base = (getattr(storage, "base_prefix", "") or "").strip("/")
|
|
129
|
+
rel_key = (
|
|
130
|
+
full_key[len(base) + 1:]
|
|
131
|
+
if base and full_key.startswith(base + "/")
|
|
132
|
+
else full_key
|
|
133
|
+
)
|
|
134
|
+
try:
|
|
135
|
+
url = presign_fn(rel_key)
|
|
136
|
+
if isinstance(url, str) and url:
|
|
137
|
+
return url
|
|
138
|
+
except Exception as e: # pragma: no cover - defensive
|
|
139
|
+
logger.debug(f"[spark.thrift] presign failed for {rel_key!r}; using s3a: {e}")
|
|
140
|
+
return s3a
|
|
141
|
+
|
|
142
|
+
|
|
96
143
|
# =========================================================
|
|
97
144
|
# Spark SQL helpers
|
|
98
145
|
# =========================================================
|
|
@@ -209,6 +256,7 @@ def _spark_create_tombstone_view(
|
|
|
209
256
|
source_table: str,
|
|
210
257
|
view_name: str,
|
|
211
258
|
tombstone_def,
|
|
259
|
+
storage=None,
|
|
212
260
|
) -> None:
|
|
213
261
|
"""Create a view that hides system columns and drops tombstoned rows.
|
|
214
262
|
|
|
@@ -239,11 +287,12 @@ def _spark_create_tombstone_view(
|
|
|
239
287
|
has_rowid = "__rowid__" in src_cols
|
|
240
288
|
|
|
241
289
|
if tomb_path and has_rowid:
|
|
242
|
-
#
|
|
243
|
-
# loop in the executor)
|
|
244
|
-
#
|
|
245
|
-
#
|
|
246
|
-
|
|
290
|
+
# Resolve the deletion-vector pointer the same way as the data files
|
|
291
|
+
# (see the data-file loop in the executor) so Spark reads it through the
|
|
292
|
+
# same access method — direct s3a:// by default, presigned when
|
|
293
|
+
# SUPERTABLE_SPARK_PRESIGNED is on — instead of a bare key or a
|
|
294
|
+
# DuckDB-shaped presigned URL.
|
|
295
|
+
escaped = _resolve_spark_file(storage, tomb_path).replace("'", "''")
|
|
247
296
|
sql = (
|
|
248
297
|
f"CREATE OR REPLACE TEMPORARY VIEW {view_name} AS "
|
|
249
298
|
f"SELECT {select_cols} FROM {source_table} AS src "
|
|
@@ -806,8 +855,10 @@ class SparkThriftExecutor:
|
|
|
806
855
|
if sup.files and table_name not in table_repr_file:
|
|
807
856
|
table_repr_file[table_name] = sup.files[0]
|
|
808
857
|
|
|
809
|
-
#
|
|
810
|
-
|
|
858
|
+
# Resolve to Spark's access form: direct s3a:// by default, or
|
|
859
|
+
# presigned URLs when SUPERTABLE_SPARK_PRESIGNED is on (handles
|
|
860
|
+
# s3://, presigned HTTP URLs and bare keys either way).
|
|
861
|
+
files = [_resolve_spark_file(self.storage, f) for f in sup.files]
|
|
811
862
|
|
|
812
863
|
logger.debug(
|
|
813
864
|
f"{log_prefix}[spark.thrift] creating view {table_name} "
|
|
@@ -898,7 +949,7 @@ class SparkThriftExecutor:
|
|
|
898
949
|
source = query_alias_to_name[alias]
|
|
899
950
|
tomb_def = tombstone_views.get(alias)
|
|
900
951
|
tomb_view = f"tomb_{source}_{query_suffix}"
|
|
901
|
-
_spark_create_tombstone_view(cursor, source, tomb_view, tomb_def)
|
|
952
|
+
_spark_create_tombstone_view(cursor, source, tomb_view, tomb_def, self.storage)
|
|
902
953
|
created_views.append(tomb_view)
|
|
903
954
|
query_alias_to_name[alias] = tomb_view
|
|
904
955
|
|