supertable 2.3.8__tar.gz → 2.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (201) hide show
  1. {supertable-2.3.8/supertable.egg-info → supertable-2.4.0}/PKG-INFO +1 -1
  2. {supertable-2.3.8 → supertable-2.4.0}/pyproject.toml +1 -1
  3. {supertable-2.3.8 → supertable-2.4.0}/setup.py +1 -1
  4. {supertable-2.3.8 → supertable-2.4.0}/supertable/__init__.py +1 -1
  5. {supertable-2.3.8 → supertable-2.4.0}/supertable/config/settings.py +31 -0
  6. {supertable-2.3.8 → supertable-2.4.0}/supertable/data_writer.py +21 -2
  7. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/data_estimator.py +205 -14
  8. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/engine_common.py +45 -3
  9. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/spark_thrift.py +59 -8
  10. {supertable-2.3.8 → supertable-2.4.0}/supertable/processing.py +164 -47
  11. {supertable-2.3.8 → supertable-2.4.0}/supertable/quality/checker.py +27 -2
  12. {supertable-2.3.8 → supertable-2.4.0}/supertable/quality/scheduler.py +7 -3
  13. supertable-2.4.0/supertable/tests/test_data_estimator_projection.py +201 -0
  14. supertable-2.4.0/supertable/tests/test_metadata_partitioning.py +221 -0
  15. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_processing_stats.py +75 -0
  16. supertable-2.4.0/supertable/tests/test_quality_checker.py +120 -0
  17. supertable-2.4.0/supertable/tests/test_spark_file_resolution.py +148 -0
  18. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_stats_cache.py +91 -1
  19. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_stats_schema_snapshot.py +7 -0
  20. supertable-2.4.0/supertable/tests/test_tombstone_cache.py +162 -0
  21. supertable-2.4.0/supertable/utils/helper.py +63 -0
  22. supertable-2.4.0/supertable/utils/tests/test_hourly_partition.py +63 -0
  23. {supertable-2.3.8 → supertable-2.4.0/supertable.egg-info}/PKG-INFO +1 -1
  24. {supertable-2.3.8 → supertable-2.4.0}/supertable.egg-info/SOURCES.txt +6 -0
  25. supertable-2.3.8/supertable/utils/helper.py +0 -39
  26. {supertable-2.3.8 → supertable-2.4.0}/LICENSE +0 -0
  27. {supertable-2.3.8 → supertable-2.4.0}/README.md +0 -0
  28. {supertable-2.3.8 → supertable-2.4.0}/requirements.txt +0 -0
  29. {supertable-2.3.8 → supertable-2.4.0}/setup.cfg +0 -0
  30. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/__init__.py +0 -0
  31. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/admin.py +0 -0
  32. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/chain.py +0 -0
  33. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/consumers.py +0 -0
  34. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/crypto.py +0 -0
  35. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/events.py +0 -0
  36. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/export.py +0 -0
  37. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/logger.py +0 -0
  38. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/middleware.py +0 -0
  39. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/reader.py +0 -0
  40. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/retention.py +0 -0
  41. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/tests/__init__.py +0 -0
  42. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/tests/test_chain.py +0 -0
  43. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/tests/test_crypto.py +0 -0
  44. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/tests/test_emit.py +0 -0
  45. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/tests/test_events.py +0 -0
  46. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/tests/test_retention.py +0 -0
  47. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/writer_parquet.py +0 -0
  48. {supertable-2.3.8 → supertable-2.4.0}/supertable/audit/writer_redis.py +0 -0
  49. {supertable-2.3.8 → supertable-2.4.0}/supertable/config/__init__.py +0 -0
  50. {supertable-2.3.8 → supertable-2.4.0}/supertable/config/defaults.py +0 -0
  51. {supertable-2.3.8 → supertable-2.4.0}/supertable/config/homedir.py +0 -0
  52. {supertable-2.3.8 → supertable-2.4.0}/supertable/config/tests/__init__.py +0 -0
  53. {supertable-2.3.8 → supertable-2.4.0}/supertable/config/tests/test_defaults.py +0 -0
  54. {supertable-2.3.8 → supertable-2.4.0}/supertable/config/tests/test_homedir.py +0 -0
  55. {supertable-2.3.8 → supertable-2.4.0}/supertable/config/tests/test_settings.py +0 -0
  56. {supertable-2.3.8 → supertable-2.4.0}/supertable/data_classes.py +0 -0
  57. {supertable-2.3.8 → supertable-2.4.0}/supertable/data_reader.py +0 -0
  58. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/__init__.py +0 -0
  59. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/__init__.py +0 -0
  60. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/__main__.py +0 -0
  61. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/check_filter_builder.py +0 -0
  62. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/controller.py +0 -0
  63. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/data_writer_helpers.py +0 -0
  64. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/defaults.py +0 -0
  65. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/dummy_data.py +0 -0
  66. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/read_parquet_header.py +0 -0
  67. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s01_01_01_create_super_table.py +0 -0
  68. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s01_01_02_enable_mirroring_formats.py +0 -0
  69. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s01_02_create_roles.py +0 -0
  70. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s01_03_create_users.py +0 -0
  71. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_01_write_dummy_data.py +0 -0
  72. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_02_write_single_data.py +0 -0
  73. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_03_01_write_staging.py +0 -0
  74. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_03_02_create_pipe.py +0 -0
  75. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_04_01_write_monitoring_simple.py +0 -0
  76. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_04_02_write_monitoring_parallel.py +0 -0
  77. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s02_05_write_tombstone.py +0 -0
  78. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_01_read_data_error.py +0 -0
  79. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_02_01_read_super_data_ok.py +0 -0
  80. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_02_02_read_table_data_ok.py +0 -0
  81. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_03_read_meta.py +0 -0
  82. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_04_read_staging.py +0 -0
  83. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_06_01_read_roles.py +0 -0
  84. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_06_02_read_user.py +0 -0
  85. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_07_01_estimate_read.py +0 -0
  86. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_07_02_estimate_files.py +0 -0
  87. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s03_08_read_snapshot_history.py +0 -0
  88. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s04_01_03_delete_pipe.py +0 -0
  89. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s05_01_delete_table.py +0 -0
  90. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/quickstart/s05_02_delete_super_table.py +0 -0
  91. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/webshop/__init__.py +0 -0
  92. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/webshop/core.py +0 -0
  93. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/webshop/defaults.py +0 -0
  94. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/webshop/generate.py +0 -0
  95. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/webshop/load.py +0 -0
  96. {supertable-2.3.8 → supertable-2.4.0}/supertable/demo/webshop/topup.py +0 -0
  97. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/__init__.py +0 -0
  98. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/duckdb_lite.py +0 -0
  99. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/duckdb_pro.py +0 -0
  100. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/engine_config.py +0 -0
  101. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/engine_enum.py +0 -0
  102. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/executor.py +0 -0
  103. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/plan_stats.py +0 -0
  104. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/tests/__init__.py +0 -0
  105. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/tests/conftest.py +0 -0
  106. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/tests/test_engine.py +0 -0
  107. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/tests/test_engine_config.py +0 -0
  108. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/tests/test_engine_routing.py +0 -0
  109. {supertable-2.3.8 → supertable-2.4.0}/supertable/engine/tests/test_engine_spill.py +0 -0
  110. {supertable-2.3.8 → supertable-2.4.0}/supertable/errors.py +0 -0
  111. {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/__init__.py +0 -0
  112. {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/benchmarks/__init__.py +0 -0
  113. {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/benchmarks/benchmark_locking.py +0 -0
  114. {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/benchmarks/measure_lock_speed.py +0 -0
  115. {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/benchmarks/measure_lock_time.py +0 -0
  116. {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/file_lock.py +0 -0
  117. {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/redis_lock.py +0 -0
  118. {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/tests/__init__.py +0 -0
  119. {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/tests/test_file_lock.py +0 -0
  120. {supertable-2.3.8 → supertable-2.4.0}/supertable/locking/tests/test_redis_lock.py +0 -0
  121. {supertable-2.3.8 → supertable-2.4.0}/supertable/logging.py +0 -0
  122. {supertable-2.3.8 → supertable-2.4.0}/supertable/meta_reader.py +0 -0
  123. {supertable-2.3.8 → supertable-2.4.0}/supertable/mirroring/__init__.py +0 -0
  124. {supertable-2.3.8 → supertable-2.4.0}/supertable/mirroring/mirror_delta.py +0 -0
  125. {supertable-2.3.8 → supertable-2.4.0}/supertable/mirroring/mirror_formats.py +0 -0
  126. {supertable-2.3.8 → supertable-2.4.0}/supertable/mirroring/mirror_iceberg.py +0 -0
  127. {supertable-2.3.8 → supertable-2.4.0}/supertable/mirroring/mirror_parquet.py +0 -0
  128. {supertable-2.3.8 → supertable-2.4.0}/supertable/monitoring/__init__.py +0 -0
  129. {supertable-2.3.8 → supertable-2.4.0}/supertable/monitoring/partitions.py +0 -0
  130. {supertable-2.3.8 → supertable-2.4.0}/supertable/monitoring_writer.py +0 -0
  131. {supertable-2.3.8 → supertable-2.4.0}/supertable/plan_extender.py +0 -0
  132. {supertable-2.3.8 → supertable-2.4.0}/supertable/quality/__init__.py +0 -0
  133. {supertable-2.3.8 → supertable-2.4.0}/supertable/quality/anomaly.py +0 -0
  134. {supertable-2.3.8 → supertable-2.4.0}/supertable/quality/config.py +0 -0
  135. {supertable-2.3.8 → supertable-2.4.0}/supertable/quality/history.py +0 -0
  136. {supertable-2.3.8 → supertable-2.4.0}/supertable/query_plan_manager.py +0 -0
  137. {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/__init__.py +0 -0
  138. {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/access_control.py +0 -0
  139. {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/filter_builder.py +0 -0
  140. {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/permissions.py +0 -0
  141. {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/role_manager.py +0 -0
  142. {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/row_column_security.py +0 -0
  143. {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/tests/test_filter_builder.py +0 -0
  144. {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/tests/test_rbac.py +0 -0
  145. {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/tests/test_rbac_per_table.py +0 -0
  146. {supertable-2.3.8 → supertable-2.4.0}/supertable/rbac/user_manager.py +0 -0
  147. {supertable-2.3.8 → supertable-2.4.0}/supertable/redis_catalog.py +0 -0
  148. {supertable-2.3.8 → supertable-2.4.0}/supertable/redis_connector.py +0 -0
  149. {supertable-2.3.8 → supertable-2.4.0}/supertable/redis_infra.py +0 -0
  150. {supertable-2.3.8 → supertable-2.4.0}/supertable/redis_keys.py +0 -0
  151. {supertable-2.3.8 → supertable-2.4.0}/supertable/simple_table.py +0 -0
  152. {supertable-2.3.8 → supertable-2.4.0}/supertable/staging_area.py +0 -0
  153. {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/__init__.py +0 -0
  154. {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/azure_storage.py +0 -0
  155. {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/gcp_storage.py +0 -0
  156. {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/local_storage.py +0 -0
  157. {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/minio_storage.py +0 -0
  158. {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/s3_storage.py +0 -0
  159. {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/storage_factory.py +0 -0
  160. {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/storage_interface.py +0 -0
  161. {supertable-2.3.8 → supertable-2.4.0}/supertable/storage/tests/test_storage.py +0 -0
  162. {supertable-2.3.8 → supertable-2.4.0}/supertable/super_pipe.py +0 -0
  163. {supertable-2.3.8 → supertable-2.4.0}/supertable/super_table.py +0 -0
  164. {supertable-2.3.8 → supertable-2.4.0}/supertable/system_query.py +0 -0
  165. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/__init__.py +0 -0
  166. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_align_to_schema_fix.py +0 -0
  167. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_compaction_selection.py +0 -0
  168. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_create_if_missing.py +0 -0
  169. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_data_reader.py +0 -0
  170. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_data_reader_preflight.py +0 -0
  171. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_data_writer.py +0 -0
  172. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_data_writer_compact.py +0 -0
  173. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_data_writer_comprehensive.py +0 -0
  174. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_errors.py +0 -0
  175. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_meta_reader.py +0 -0
  176. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_monitoring_partitions.py +0 -0
  177. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_monitoring_sink_guard.py +0 -0
  178. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_newer_than.py +0 -0
  179. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_parquet_statistics.py +0 -0
  180. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_processing.py +0 -0
  181. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_processing_compact_resources.py +0 -0
  182. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_query_sql.py +0 -0
  183. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_read_pruning_differential.py +0 -0
  184. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_read_pruning_integration.py +0 -0
  185. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_redis_key_prefix.py +0 -0
  186. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_resolve_overwrite_writes.py +0 -0
  187. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_simple_table.py +0 -0
  188. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_stats_pruning.py +0 -0
  189. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_super_table.py +0 -0
  190. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_supertable_all.py +0 -0
  191. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_system_query.py +0 -0
  192. {supertable-2.3.8 → supertable-2.4.0}/supertable/tests/test_write_probe_gate.py +0 -0
  193. {supertable-2.3.8 → supertable-2.4.0}/supertable/utils/__init__.py +0 -0
  194. {supertable-2.3.8 → supertable-2.4.0}/supertable/utils/profiler.py +0 -0
  195. {supertable-2.3.8 → supertable-2.4.0}/supertable/utils/sql_parser.py +0 -0
  196. {supertable-2.3.8 → supertable-2.4.0}/supertable/utils/tests/test_sql_parser_columns.py +0 -0
  197. {supertable-2.3.8 → supertable-2.4.0}/supertable/utils/timer.py +0 -0
  198. {supertable-2.3.8 → supertable-2.4.0}/supertable.egg-info/dependency_links.txt +0 -0
  199. {supertable-2.3.8 → supertable-2.4.0}/supertable.egg-info/entry_points.txt +0 -0
  200. {supertable-2.3.8 → supertable-2.4.0}/supertable.egg-info/requires.txt +0 -0
  201. {supertable-2.3.8 → supertable-2.4.0}/supertable.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: supertable
3
- Version: 2.3.8
3
+ Version: 2.4.0
4
4
  Summary: SuperTable — versioned data lake library for SQL analytics on Parquet + Redis.
5
5
  Author: Levente Kupas
6
6
  Author-email: Levente Kupas <lkupas@kladnasoft.com>
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "supertable"
7
- version = "2.3.8"
7
+ version = "2.4.0"
8
8
  description = "SuperTable — versioned data lake library for SQL analytics on Parquet + Redis."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -19,7 +19,7 @@ long_description = readme.read_text(encoding="utf-8") if readme.exists() else ""
19
19
 
20
20
  setup(
21
21
  name="supertable",
22
- version="2.3.8",
22
+ version="2.4.0",
23
23
  description="SuperTable — versioned data lake library for SQL analytics on Parquet + Redis.",
24
24
  long_description=long_description,
25
25
  long_description_content_type="text/markdown",
@@ -25,7 +25,7 @@ See the ``supertable.demo`` package for runnable end-to-end demos and the
25
25
  project documentation for the full API surface.
26
26
  """
27
27
 
28
- __version__ = "2.3.8"
28
+ __version__ = "2.4.0"
29
29
 
30
30
  # Re-export the core public surface so users can do ``from supertable import …``
31
31
  # instead of remembering submodule paths.
@@ -157,6 +157,14 @@ class Settings:
157
157
  SUPERTABLE_DUCKDB_MATERIALIZE: str = "view" # SUPERTABLE_DUCKDB_MATERIALIZE
158
158
  SUPERTABLE_DUCKDB_PRESIGNED: bool = False # SUPERTABLE_DUCKDB_PRESIGNED
159
159
  SUPERTABLE_DUCKDB_USE_HTTPFS: bool = False # SUPERTABLE_DUCKDB_USE_HTTPFS
160
+ # Allow DuckDB to DOWNLOAD a missing extension (httpfs) from
161
+ # extensions.duckdb.org at query time. Default OFF: httpfs is baked into
162
+ # the image and seeded into the extension dir, and an implicit network
163
+ # install on an offline/firewalled node can hang for minutes instead of
164
+ # failing. When OFF, a genuinely missing extension raises immediately with
165
+ # an actionable message; set ON only on a networked box that must
166
+ # self-install.
167
+ SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD: bool = False # SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD
160
168
  # Write-path overwrite/delete resolution via the DuckDB pushdown probe.
161
169
  # Disabled by default: the polars fallback reads only the projected key
162
170
  # columns through the storage SDK and needs no httpfs extension, so it works
@@ -196,6 +204,12 @@ class Settings:
196
204
  SUPERTABLE_SPARK_STATEMENT_TIMEOUT: int = 120 # SUPERTABLE_SPARK_STATEMENT_TIMEOUT
197
205
  SUPERTABLE_SPARK_CONNECT_TIMEOUT: int = 30 # SUPERTABLE_SPARK_CONNECT_TIMEOUT
198
206
  SUPERTABLE_SPARK_BATCH_SIZE: int = 50 # SUPERTABLE_SPARK_BATCH_SIZE
207
+ # Spark file access. Default False → Spark scans direct s3a://bucket/key
208
+ # paths using its own fs.s3a.* credentials, independent of the DuckDB
209
+ # presign setting that shapes the shared reflection file list. True →
210
+ # Spark scans presigned http(s):// URLs minted per request (use when the
211
+ # cluster cannot reach the object store with its own credentials).
212
+ SUPERTABLE_SPARK_PRESIGNED: bool = False # SUPERTABLE_SPARK_PRESIGNED
199
213
 
200
214
  # ── Redis ────────────────────────────────────────────────────────
201
215
  SUPERTABLE_REDIS_URL: str = "" # SUPERTABLE_REDIS_URL
@@ -291,11 +305,24 @@ class Settings:
291
305
  # always read fresh and never cached). 0 disables caching.
292
306
  SUPERTABLE_STATS_CACHE_MAX_TABLES: int = 64 # SUPERTABLE_STATS_CACHE_MAX_TABLES
293
307
 
308
+ # Max number of tables whose *latest* tombstone (deletion-vector) artifact is
309
+ # held in the in-process tombstone cache (one DataFrame per table; older
310
+ # versions are always read fresh and never cached). Mirrors the stats cache
311
+ # so a process writing in a loop skips the carry-forward read. 0 disables.
312
+ SUPERTABLE_TOMBSTONE_CACHE_MAX_TABLES: int = 64 # SUPERTABLE_TOMBSTONE_CACHE_MAX_TABLES
313
+
294
314
  # Read-path file pruning: when True the estimator uses the stats artifact to
295
315
  # drop parquet files that provably can't satisfy a query's WHERE predicates.
296
316
  # Conservative (never drops a file that could match); set False to disable.
297
317
  SUPERTABLE_READ_PRUNING_ENABLED: bool = True # SUPERTABLE_READ_PRUNING_ENABLED
298
318
 
319
+ # Projection-aware read sizing: when True the estimator charges a query only
320
+ # the on-disk (compressed) bytes of the columns it selects — via per-column
321
+ # ``compressed_bytes`` in the stats artifact, falling back to a type-width
322
+ # ratio of the whole file. Drives AUTO engine routing (lite/pro/spark). Set
323
+ # False to fall back to whole-file sizing (every query charged all columns).
324
+ SUPERTABLE_READ_PROJECTION_SIZING_ENABLED: bool = True
325
+
299
326
  # ── Audit ────────────────────────────────────────────────────────
300
327
  # Audit is OFF by default. Enable per-organization in the WebUI
301
328
  # /ui/audit → Compliance tab (persisted at supertable:{org}:system:audit:config),
@@ -444,6 +471,7 @@ def _build_settings() -> Settings:
444
471
  SUPERTABLE_DUCKDB_MATERIALIZE=_env_str("SUPERTABLE_DUCKDB_MATERIALIZE", "view"),
445
472
  SUPERTABLE_DUCKDB_PRESIGNED=_env_bool("SUPERTABLE_DUCKDB_PRESIGNED", False),
446
473
  SUPERTABLE_DUCKDB_USE_HTTPFS=_env_bool("SUPERTABLE_DUCKDB_USE_HTTPFS", False),
474
+ SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD=_env_bool("SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD", False),
447
475
  SUPERTABLE_DUCKDB_WRITE_PROBE=_env_bool("SUPERTABLE_DUCKDB_WRITE_PROBE", False),
448
476
  SUPERTABLE_DUCKDB_TOMBSTONE_CACHE_MAX_PER_TABLE=_env_int("SUPERTABLE_DUCKDB_TOMBSTONE_CACHE_MAX_PER_TABLE", 8),
449
477
  SUPERTABLE_DUCKDB_TOMBSTONE_CACHE_TTL_SEC=_env_int("SUPERTABLE_DUCKDB_TOMBSTONE_CACHE_TTL_SEC", 300),
@@ -460,6 +488,7 @@ def _build_settings() -> Settings:
460
488
  SUPERTABLE_SPARK_STATEMENT_TIMEOUT=_env_int("SUPERTABLE_SPARK_STATEMENT_TIMEOUT", 120),
461
489
  SUPERTABLE_SPARK_CONNECT_TIMEOUT=_env_int("SUPERTABLE_SPARK_CONNECT_TIMEOUT", 30),
462
490
  SUPERTABLE_SPARK_BATCH_SIZE=_env_int("SUPERTABLE_SPARK_BATCH_SIZE", 50),
491
+ SUPERTABLE_SPARK_PRESIGNED=_env_bool("SUPERTABLE_SPARK_PRESIGNED", False),
463
492
 
464
493
  # ── Redis ────────────────────────────────────────────────────
465
494
  SUPERTABLE_REDIS_URL=_env_str("SUPERTABLE_REDIS_URL"),
@@ -551,7 +580,9 @@ def _build_settings() -> Settings:
551
580
  # ── Meta Reader / Caching ────────────────────────────────────
552
581
  SUPERTABLE_SUPER_META_CACHE_TTL_S=meta_ttl,
553
582
  SUPERTABLE_STATS_CACHE_MAX_TABLES=_env_int("SUPERTABLE_STATS_CACHE_MAX_TABLES", 64),
583
+ SUPERTABLE_TOMBSTONE_CACHE_MAX_TABLES=_env_int("SUPERTABLE_TOMBSTONE_CACHE_MAX_TABLES", 64),
554
584
  SUPERTABLE_READ_PRUNING_ENABLED=_env_bool("SUPERTABLE_READ_PRUNING_ENABLED", True),
585
+ SUPERTABLE_READ_PROJECTION_SIZING_ENABLED=_env_bool("SUPERTABLE_READ_PROJECTION_SIZING_ENABLED", True),
555
586
 
556
587
  # ── Audit ────────────────────────────────────────────────────
557
588
  SUPERTABLE_AUDIT_ENABLED=_env_bool("SUPERTABLE_AUDIT_ENABLED", False),
@@ -34,6 +34,8 @@ from supertable.processing import (
34
34
  prune_overlapping_files_by_stats,
35
35
  load_stats,
36
36
  cache_stats,
37
+ load_tombstone,
38
+ cache_tombstone,
37
39
  write_parquet_and_collect_resources,
38
40
  compact_resources,
39
41
  compact_tombstones,
@@ -525,8 +527,13 @@ class DataWriter:
525
527
  # required=True: a DV that exists but cannot be read must abort
526
528
  # the write, never be treated as empty — silently dropping the
527
529
  # carried-forward vector would resurrect previously deleted rows.
530
+ # Cache-first: if this process wrote the table on a prior loop
531
+ # iteration, the current deletion-vector is already in memory
532
+ # (seeded below after the pointer is pinned), so this is a hit
533
+ # with no storage round-trip. required=True preserves the
534
+ # carry-forward safety on a genuine miss (abort, never truncate).
528
535
  prev_dv_df = (
529
- _read_parquet_safe(prev_tombstone_path, profiler=profiler, required=True)
536
+ load_tombstone(prev_tombstone_path, allow_cache=True, required=True, profiler=profiler)
530
537
  if prev_tombstone_path else None
531
538
  )
532
539
  # The rowid set is consumed only by the idempotency filter below,
@@ -790,6 +797,15 @@ class DataWriter:
790
797
  # and its row count.
791
798
  last_simple_table["tombstone"] = tombstone_path
792
799
  last_simple_table["tombstone_rows"] = tombstone_rows
800
+ # Seed the in-process cache so the NEXT write's carry-forward read
801
+ # (prev_dv_df above) is a pure memory hit. The frame is the fresh
802
+ # build/reclaim result when this write changed the vector, else the
803
+ # already-loaded prev frame for a pure carry-forward. No-op when
804
+ # the vector was fully consumed this write (tombstone_path is None).
805
+ cache_tombstone(
806
+ tombstone_path,
807
+ combined_tombstone_df if combined_tombstone_df is not None else prev_dv_df,
808
+ )
793
809
  mark("compact_tombstones")
794
810
 
795
811
  # Phase B — auto small-file compaction. Merge the accumulated
@@ -1328,7 +1344,10 @@ class DataWriter:
1328
1344
  # new file while the vector kept pointing at the sunset __file__ —
1329
1345
  # leaving them permanently unreclaimable. Failing loud leaves the
1330
1346
  # prior snapshot + vector intact for a retry, and matches the
1331
- # write-path carry-forward read (required=True) above.
1347
+ # write-path carry-forward read (required=True) above. Read directly
1348
+ # (not via the in-process cache): compact() always drains the vector,
1349
+ # so there is no carry-forward hit to gain and it never re-seeds the
1350
+ # cache after draining — the loop-caching win lives on the write path.
1332
1351
  tombstone_df = (
1333
1352
  _read_parquet_safe(tombstone_path, required=True)
1334
1353
  if tombstone_path else None
@@ -7,6 +7,8 @@ from collections import defaultdict
7
7
  from typing import Iterable, Set, List, Dict, Optional, Tuple
8
8
  from urllib.parse import urlparse
9
9
 
10
+ import polars
11
+
10
12
  from supertable.config.defaults import logger
11
13
  from supertable.config.settings import settings
12
14
  from supertable.data_classes import Reflection, SuperSnapshot
@@ -18,7 +20,12 @@ from supertable.utils.profiler import Profiler
18
20
  from supertable.redis_catalog import RedisCatalog # Redis leaf pointers for snapshots
19
21
 
20
22
  from supertable.utils.sql_parser import TableDefinition
21
- from supertable.processing import load_stats, prune_files_by_predicates
23
+ from supertable.processing import (
24
+ load_stats,
25
+ prune_files_by_predicates,
26
+ ROWID_COL,
27
+ TIMESTAMP_COL,
28
+ )
22
29
 
23
30
 
24
31
  from typing import Dict, List, Optional, Set, Tuple
@@ -306,22 +313,25 @@ class DataEstimator:
306
313
  super_name: str,
307
314
  simple_name: str,
308
315
  raw_keys: List[str],
309
- stats_file: Optional[str],
316
+ stats_df: Optional["polars.DataFrame"],
310
317
  profiler: Optional[Profiler] = None,
311
318
  ) -> List[str]:
312
319
  """Narrow *raw_keys* to those that could satisfy the query predicates.
313
320
 
314
- Returns *raw_keys* unchanged whenever pruning is disabled, there's no
315
- stats artifact, or the query carries no usable constraint for this
316
- table — and never raises (a pruning failure must not break a read).
321
+ Takes the already-loaded *stats_df* (loaded once per table by
322
+ :meth:`estimate` and shared with projection sizing) rather than a path,
323
+ so the artifact is read at most once. Returns *raw_keys* unchanged
324
+ whenever pruning is disabled, there's no stats artifact, or the query
325
+ carries no usable constraint for this table — and never raises (a
326
+ pruning failure must not break a read).
317
327
 
318
328
  *profiler*, when supplied, accumulates the same IO/pruning counters the
319
- write path emits (``stats_cache_hit``/``stats_cache_miss``,
320
- ``read_pruned_files``) so the read monitoring payload can surface them.
329
+ write path emits (``read_pruned_files``) so the read monitoring payload
330
+ can surface them.
321
331
  """
322
332
  if not settings.SUPERTABLE_READ_PRUNING_ENABLED:
323
333
  return raw_keys
324
- if not stats_file or not raw_keys:
334
+ if stats_df is None or not raw_keys:
325
335
  return raw_keys
326
336
  occurrences = self.predicate_constraints.get(
327
337
  (super_name.lower(), simple_name.lower())
@@ -329,7 +339,6 @@ class DataEstimator:
329
339
  if not occurrences:
330
340
  return raw_keys
331
341
  try:
332
- stats_df = load_stats(stats_file, allow_cache=True, profiler=profiler)
333
342
  return prune_files_by_predicates(
334
343
  raw_keys, stats_df, occurrences, profiler=profiler,
335
344
  )
@@ -337,6 +346,138 @@ class DataEstimator:
337
346
  logger.warning(f"[estimate.prune] pruning skipped for {super_name}.{simple_name}: {e}")
338
347
  return raw_keys
339
348
 
349
+ # ----------------------- projection-aware sizing -----------------------
350
+
351
+ # Rough per-value byte widths for the type-width *fallback* (used only when
352
+ # per-column ``compressed_bytes`` are unavailable — e.g. a stats file that
353
+ # predates the column, or no stats artifact at all). Substring-matched
354
+ # against the stored (DuckDB/polars) type name, most-specific first.
355
+ _STRING_AVG_WIDTH = 16
356
+
357
+ def _type_width(self, type_name: Optional[str]) -> int:
358
+ """Estimated on-disk bytes-per-value for a column type (fallback only)."""
359
+ t = (type_name or "").strip().lower()
360
+ if not t:
361
+ return 8
362
+ if "bool" in t:
363
+ return 1
364
+ if "timestamp" in t or "datetime" in t:
365
+ return 8
366
+ if "date" in t:
367
+ return 4
368
+ if any(s in t for s in ("varchar", "char", "utf8", "string", "text",
369
+ "json", "blob", "binary", "bytea")):
370
+ return self._STRING_AVG_WIDTH
371
+ if "double" in t or "float64" in t:
372
+ return 8
373
+ if "float" in t or "real" in t:
374
+ return 4
375
+ if "decimal" in t or "numeric" in t:
376
+ return 8
377
+ if "bigint" in t or "int64" in t or "long" in t:
378
+ return 8
379
+ if "smallint" in t or "int16" in t:
380
+ return 2
381
+ if "tinyint" in t or "int8" in t:
382
+ return 1
383
+ if "int" in t:
384
+ return 4
385
+ return 8
386
+
387
+ def _selected_columns(self, super_name: str, simple_name: str) -> Optional[Set[str]]:
388
+ """Lowercased set of user columns this query projects from the table.
389
+
390
+ Returns ``None`` to mean "every column" — i.e. ``SELECT *`` / ``t.*``
391
+ (an empty ``TableDefinition.columns``), an unrecognised table, or a
392
+ projection that resolves to only system columns. In every ``None`` case
393
+ the caller uses the whole-file size (no projection savings), which is
394
+ the safe over-estimate. System columns (``__rowid__`` / ``__timestamp__``)
395
+ are stripped: the read view hides them, so they're never scanned.
396
+ """
397
+ key = (super_name.lower(), simple_name.lower())
398
+ selected: Set[str] = set()
399
+ matched = False
400
+ for t in self.tables:
401
+ if (t.super_name.lower(), t.simple_name.lower()) != key:
402
+ continue
403
+ matched = True
404
+ if not t.columns: # [] => SELECT * / t.* => whole table
405
+ return None
406
+ for c in t.columns:
407
+ cl = c.lower()
408
+ if cl not in (ROWID_COL, TIMESTAMP_COL):
409
+ selected.add(cl)
410
+ if not matched or not selected:
411
+ return None
412
+ return selected
413
+
414
+ def _projected_bytes_index(
415
+ self,
416
+ stats_df: Optional["polars.DataFrame"],
417
+ selected_cols: Set[str],
418
+ ) -> Tuple[Set[str], Dict[str, int]]:
419
+ """Sum per-column ``compressed_bytes`` for the selected columns per file.
420
+
421
+ Returns ``(tier3_files, proj)`` where *proj* maps a file path to the
422
+ summed on-disk bytes of its selected columns, and *tier3_files* is the
423
+ subset of files for which that sum is *trustworthy* — every matched row
424
+ carried a non-NULL ``compressed_bytes``. A file with any NULL (an older
425
+ carried-forward row) is omitted so the caller falls back to a whole-file
426
+ ratio rather than under-counting it as zero.
427
+ """
428
+ if (
429
+ stats_df is None
430
+ or stats_df.height == 0
431
+ or "compressed_bytes" not in stats_df.columns
432
+ ):
433
+ return set(), {}
434
+ sel = (
435
+ stats_df.select(["file_path", "column_name", "compressed_bytes"])
436
+ .with_columns(polars.col("column_name").str.to_lowercase().alias("__cn"))
437
+ .filter(polars.col("__cn").is_in(list(selected_cols)))
438
+ )
439
+ if sel.height == 0:
440
+ return set(), {}
441
+ agg = sel.group_by("file_path").agg(
442
+ [
443
+ polars.col("compressed_bytes").sum().alias("__b"),
444
+ polars.col("compressed_bytes").is_null().sum().alias("__nulls"),
445
+ ]
446
+ )
447
+ tier3_files: Set[str] = set()
448
+ proj: Dict[str, int] = {}
449
+ for r in agg.iter_rows(named=True):
450
+ if int(r["__nulls"] or 0) == 0:
451
+ fp = r["file_path"]
452
+ tier3_files.add(fp)
453
+ proj[fp] = int(r["__b"] or 0)
454
+ return tier3_files, proj
455
+
456
+ def _ratio_bytes(
457
+ self,
458
+ file_key: str,
459
+ key_size: Dict[str, int],
460
+ selected_cols: Set[str],
461
+ schema_types: Dict[str, str],
462
+ ) -> int:
463
+ """Fallback size: scale the whole-file bytes by the selected columns'
464
+ type-width share of the table schema. Used only when precise
465
+ per-column ``compressed_bytes`` are unavailable for *file_key*."""
466
+ full = int(key_size.get(file_key, 0))
467
+ if full <= 0 or not schema_types:
468
+ return full
469
+ all_cols = {
470
+ c: ty for c, ty in schema_types.items()
471
+ if c not in (ROWID_COL, TIMESTAMP_COL)
472
+ }
473
+ total_w = sum(self._type_width(ty) for ty in all_cols.values())
474
+ if total_w <= 0:
475
+ return full
476
+ sel_w = sum(self._type_width(all_cols[c]) for c in selected_cols if c in all_cols)
477
+ if sel_w <= 0:
478
+ return full
479
+ return int(full * sel_w / total_w)
480
+
340
481
  # ----------------------- main API -----------------------
341
482
  def estimate(self) -> Reflection:
342
483
  """
@@ -352,7 +493,8 @@ class DataEstimator:
352
493
  prune_profiler = Profiler()
353
494
 
354
495
  supers: List[SuperSnapshot] = []
355
- reflection_file_size = 0
496
+ reflection_file_size = 0 # projected (selected-column) bytes — routing
497
+ reflection_file_size_raw = 0 # whole-file bytes — physical footprint
356
498
  max_freshness_ms = 0
357
499
  files_before_prune = 0
358
500
  files_pruned = 0
@@ -378,6 +520,7 @@ class DataEstimator:
378
520
  )
379
521
 
380
522
  schema: Set[str] = set()
523
+ schema_types: Dict[str, str] = {}
381
524
  raw_keys: List[str] = []
382
525
  key_size: Dict[str, int] = {}
383
526
  stats_file: Optional[str] = None
@@ -395,7 +538,12 @@ class DataEstimator:
395
538
 
396
539
  current_version = current_snapshot_data.get("snapshot_version", 0)
397
540
  current_schema = self._schema_to_dict(current_snapshot_data.get("schema", {}))
398
- schema.update(dict_keys_to_lowercase(current_schema).keys())
541
+ lowered_schema = dict_keys_to_lowercase(current_schema)
542
+ schema.update(lowered_schema.keys())
543
+ # Retain name->type for the projection ratio fallback. First
544
+ # writer wins for a given column (schemas are stable per table).
545
+ for _cname, _ctype in lowered_schema.items():
546
+ schema_types.setdefault(_cname, _ctype)
399
547
  sf = current_snapshot_data.get("stats_file")
400
548
  if sf:
401
549
  stats_file = sf
@@ -408,23 +556,62 @@ class DataEstimator:
408
556
  raw_keys.append(file_key)
409
557
  key_size[file_key] = int(resource.get("file_size", 0))
410
558
 
559
+ # Which columns does the query actually read? None => SELECT *
560
+ # (whole table, no projection savings).
561
+ selected_cols = self._selected_columns(super_name, simple_name)
562
+ need_projection = (
563
+ selected_cols is not None
564
+ and settings.SUPERTABLE_READ_PROJECTION_SIZING_ENABLED
565
+ )
566
+ has_predicate = bool(
567
+ self.predicate_constraints.get((super_name.lower(), simple_name.lower()))
568
+ )
569
+
570
+ # Load the stats artifact ONCE per table and reuse it for BOTH
571
+ # predicate pruning and projection sizing (cache-backed: a repeated
572
+ # read of the same table is a memory hit). Only load when at least
573
+ # one consumer needs it, so a SELECT * with no WHERE stays free.
574
+ stats_df: Optional["polars.DataFrame"] = None
575
+ if stats_file and (need_projection or has_predicate):
576
+ stats_df = load_stats(stats_file, allow_cache=True, profiler=prune_profiler)
577
+
411
578
  # Read-path pruning: drop raw keys whose stats prove they cannot
412
579
  # satisfy the query's WHERE before resolving them to scan URLs.
413
580
  # The span accumulates the wall-clock of the whole pruning step
414
- # (stats load + predicate eval) across every table in the query.
581
+ # (predicate eval) across every table in the query.
415
582
  with prune_profiler.span("read.prune"):
416
583
  survivors = self._prune_files(
417
- super_name, simple_name, raw_keys, stats_file,
584
+ super_name, simple_name, raw_keys, stats_df,
418
585
  profiler=prune_profiler,
419
586
  )
420
587
  files_before_prune += len(raw_keys)
421
588
  files_pruned += len(raw_keys) - len(survivors)
422
589
  files_kept += len(survivors)
423
590
 
591
+ # Projection-aware size: a query selecting specific columns scans
592
+ # only those columns' on-disk (compressed) chunks, not the whole
593
+ # multi-column file. Precise path sums per-column compressed_bytes
594
+ # from the stats artifact; files predating that column (or with no
595
+ # stats) fall back to a type-width ratio of the whole-file size.
596
+ # SELECT * keeps every column (full file).
597
+ tier3_files: Set[str] = set()
598
+ proj: Dict[str, int] = {}
599
+ if need_projection:
600
+ tier3_files, proj = self._projected_bytes_index(stats_df, selected_cols)
601
+
424
602
  parquet_files: List[str] = []
425
603
  for file_key in survivors:
426
604
  parquet_files.append(self._to_duckdb_path(file_key))
427
- reflection_file_size += key_size.get(file_key, 0)
605
+ full = int(key_size.get(file_key, 0))
606
+ reflection_file_size_raw += full
607
+ if not need_projection:
608
+ reflection_file_size += full
609
+ elif file_key in tier3_files:
610
+ reflection_file_size += proj.get(file_key, 0)
611
+ else:
612
+ reflection_file_size += self._ratio_bytes(
613
+ file_key, key_size, selected_cols, schema_types
614
+ )
428
615
 
429
616
  # SuperSnapshot is created ONCE per (super_name, simple_name) after
430
617
  # all snapshot iterations have accumulated their files and schema.
@@ -465,7 +652,11 @@ class DataEstimator:
465
652
  self.timer.capture_and_reset_timing(event="ESTIMATE")
466
653
 
467
654
  self.plan_stats.add_stat({"REFLECTIONS": total_reflections})
655
+ # REFLECTION_SIZE is the projected (selected-column) size that drives
656
+ # engine routing; REFLECTION_SIZE_RAW is the whole-file footprint, kept
657
+ # for observability so the two are comparable in the plans payload.
468
658
  self.plan_stats.add_stat({"REFLECTION_SIZE": reflection_file_size})
659
+ self.plan_stats.add_stat({"REFLECTION_SIZE_RAW": reflection_file_size_raw})
469
660
 
470
661
  # Read-path pruning observability — only when pruning is engaged, so a
471
662
  # disabled-pruning read doesn't litter the payload with noise. Mirrors
@@ -215,12 +215,43 @@ def configure_httpfs_and_s3(
215
215
  if not for_paths:
216
216
  return
217
217
 
218
- # Load httpfs; install only when not already available.
218
+ # Load httpfs. It is baked into the image and seeded into the DuckDB
219
+ # extension dir (see the container entrypoint), so LOAD normally succeeds
220
+ # with no network access.
221
+ #
222
+ # Why this is NOT a blind ``INSTALL`` fallback: ``INSTALL httpfs`` performs
223
+ # an HTTP GET to extensions.duckdb.org. On an offline / firewalled node
224
+ # that socket can stall for minutes — or hang indefinitely on a blackholed
225
+ # route — turning a should-be-instant failure into an unbounded query hang.
226
+ # So we fail fast instead:
227
+ # * SET autoinstall_known_extensions=false makes LOAD raise immediately
228
+ # when the extension is absent, rather than silently downloading it;
229
+ # * a network INSTALL is attempted ONLY when explicitly opted in via
230
+ # SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD;
231
+ # * otherwise we raise a clear, actionable error the caller returns.
219
232
  try:
220
- con.execute("LOAD httpfs;")
233
+ con.execute("SET autoinstall_known_extensions=false;")
221
234
  except Exception:
222
- con.execute("INSTALL httpfs;")
235
+ pass
236
+
237
+ try:
223
238
  con.execute("LOAD httpfs;")
239
+ except Exception as load_err:
240
+ if settings.SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD:
241
+ # Operator explicitly allowed reaching the network for a one-off
242
+ # install (e.g. an online dev box without a baked extension).
243
+ con.execute("INSTALL httpfs;")
244
+ con.execute("LOAD httpfs;")
245
+ else:
246
+ raise RuntimeError(
247
+ "DuckDB 'httpfs' extension is not available locally and network "
248
+ "auto-download is disabled, so this query cannot run. Bake/seed "
249
+ "httpfs into "
250
+ f"'{get_app_home()}/.duckdb/extensions/v<duckdb_version>/<platform>/' "
251
+ "(the container entrypoint restores it from /opt/duckdb-extensions), "
252
+ "or set SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD=true to permit a "
253
+ f"one-time online install. Underlying DuckDB error: {load_err}"
254
+ ) from load_err
224
255
 
225
256
  any_s3 = any(str(p).lower().startswith("s3://") for p in for_paths)
226
257
  any_http = any(str(p).lower().startswith(("http://", "https://")) for p in for_paths)
@@ -634,6 +665,17 @@ def init_connection(
634
665
  except Exception as e:
635
666
  logger.warning(f"[duckdb.init] home_directory pin failed: {e}")
636
667
 
668
+ # Never let DuckDB auto-DOWNLOAD an extension. Everything we need (httpfs)
669
+ # is baked/seeded into the local extension dir; an implicit network install
670
+ # would reach out to extensions.duckdb.org and can hang for minutes on an
671
+ # offline/firewalled node — turning a should-be-instant error into an
672
+ # unbounded query hang. configure_httpfs_and_s3() owns the explicit,
673
+ # opt-in install path (SUPERTABLE_DUCKDB_ALLOW_EXTENSION_DOWNLOAD).
674
+ try:
675
+ con.execute("SET autoinstall_known_extensions=false;")
676
+ except Exception as e:
677
+ logger.debug(f"[duckdb.init] disabling extension auto-install failed: {e}")
678
+
637
679
  # Resolve memory limit.
638
680
  # Single env var SUPERTABLE_DUCKDB_MEMORY_LIMIT controls both executors.
639
681
  # The `memory_limit` argument is the caller's fallback when the env var is absent.
@@ -93,6 +93,53 @@ def _to_s3a_path(file_path: str) -> str:
93
93
  return file_path
94
94
 
95
95
 
96
+ def _resolve_spark_file(storage, file_path: str) -> str:
97
+ """Resolve one snapshot file path to the form Spark should scan.
98
+
99
+ ``sup.files`` are resolved once by the estimator using the *DuckDB* presign
100
+ setting (``SUPERTABLE_DUCKDB_PRESIGNED``) and shared by every engine, so
101
+ Spark re-resolves here to make its access method depend solely on
102
+ ``SUPERTABLE_SPARK_PRESIGNED``:
103
+
104
+ * default (False) → a direct ``s3a://bucket/key`` path that Spark reads
105
+ with its own ``fs.s3a.*`` credentials (``_to_s3a_path`` already
106
+ normalises ``s3://``, presigned ``http(s)://`` and ``s3a://`` inputs);
107
+ * opt-in (True) → a freshly minted presigned ``http(s)://`` URL, so
108
+ Spark works even when the cluster has no object-store credentials — and
109
+ regardless of whether ``sup.files`` were presigned for DuckDB.
110
+
111
+ Presigning is best-effort: a missing ``presign``, a local path, or an
112
+ unparseable key falls back to the s3a form.
113
+ """
114
+ s3a = _to_s3a_path(file_path)
115
+ if not settings.SUPERTABLE_SPARK_PRESIGNED or not s3a.startswith("s3a://"):
116
+ return s3a
117
+
118
+ presign_fn = getattr(storage, "presign", None)
119
+ if not callable(presign_fn):
120
+ return s3a
121
+
122
+ # s3a://bucket/full_key → object key. storage.presign() re-applies
123
+ # base_prefix (like read_bytes), so strip it here — mirroring
124
+ # _read_parquet_schema — to avoid doubling it.
125
+ full_key = s3a[len("s3a://"):].partition("/")[2]
126
+ if not full_key:
127
+ return s3a
128
+ base = (getattr(storage, "base_prefix", "") or "").strip("/")
129
+ rel_key = (
130
+ full_key[len(base) + 1:]
131
+ if base and full_key.startswith(base + "/")
132
+ else full_key
133
+ )
134
+ try:
135
+ url = presign_fn(rel_key)
136
+ if isinstance(url, str) and url:
137
+ return url
138
+ except Exception as e: # pragma: no cover - defensive
139
+ logger.debug(f"[spark.thrift] presign failed for {rel_key!r}; using s3a: {e}")
140
+ return s3a
141
+
142
+
96
143
  # =========================================================
97
144
  # Spark SQL helpers
98
145
  # =========================================================
@@ -209,6 +256,7 @@ def _spark_create_tombstone_view(
209
256
  source_table: str,
210
257
  view_name: str,
211
258
  tombstone_def,
259
+ storage=None,
212
260
  ) -> None:
213
261
  """Create a view that hides system columns and drops tombstoned rows.
214
262
 
@@ -239,11 +287,12 @@ def _spark_create_tombstone_view(
239
287
  has_rowid = "__rowid__" in src_cols
240
288
 
241
289
  if tomb_path and has_rowid:
242
- # The data files are converted to s3a:// for Spark (see the data-file
243
- # loop in the executor); apply the same conversion to the already
244
- # resolved deletion-vector pointer so Spark reads it via its own S3A
245
- # client instead of choking on a presigned HTTP URL or bare key.
246
- escaped = _to_s3a_path(tomb_path).replace("'", "''")
290
+ # Resolve the deletion-vector pointer the same way as the data files
291
+ # (see the data-file loop in the executor) so Spark reads it through the
292
+ # same access method — direct s3a:// by default, presigned when
293
+ # SUPERTABLE_SPARK_PRESIGNED is on — instead of a bare key or a
294
+ # DuckDB-shaped presigned URL.
295
+ escaped = _resolve_spark_file(storage, tomb_path).replace("'", "''")
247
296
  sql = (
248
297
  f"CREATE OR REPLACE TEMPORARY VIEW {view_name} AS "
249
298
  f"SELECT {select_cols} FROM {source_table} AS src "
@@ -806,8 +855,10 @@ class SparkThriftExecutor:
806
855
  if sup.files and table_name not in table_repr_file:
807
856
  table_repr_file[table_name] = sup.files[0]
808
857
 
809
- # Convert to s3a:// paths for Spark (handles s3://, presigned HTTP URLs, etc.)
810
- files = [_to_s3a_path(f) for f in sup.files]
858
+ # Resolve to Spark's access form: direct s3a:// by default, or
859
+ # presigned URLs when SUPERTABLE_SPARK_PRESIGNED is on (handles
860
+ # s3://, presigned HTTP URLs and bare keys either way).
861
+ files = [_resolve_spark_file(self.storage, f) for f in sup.files]
811
862
 
812
863
  logger.debug(
813
864
  f"{log_prefix}[spark.thrift] creating view {table_name} "
@@ -898,7 +949,7 @@ class SparkThriftExecutor:
898
949
  source = query_alias_to_name[alias]
899
950
  tomb_def = tombstone_views.get(alias)
900
951
  tomb_view = f"tomb_{source}_{query_suffix}"
901
- _spark_create_tombstone_view(cursor, source, tomb_view, tomb_def)
952
+ _spark_create_tombstone_view(cursor, source, tomb_view, tomb_def, self.storage)
902
953
  created_views.append(tomb_view)
903
954
  query_alias_to_name[alias] = tomb_view
904
955