vgi-python 0.35.0__py3-none-any.whl → 0.36.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1003,7 +1003,10 @@ class FilterEchoPartitionedFunction(TableFunctionGenerator[_FilterEchoPartitione
1003
1003
  end_idx = min(start_idx + chunk, params.args.count)
1004
1004
  work_items.append(struct.pack(">QQ", start_idx, end_idx))
1005
1005
  params.storage.queue_push(work_items)
1006
- return GlobalInitResponse()
1006
+ # No more readers than work items. Left at the default (unbounded) the
1007
+ # client opens one stream per DuckDB thread -- 48 on a 48-core host for at
1008
+ # most MAX_PARTITIONS items, every extra one an init plus an empty drain.
1009
+ return GlobalInitResponse(max_workers=max(1, len(work_items)))
1007
1010
 
1008
1011
  @classmethod
1009
1012
  def initial_state(cls, params: ProcessParams[_FilterEchoPartitionedArgs]) -> _FilterEchoPartitionedState:
@@ -516,7 +516,10 @@ class PartitionedSequenceFunction(
516
516
 
517
517
  # Always enqueue (even if empty) to register the invocation
518
518
  params.storage.queue_push(work_items)
519
- return GlobalInitResponse()
519
+ # No more readers than work items. Left at the default (unbounded) the
520
+ # client opens one stream per DuckDB thread -- 48 on a 48-core host for at
521
+ # most MAX_PARTITIONS items, every extra one an init plus an empty drain.
522
+ return GlobalInitResponse(max_workers=max(1, len(work_items)))
520
523
 
521
524
  @classmethod
522
525
  def initial_state(cls, params: ProcessParams[PartitionedSequenceFunctionArguments]) -> PartitionedSequenceState:
@@ -35,6 +35,7 @@ from vgi_rpc.rpc import OutputCollector
35
35
 
36
36
  from vgi.arguments import Arg
37
37
  from vgi.cache_control import CacheControl
38
+ from vgi.filter_v2 import replaying_accepted_state
38
39
  from vgi.metadata import FunctionExample, PartitionKind
39
40
  from vgi.protocol import PlanResponse, ScanSplit, VgiOutputCollector
40
41
  from vgi.schema_utils import partition_field, schema
@@ -1055,11 +1056,15 @@ class SplitDynamicFilterFunction(TableFunctionGenerator[SplitSequenceArgs, Split
1055
1056
  init = params.init_call
1056
1057
  merged = None
1057
1058
  if init is not None and init.pushdown_filters is not None:
1058
- merged = cls.pushdown_filters(
1059
- init.pushdown_filters,
1060
- join_keys=init.join_keys,
1061
- output_schema=init.output_schema,
1062
- )
1059
+ # The framework validated these at init; re-binding every
1060
+ # predicate through the evaluator on every tick is what made
1061
+ # this fixture cost ~10 ms a batch.
1062
+ with replaying_accepted_state():
1063
+ merged = cls.pushdown_filters(
1064
+ init.pushdown_filters,
1065
+ join_keys=init.join_keys,
1066
+ output_schema=init.output_schema,
1067
+ )
1063
1068
  if merged is None:
1064
1069
  merged = params.current_pushdown_filters
1065
1070
  filter_str = _render_filters_canonical(merged)
vgi/auth.py CHANGED
@@ -18,8 +18,9 @@ JWT auth (requires ``vgi[oauth]``):
18
18
 
19
19
  from __future__ import annotations
20
20
 
21
- import contextlib
21
+ import importlib
22
22
  from collections.abc import Callable
23
+ from typing import TYPE_CHECKING, Any
23
24
 
24
25
  from vgi_rpc.rpc import AuthContext, CallContext
25
26
 
@@ -43,34 +44,89 @@ __all__ = [
43
44
  "TokenResolver",
44
45
  ]
45
46
 
46
- # HTTP auth helpers — available when vgi[http] is installed.
47
- with contextlib.suppress(ImportError):
48
- from vgi_rpc.http import ( # noqa: F401
49
- AuthUnavailableError,
50
- OAuthResourceMetadata,
51
- bearer_authenticate,
52
- bearer_authenticate_static,
53
- chain_authenticate,
54
- parse_client_id,
55
- parse_client_secret,
56
- parse_device_code_client_id,
57
- parse_device_code_client_secret,
47
+ # HTTP auth helpers (``vgi[http]``) and JWT auth (``vgi[oauth]``), resolved on
48
+ # first access rather than at import. ``vgi/__init__`` imports this module, and
49
+ # ``vgi_rpc.http`` brings in the HTTP server and client stack (falcon, httpx2,
50
+ # cryptography, joserfc): importing it here put about 180 ms on every
51
+ # ``import vgi``, and so on the start of every subprocess worker, which never
52
+ # uses it. ``from vgi.auth import bearer_authenticate`` works as before; when
53
+ # the extra is missing it raises ImportError, as it did.
54
+ _LAZY_EXPORTS = {
55
+ "AuthUnavailableError": "vgi_rpc.http",
56
+ "OAuthResourceMetadata": "vgi_rpc.http",
57
+ "bearer_authenticate": "vgi_rpc.http",
58
+ "bearer_authenticate_static": "vgi_rpc.http",
59
+ "chain_authenticate": "vgi_rpc.http",
60
+ "parse_client_id": "vgi_rpc.http",
61
+ "parse_client_secret": "vgi_rpc.http",
62
+ "parse_device_code_client_id": "vgi_rpc.http",
63
+ "parse_device_code_client_secret": "vgi_rpc.http",
64
+ "jwt_authenticate": "vgi_rpc.http._oauth_jwt",
65
+ }
66
+
67
+ if TYPE_CHECKING:
68
+ # ``X as X``: explicit re-exports, so type checkers see these as part of
69
+ # ``vgi.auth`` exactly as they were before they became lazy.
70
+ from vgi_rpc.http import (
71
+ AuthUnavailableError as AuthUnavailableError,
72
+ )
73
+ from vgi_rpc.http import (
74
+ OAuthResourceMetadata as OAuthResourceMetadata,
75
+ )
76
+ from vgi_rpc.http import (
77
+ bearer_authenticate as bearer_authenticate,
78
+ )
79
+ from vgi_rpc.http import (
80
+ bearer_authenticate_static as bearer_authenticate_static,
81
+ )
82
+ from vgi_rpc.http import (
83
+ chain_authenticate as chain_authenticate,
84
+ )
85
+ from vgi_rpc.http import (
86
+ parse_client_id as parse_client_id,
87
+ )
88
+ from vgi_rpc.http import (
89
+ parse_client_secret as parse_client_secret,
90
+ )
91
+ from vgi_rpc.http import (
92
+ parse_device_code_client_id as parse_device_code_client_id,
58
93
  )
94
+ from vgi_rpc.http import (
95
+ parse_device_code_client_secret as parse_device_code_client_secret,
96
+ )
97
+ from vgi_rpc.http._oauth_jwt import jwt_authenticate as jwt_authenticate
98
+
99
+
100
+ def __getattr__(name: str) -> Any:
101
+ """Import an HTTP or JWT auth helper on first access (PEP 562).
102
+
103
+ Args:
104
+ name: The attribute being looked up.
105
+
106
+ Returns:
107
+ The helper, which is then cached in this module's namespace.
108
+
109
+ Raises:
110
+ AttributeError: If ``name`` is not a helper this module re-exports, or
111
+ its extra is not installed.
112
+ """
113
+ module_name = _LAZY_EXPORTS.get(name)
114
+ if module_name is None:
115
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
116
+ try:
117
+ value = getattr(importlib.import_module(module_name), name)
118
+ except ImportError as exc:
119
+ raise AttributeError(
120
+ f"module {__name__!r} has no attribute {name!r}: it needs {module_name}, which failed to import ({exc})"
121
+ ) from exc
122
+ globals()[name] = value
123
+ return value
124
+
125
+
126
+ def __dir__() -> list[str]:
127
+ """List this module's names, the lazily imported helpers included.
59
128
 
60
- __all__ += [
61
- "AuthUnavailableError",
62
- "OAuthResourceMetadata",
63
- "bearer_authenticate",
64
- "bearer_authenticate_static",
65
- "chain_authenticate",
66
- "parse_client_id",
67
- "parse_client_secret",
68
- "parse_device_code_client_id",
69
- "parse_device_code_client_secret",
70
- ]
71
-
72
- # JWT auth — available when vgi[oauth] is installed (requires authlib).
73
- with contextlib.suppress(ImportError):
74
- from vgi_rpc.http._oauth_jwt import jwt_authenticate # noqa: F401
75
-
76
- __all__ += ["jwt_authenticate"]
129
+ Returns:
130
+ The module's attribute names.
131
+ """
132
+ return sorted({*globals(), *_LAZY_EXPORTS})
@@ -2698,6 +2698,8 @@ class ReadOnlyCatalogInterface(CatalogInterface):
2698
2698
  _function_registry: "dict[tuple[SchemaKey, str], list[type]] | None" = None
2699
2699
  _macro_registry: "dict[tuple[SchemaKey, str], Macro] | None" = None
2700
2700
  _index_registry: "dict[tuple[SchemaKey, str], Index] | None" = None
2701
+ # Function listings by (schema, listing type), filled in by ``_function_infos``.
2702
+ _function_info_cache: "dict[tuple[SchemaKey, SchemaObjectType], tuple[FunctionInfo, ...]] | None" = None
2701
2703
  # Lazy registry build is one-time but the fixture HTTP server is
2702
2704
  # multi-threaded and shares one catalog instance, so concurrent
2703
2705
  # first-requests can race the build. Serialize it under a lock and flip
@@ -2738,6 +2740,7 @@ class ReadOnlyCatalogInterface(CatalogInterface):
2738
2740
  self._function_registry = {}
2739
2741
  self._macro_registry = {}
2740
2742
  self._index_registry = {}
2743
+ self._function_info_cache = {}
2741
2744
 
2742
2745
  def _register_table(schema_key: SchemaKey, table: "Table") -> None:
2743
2746
  key = (schema_key, table.name.lower())
@@ -3320,29 +3323,46 @@ class ReadOnlyCatalogInterface(CatalogInterface):
3320
3323
  if sn == path_key and macro.macro_type == target_macro_type:
3321
3324
  results.append(macro.to_macro_info(schema_path))
3322
3325
  else:
3323
- # SCALAR_FUNCTION or TABLE_FUNCTION
3324
- for (sn, _), func_classes in self._function_registry.items():
3325
- if sn != path_key:
3326
- continue
3327
- for func_cls in func_classes:
3328
- func_info = self._function_to_info(func_cls, schema_path)
3329
- # Filter by function type
3330
- if type_enum == SchemaObjectType.SCALAR_FUNCTION and func_info.function_type != FunctionType.SCALAR:
3331
- continue
3332
- if type_enum == SchemaObjectType.TABLE_FUNCTION and func_info.function_type not in (
3333
- FunctionType.TABLE,
3334
- FunctionType.TABLE_BUFFERING,
3335
- ):
3336
- continue
3337
- if (
3338
- type_enum == SchemaObjectType.AGGREGATE_FUNCTION
3339
- and func_info.function_type != FunctionType.AGGREGATE
3340
- ):
3341
- continue
3342
- results.append(func_info)
3326
+ # SCALAR_FUNCTION, TABLE_FUNCTION or AGGREGATE_FUNCTION
3327
+ results.extend(self._function_infos(path_key, schema_path, type_enum))
3343
3328
 
3344
3329
  return results
3345
3330
 
3331
+ def _function_infos(
3332
+ self, path_key: SchemaKey, schema_path: SchemaPath, type_enum: SchemaObjectType
3333
+ ) -> tuple[FunctionInfo, ...]:
3334
+ """Return the schema's functions of one listing type, built on the first request.
3335
+
3336
+ A listing depends only on the registered classes' static metadata, so
3337
+ every request for it returns the same `FunctionInfo` instances. Rebuilding
3338
+ it on each request re-derived every function in the schema to keep the
3339
+ requested type's. It also made the worker encode each one again, at
3340
+ about 0.8 ms per function, because ``FunctionsResponse.from_infos``
3341
+ stores an instance's encoding on the instance.
3342
+ The instances are shared between requests, so they must not be mutated.
3343
+ """
3344
+ assert self._function_registry is not None
3345
+ assert self._function_info_cache is not None
3346
+ key = (path_key, type_enum)
3347
+ cached = self._function_info_cache.get(key)
3348
+ if cached is not None:
3349
+ return cached
3350
+ wanted = {
3351
+ SchemaObjectType.SCALAR_FUNCTION: (FunctionType.SCALAR,),
3352
+ SchemaObjectType.TABLE_FUNCTION: (FunctionType.TABLE, FunctionType.TABLE_BUFFERING),
3353
+ SchemaObjectType.AGGREGATE_FUNCTION: (FunctionType.AGGREGATE,),
3354
+ }.get(type_enum, ())
3355
+ infos: list[FunctionInfo] = []
3356
+ for (sn, _), func_classes in self._function_registry.items():
3357
+ if sn != path_key:
3358
+ continue
3359
+ for func_cls in func_classes:
3360
+ func_info = self._function_to_info(func_cls, schema_path)
3361
+ if func_info.function_type in wanted:
3362
+ infos.append(func_info)
3363
+ # Concurrent first requests may both build it; keep whichever landed first.
3364
+ return self._function_info_cache.setdefault(key, tuple(infos))
3365
+
3346
3366
  def copy_from_formats(
3347
3367
  self,
3348
3368
  *,
vgi/client/client.py CHANGED
@@ -33,10 +33,14 @@ the HTTP flow and skip the subprocess branch.
33
33
 
34
34
  Parallel processing
35
35
  -------------------
36
- When a bind returns ``max_workers > 1`` the client spawns additional
37
- worker connections and distributes input batches round-robin. Output
38
- order is non-deterministic in parallel mode. This is optimization; a
39
- minimal port can ignore it and always use one connection.
36
+ When an init returns ``max_workers > 1`` the client may use that many
37
+ worker connections, capped by the CPU count and ``worker_limit``. For
38
+ scalar and table-in-out calls it deals input batches round-robin and opens
39
+ a secondary connection only when the first batch is dealt to it, so ``n``
40
+ input batches use at most ``n`` connections. A table function has no input,
41
+ so it opens every allowed connection at init. Output order is
42
+ non-deterministic in parallel mode. This is optimization; a minimal port
43
+ can ignore it and always use one connection.
40
44
 
41
45
  Key classes
42
46
  -----------
@@ -258,6 +262,37 @@ class WorkerConnection:
258
262
  _launch_ctx: AbstractContextManager[Any] | None = field(default=None, repr=False)
259
263
 
260
264
 
265
+ @dataclass(frozen=True, slots=True)
266
+ class _FanOut:
267
+ """How far a stream may fan out, and the init every secondary connection sends.
268
+
269
+ Built from the primary's bind and init. A secondary connection's init
270
+ echoes the primary's request plus its ``execution_id`` and init opaque
271
+ data, so all connections are parts of one execution.
272
+
273
+ Attributes:
274
+ max_workers: The total connections allowed, the primary included, after
275
+ ``_determine_max_workers`` applies the client's limits.
276
+ bind_request: The primary's bind request, carried inside each init.
277
+ bind_response: The primary's bind response.
278
+ init_response: The primary's init header. Supplies ``execution_id`` and
279
+ ``opaque_data``.
280
+ projection_ids: Projection, repeated on every secondary init.
281
+ pushdown_filters_batch: Pushed-down filters, repeated on every secondary init.
282
+ phase: Table-in-out phase, or ``None`` for other function kinds.
283
+ join_keys: Join-key batches, repeated on every secondary init.
284
+ """
285
+
286
+ max_workers: int
287
+ bind_request: BindRequest
288
+ bind_response: BindResponse
289
+ init_response: GlobalInitResponse
290
+ projection_ids: list[int] | None
291
+ pushdown_filters_batch: pa.RecordBatch | None
292
+ phase: TableInOutFunctionInitPhase | None
293
+ join_keys: list[pa.RecordBatch] | None
294
+
295
+
261
296
  class Client(CatalogClientMixin, AggregateClientMixin):
262
297
  """Canonical VGI client — HTTP is the path other-language ports mirror.
263
298
 
@@ -1731,7 +1766,8 @@ class Client(CatalogClientMixin, AggregateClientMixin):
1731
1766
  join_keys: list[pa.RecordBatch] | None = None,
1732
1767
  at_unit: str | None = None,
1733
1768
  at_value: str | None = None,
1734
- ) -> tuple[BindRequest, BindResponse, GlobalInitResponse]:
1769
+ fan_out_on_demand: bool = False,
1770
+ ) -> tuple[BindRequest, BindResponse, GlobalInitResponse, _FanOut]:
1735
1771
  """Run the canonical bind → init → fan-out-workers sequence.
1736
1772
 
1737
1773
  All three function entry points (``scalar_function``,
@@ -1744,7 +1780,16 @@ class Client(CatalogClientMixin, AggregateClientMixin):
1744
1780
  4. Read the `[`GlobalInitResponse`][]` header (carries ``max_workers``
1745
1781
  + ``execution_id`` for secondary workers).
1746
1782
  5. Spawn any additional workers and drive their ``init`` with the
1747
- primary's execution identity.
1783
+ primary's execution identity — unless ``fan_out_on_demand``, in
1784
+ which case the returned `_FanOut` is handed to
1785
+ ``_distribute_and_collect``, which opens a secondary connection only
1786
+ when an input batch is dealt to it. Input-driven calls (scalar and
1787
+ table-in-out) fan out that way: the server's ``max_workers`` is an
1788
+ upper bound (by default unbounded, so the client's CPU count), and
1789
+ opening that many connections up front for a call with one input
1790
+ batch costs a worker process each on the subprocess transport, all
1791
+ but one of them idle. ``table_function`` has no input to deal, so
1792
+ its connections are all opened here.
1748
1793
 
1749
1794
  Centralizing this keeps HTTP/subprocess differences and protocol
1750
1795
  changes (e.g. future scoped-secret re-bind, init hints) in one
@@ -1809,69 +1854,71 @@ class Client(CatalogClientMixin, AggregateClientMixin):
1809
1854
 
1810
1855
  init_response = stream.typed_header(GlobalInitResponse)
1811
1856
  max_workers = 1 if split_tokens is not None else self._determine_max_workers(init_response.max_workers)
1812
-
1813
- self._spawn_additional_workers(
1814
- max_workers,
1815
- bind_request,
1816
- bind_response,
1817
- init_response,
1857
+ fan_out = _FanOut(
1858
+ max_workers=max_workers,
1859
+ bind_request=bind_request,
1860
+ bind_response=bind_response,
1861
+ init_response=init_response,
1818
1862
  projection_ids=projection_ids,
1819
1863
  pushdown_filters_batch=pushdown_filters_batch,
1820
1864
  phase=phase,
1821
1865
  join_keys=join_keys,
1822
1866
  )
1867
+ if not fan_out_on_demand:
1868
+ self._spawn_additional_workers(fan_out)
1823
1869
 
1824
- return bind_request, bind_response, init_response
1870
+ return bind_request, bind_response, init_response, fan_out
1825
1871
 
1826
- def _spawn_additional_workers(
1827
- self,
1828
- max_workers: int,
1829
- bind_request: BindRequest,
1830
- bind_response: BindResponse,
1831
- global_init_response: GlobalInitResponse,
1832
- *,
1833
- projection_ids: list[int] | None = None,
1834
- pushdown_filters_batch: pa.RecordBatch | None = None,
1835
- phase: TableInOutFunctionInitPhase | None = None,
1836
- join_keys: list[pa.RecordBatch] | None = None,
1837
- ) -> None:
1838
- """Spawn and initialize additional worker subprocesses in parallel.
1872
+ def _init_secondary_worker(self, worker: WorkerConnection, fan_out: _FanOut) -> None:
1873
+ """Send a secondary worker its ``init``, making it part of the primary's execution.
1874
+
1875
+ Args:
1876
+ worker: The newly opened secondary connection.
1877
+ fan_out: The primary's bind and init, which the secondary init repeats.
1878
+ """
1879
+ worker.substream_id = _substream_id_for(fan_out.phase)
1880
+ worker.stream = self._do_init(
1881
+ worker.proxy,
1882
+ fan_out.bind_request,
1883
+ fan_out.bind_response,
1884
+ projection_ids=fan_out.projection_ids,
1885
+ pushdown_filters_batch=fan_out.pushdown_filters_batch,
1886
+ phase=fan_out.phase,
1887
+ join_keys=fan_out.join_keys,
1888
+ execution_id=fan_out.init_response.execution_id,
1889
+ init_opaque_data=fan_out.init_response.opaque_data,
1890
+ substream_id=worker.substream_id,
1891
+ )
1839
1892
 
1840
- First spawns all worker subprocesses sequentially (fast operation), then
1841
- initializes all workers in parallel using threads. Each additional worker
1842
- receives a secondary init with the execution_id from the primary worker.
1893
+ def _spawn_additional_workers(self, fan_out: _FanOut) -> None:
1894
+ """Open and initialize every secondary connection ``fan_out`` allows, in parallel.
1843
1895
 
1844
- The spawned workers are appended to self._additional_workers list.
1896
+ Opens the connections one after another, which is fast: a subprocess is
1897
+ only started here, not waited for. Then initializes them in parallel
1898
+ threads, overlapping the workers' startup. Each secondary init carries
1899
+ the primary's ``execution_id``. The connections are appended to
1900
+ ``self._additional_workers``.
1845
1901
 
1846
- If max_workers is 1 or less, this method returns immediately without
1847
- spawning any workers.
1902
+ ``table_function`` fans out this way because it has no input to deal.
1903
+ Input-driven calls use `_distribute_and_collect`, which opens a
1904
+ connection only when a batch is dealt to it. When ``max_workers`` is 1
1905
+ or less, this method returns without opening anything.
1848
1906
 
1849
1907
  Args:
1850
- max_workers: Total number of workers desired (including the primary
1851
- worker). For example, if max_workers=4, this method spawns
1852
- 3 additional workers (indices 1, 2, 3).
1853
- bind_request: The original bind request to embed in init.
1854
- bind_response: The bind response with output schema.
1855
- global_init_response: The primary worker's init response containing
1856
- execution_id and opaque_data for secondary init.
1857
- projection_ids: Optional column indices for projection.
1858
- pushdown_filters_batch: Optional deserialized filter predicates.
1859
- phase: Table-in-out function phase (INPUT or FINALIZE).
1860
- join_keys: Optional serialized join-key batches pushed down for
1861
- join filtering, echoed to every secondary worker exactly like
1862
- ``pushdown_filters_batch``.
1908
+ fan_out: The connection limit and the primary's bind and init, which
1909
+ every secondary init repeats.
1863
1910
 
1864
1911
  Raises:
1865
1912
  [`ClientError`][]: If any worker fails to initialize. The exception wraps
1866
1913
  the first initialization error encountered.
1867
1914
 
1868
1915
  """
1869
- if max_workers <= 1:
1916
+ if fan_out.max_workers <= 1:
1870
1917
  return
1871
1918
 
1872
1919
  # Spawn all worker subprocesses first (fast)
1873
1920
  new_workers: list[WorkerConnection] = []
1874
- for worker_index in range(1, max_workers):
1921
+ for worker_index in range(1, fan_out.max_workers):
1875
1922
  worker = self._spawn_worker(worker_index)
1876
1923
  new_workers.append(worker)
1877
1924
  self._additional_workers.append(worker)
@@ -1881,20 +1928,7 @@ class Client(CatalogClientMixin, AggregateClientMixin):
1881
1928
 
1882
1929
  def do_init(worker: WorkerConnection) -> None:
1883
1930
  try:
1884
- worker.substream_id = _substream_id_for(phase)
1885
- stream = self._do_init(
1886
- worker.proxy,
1887
- bind_request,
1888
- bind_response,
1889
- projection_ids=projection_ids,
1890
- pushdown_filters_batch=pushdown_filters_batch,
1891
- phase=phase,
1892
- join_keys=join_keys,
1893
- execution_id=global_init_response.execution_id,
1894
- init_opaque_data=global_init_response.opaque_data,
1895
- substream_id=worker.substream_id,
1896
- )
1897
- worker.stream = stream
1931
+ self._init_secondary_worker(worker, fan_out)
1898
1932
  except Exception as e:
1899
1933
  init_errors.append(e)
1900
1934
 
@@ -2111,13 +2145,16 @@ class Client(CatalogClientMixin, AggregateClientMixin):
2111
2145
  output_queue: Queue[tuple[int, list[pa.RecordBatch], list[list[int]] | None] | BaseException],
2112
2146
  *,
2113
2147
  decode_parent_rows: bool = False,
2148
+ init: _FanOut | None = None,
2114
2149
  ) -> None:
2115
2150
  """Thread function that processes batches for a single worker.
2116
2151
 
2117
2152
  Runs in a dedicated thread, pulling (batch_index, batch) tuples from
2118
2153
  the input queue, processing them via _process_batch_on_worker, and
2119
2154
  pushing (batch_index, output_batches, parent_rows_batches) tuples to
2120
- the output queue.
2155
+ the output queue. With ``init``, first sends the worker its secondary
2156
+ init, so a connection opened on demand starts up without holding up
2157
+ the thread dealing the batches.
2121
2158
 
2122
2159
  When None is received from input_queue, signals thread completion by
2123
2160
  pushing (-1, [], None) to output_queue and exits.
@@ -2132,9 +2169,13 @@ class Client(CatalogClientMixin, AggregateClientMixin):
2132
2169
  output_queue: Thread-safe queue for results.
2133
2170
  decode_parent_rows: Forwarded to `_process_batch_on_worker` — see
2134
2171
  its docstring.
2172
+ init: The fan-out plan to initialize a just-opened secondary worker
2173
+ from, or ``None`` for a worker already initialized.
2135
2174
 
2136
2175
  """
2137
2176
  try:
2177
+ if init is not None:
2178
+ self._init_secondary_worker(worker, init)
2138
2179
  while True:
2139
2180
  item = input_queue.get()
2140
2181
  if item is None:
@@ -2154,7 +2195,7 @@ class Client(CatalogClientMixin, AggregateClientMixin):
2154
2195
  def _distribute_and_collect(
2155
2196
  self,
2156
2197
  *,
2157
- all_workers: list[WorkerConnection],
2198
+ fan_out: _FanOut,
2158
2199
  first_batch: pa.RecordBatch,
2159
2200
  remaining_input: Iterator[pa.RecordBatch],
2160
2201
  decode_parent_rows: bool = False,
@@ -2166,8 +2207,16 @@ class Client(CatalogClientMixin, AggregateClientMixin):
2166
2207
  worker, spawns a dedicated thread that pulls batches from an input queue,
2167
2208
  sends them to the worker, and pushes results to a shared output queue.
2168
2209
 
2210
+ Batch ``i`` goes to connection ``i % fan_out.max_workers``. The primary
2211
+ is connection 0. A secondary connection is opened, and sent its init,
2212
+ when the first batch is dealt to it, so a call with ``n`` input batches
2213
+ uses ``min(n, max_workers)`` connections. It never opens one that would
2214
+ get no input. Opened connections are appended to
2215
+ ``self._additional_workers``, where the caller closes them.
2216
+
2169
2217
  Args:
2170
- all_workers: List of all workers (primary + additional).
2218
+ fan_out: The primary's init and the connection limit, from
2219
+ `_initialize_stream_common` with ``fan_out_on_demand=True``.
2171
2220
  first_batch: The first input batch, already consumed from the
2172
2221
  iterator by the calling method.
2173
2222
  remaining_input: Iterator for remaining input batches.
@@ -2190,42 +2239,48 @@ class Client(CatalogClientMixin, AggregateClientMixin):
2190
2239
  [`ClientError`][]: If a worker thread fails with an exception.
2191
2240
 
2192
2241
  """
2193
- num_workers = len(all_workers)
2194
-
2195
- _logger.debug("starting_parallel_processing num_workers=%s", num_workers)
2196
-
2197
- # Create queues for each worker
2198
- input_queues: list[Queue[tuple[int, pa.RecordBatch] | None]] = [Queue() for _ in range(num_workers)]
2242
+ assert self._primary is not None
2243
+ max_workers = max(fan_out.max_workers, 1)
2244
+ input_queues: list[Queue[tuple[int, pa.RecordBatch] | None]] = []
2199
2245
  output_queue: Queue[tuple[int, list[pa.RecordBatch], list[list[int]] | None] | BaseException] = Queue()
2200
-
2201
- # Start worker threads
2202
2246
  threads: list[threading.Thread] = []
2203
- for i, worker in enumerate(all_workers):
2247
+
2248
+ def start_worker_thread(worker: WorkerConnection, init: _FanOut | None) -> None:
2249
+ input_queue: Queue[tuple[int, pa.RecordBatch] | None] = Queue()
2250
+ input_queues.append(input_queue)
2204
2251
  thread = threading.Thread(
2205
2252
  target=self._worker_thread_loop,
2206
- args=(worker, input_queues[i], output_queue),
2207
- kwargs={"decode_parent_rows": decode_parent_rows},
2253
+ args=(worker, input_queue, output_queue),
2254
+ kwargs={"decode_parent_rows": decode_parent_rows, "init": init},
2208
2255
  daemon=True,
2209
2256
  )
2210
2257
  thread.start()
2211
2258
  threads.append(thread)
2212
2259
 
2213
- # Distribute batches round-robin across workers
2214
- batch_index = 0
2215
- batches_sent = 0
2216
-
2217
- # Send first batch
2218
- worker_idx = batch_index % num_workers
2219
- input_queues[worker_idx].put((batch_index, first_batch))
2220
- batches_sent += 1
2221
- batch_index += 1
2260
+ start_worker_thread(self._primary, None)
2222
2261
 
2223
- # Send remaining batches
2224
- for input_batch in remaining_input:
2225
- worker_idx = batch_index % num_workers
2226
- input_queues[worker_idx].put((batch_index, input_batch))
2227
- batches_sent += 1
2228
- batch_index += 1
2262
+ # Deal batches round-robin, opening a connection the first time one is
2263
+ # dealt to it. Opening is quick (a subprocess is started, not waited
2264
+ # for); the secondary init runs on the connection's own thread.
2265
+ batches_sent = 0
2266
+ try:
2267
+ for batch_index, input_batch in enumerate(itertools.chain((first_batch,), remaining_input)):
2268
+ worker_idx = batch_index % max_workers
2269
+ if worker_idx == len(input_queues):
2270
+ worker = self._spawn_worker(worker_idx)
2271
+ self._additional_workers.append(worker)
2272
+ start_worker_thread(worker, fan_out)
2273
+ input_queues[worker_idx].put((batch_index, input_batch))
2274
+ batches_sent += 1
2275
+ except BaseException:
2276
+ # The input or a connection failed: let the started threads finish
2277
+ # what they were dealt and exit rather than wait forever.
2278
+ for q in input_queues:
2279
+ q.put(None)
2280
+ raise
2281
+
2282
+ num_workers = len(input_queues)
2283
+ _logger.debug("starting_parallel_processing num_workers=%s", num_workers)
2229
2284
 
2230
2285
  # Signal end of input to all workers
2231
2286
  for q in input_queues:
@@ -2377,7 +2432,7 @@ class Client(CatalogClientMixin, AggregateClientMixin):
2377
2432
  input_schema = first_batch.schema
2378
2433
  pushdown_filters_batch = self._deserialize_pushdown_filters(pushdown_filters)
2379
2434
 
2380
- bind_request, bind_response, init_response = self._initialize_stream_common(
2435
+ bind_request, bind_response, init_response, fan_out = self._initialize_stream_common(
2381
2436
  function_name=function_name,
2382
2437
  schema_path=schema_path,
2383
2438
  arguments=arguments,
@@ -2391,12 +2446,12 @@ class Client(CatalogClientMixin, AggregateClientMixin):
2391
2446
  phase=TableInOutFunctionInitPhase.INPUT,
2392
2447
  bind_result_callback=bind_result_callback,
2393
2448
  join_keys=join_keys,
2449
+ fan_out_on_demand=True,
2394
2450
  )
2395
2451
 
2396
2452
  # Process input batches across all workers
2397
- all_workers = [self._primary] + self._additional_workers
2398
2453
  yield from self._distribute_and_collect(
2399
- all_workers=all_workers,
2454
+ fan_out=fan_out,
2400
2455
  first_batch=first_batch,
2401
2456
  remaining_input=input,
2402
2457
  decode_parent_rows=parent_row_callback is not None,
@@ -2404,7 +2459,7 @@ class Client(CatalogClientMixin, AggregateClientMixin):
2404
2459
  )
2405
2460
 
2406
2461
  # Close all input streams
2407
- for worker in all_workers:
2462
+ for worker in [self._primary, *self._additional_workers]:
2408
2463
  if worker.stream is not None:
2409
2464
  worker.stream.close()
2410
2465
  worker.stream = None
@@ -3492,7 +3547,7 @@ class Client(CatalogClientMixin, AggregateClientMixin):
3492
3547
 
3493
3548
  input_schema = first_batch.schema
3494
3549
 
3495
- self._initialize_stream_common(
3550
+ *_, fan_out = self._initialize_stream_common(
3496
3551
  function_name=function_name,
3497
3552
  schema_path=schema_path,
3498
3553
  arguments=arguments,
@@ -3505,18 +3560,18 @@ class Client(CatalogClientMixin, AggregateClientMixin):
3505
3560
  pushdown_filters_batch=None,
3506
3561
  phase=None,
3507
3562
  bind_result_callback=bind_result_callback,
3563
+ fan_out_on_demand=True,
3508
3564
  )
3509
3565
 
3510
3566
  # Process batches across all workers
3511
- all_workers = [self._primary] + self._additional_workers
3512
3567
  yield from self._distribute_and_collect(
3513
- all_workers=all_workers,
3568
+ fan_out=fan_out,
3514
3569
  first_batch=first_batch,
3515
3570
  remaining_input=input,
3516
3571
  )
3517
3572
 
3518
3573
  # Close streams and secondary workers
3519
- for worker in all_workers:
3574
+ for worker in [self._primary, *self._additional_workers]:
3520
3575
  if worker.stream is not None:
3521
3576
  worker.stream.close()
3522
3577
  worker.stream = None
vgi/filter_v2.py CHANGED
@@ -4,9 +4,12 @@
4
4
 
5
5
  from __future__ import annotations
6
6
 
7
+ import contextlib
7
8
  import json
8
9
  import re
9
10
  import threading
11
+ from collections.abc import Iterator
12
+ from contextvars import ContextVar
10
13
  from dataclasses import dataclass
11
14
  from enum import StrEnum
12
15
  from typing import Any
@@ -47,6 +50,54 @@ _KNOWN_ARROW_EXTENSIONS = {
47
50
  }
48
51
 
49
52
 
53
+ #: Set while re-deriving filter state that this worker already accepted (see
54
+ #: :func:`replaying_accepted_state`). Default off: every document arriving from a
55
+ #: client is validated in full.
56
+ _REPLAYING_ACCEPTED_STATE: ContextVar[bool] = ContextVar("vgi_filter_v2_replaying_accepted_state", default=False)
57
+
58
+
59
+ @contextlib.contextmanager
60
+ def replaying_accepted_state() -> Iterator[None]:
61
+ """Re-parse filter documents this worker has already validated and accepted.
62
+
63
+ Validating a predicate binds it against the output schema through the
64
+ embedded DuckDB evaluator -- one query, several milliseconds, per predicate.
65
+ The spec asks for that once per accepted revision. An HTTP worker, though,
66
+ rebuilds its filter state from the call and cursor tokens on every turn, and
67
+ re-validating there made every continuation pay a bind per predicate (and,
68
+ with dynamic filters, one per delta ever applied). Those tokens are sealed by
69
+ the worker, so the documents in them are exactly the ones it validated when
70
+ they first arrived: inside this context the parse is still strict about
71
+ structure, but the redundant bind is skipped.
72
+
73
+ Yields:
74
+ None.
75
+
76
+ """
77
+ token = _REPLAYING_ACCEPTED_STATE.set(True)
78
+ try:
79
+ yield
80
+ finally:
81
+ _REPLAYING_ACCEPTED_STATE.reset(token)
82
+
83
+
84
+ def delta_revisions(batch: pa.RecordBatch) -> tuple[tuple[str, int], ...]:
85
+ """Return the ``(id, revision)`` of every update a delta batch carries.
86
+
87
+ A structural read only, for bookkeeping over deltas that were already
88
+ applied (and therefore validated); it does not apply or validate anything.
89
+
90
+ Args:
91
+ batch: A delta document batch.
92
+
93
+ Returns:
94
+ One ``(id, revision)`` pair per update, in document order.
95
+
96
+ """
97
+ document = json.loads(batch.column(0)[0].as_py())
98
+ return tuple((update["id"], update["revision"]) for update in document["updates"])
99
+
100
+
50
101
  class FilterV2Error(ValueError):
51
102
  """A Filter Encoding v2 document is malformed or cannot be evaluated."""
52
103
 
@@ -666,7 +717,11 @@ class _Parser:
666
717
  raise FilterV2Error("runtime_filter predicates must be advisory")
667
718
  elif not self._is_boolean(expression):
668
719
  raise FilterV2Error("predicate root must resolve to BOOLEAN")
669
- if self.output_schema is not None and not isinstance(expression, RuntimeFilter):
720
+ if (
721
+ self.output_schema is not None
722
+ and not isinstance(expression, RuntimeFilter)
723
+ and not _REPLAYING_ACCEPTED_STATE.get()
724
+ ):
670
725
  empty = pa.RecordBatch.from_arrays(
671
726
  [pa.array([], type=field.type) for field in self.output_schema],
672
727
  schema=self.output_schema,
@@ -1110,7 +1165,12 @@ def _get_evaluation_connection(context: EvaluationContext) -> Any:
1110
1165
  f"found {engine_name} {engine_version or '<unknown>'}"
1111
1166
  )
1112
1167
 
1113
- connection = connect()
1168
+ # One thread. Each evaluates one batch against one predicate, which
1169
+ # DuckDB cannot parallelize, and there is one of these databases per
1170
+ # (server thread, context): at the default of one scheduler thread per
1171
+ # core, a waitress worker on a 48-core host carried ~290 threads, and
1172
+ # every evaluation paid to dispatch onto that pool.
1173
+ connection = connect(config={"threads": 1})
1114
1174
  _apply_context(connection, context)
1115
1175
  connections[key] = connection
1116
1176
  return connection
vgi/protocol.py CHANGED
@@ -63,6 +63,7 @@ from vgi.catalog.catalog_interface import (
63
63
  TableInfo,
64
64
  ViewInfo,
65
65
  )
66
+ from vgi.filter_v2 import delta_revisions, replaying_accepted_state
66
67
  from vgi.function import StreamStateCodec, _is_state_codec
67
68
  from vgi.function_storage import BoundStorage, FrameworkNS, attach_catalog_bytes
68
69
  from vgi.invocation import BindResponse, FunctionType, GlobalInitResponse
@@ -842,6 +843,37 @@ class TransactionBeginResponse(ArrowSerializableDataclass):
842
843
  transaction_opaque_data: bytes | None = None
843
844
 
844
845
 
846
+ #: Instance attribute where `_item_ipc_bytes` keeps an item's encoding.
847
+ _ITEM_IPC_MEMO = "_vgi_item_ipc"
848
+
849
+
850
+ def _item_ipc_bytes(info: Any) -> bytes:
851
+ """Return a catalog item's Arrow IPC encoding, encoded once per instance.
852
+
853
+ A catalog item is a frozen dataclass, so its encoding cannot change, and
854
+ encoding one costs about 0.8 ms (a one-row batch of some 40 columns, plus
855
+ its schema). A catalog that returns the same instances on every listing,
856
+ as ``CatalogInterface`` does for functions, therefore pays for the encoding
857
+ once. The encoding is kept in the instance's ``__dict__``, as
858
+ ``functools.cached_property`` does on a frozen dataclass. It is not a field,
859
+ so it takes no part in equality, ``repr``, ``dataclasses.replace`` or
860
+ serialization. An item without a ``__dict__`` is encoded on every call.
861
+
862
+ Args:
863
+ info: The catalog item (an ``ArrowSerializableDataclass``).
864
+
865
+ Returns:
866
+ The item's Arrow IPC stream bytes.
867
+ """
868
+ memo = getattr(info, "__dict__", None)
869
+ if memo is None:
870
+ return info.serialize_to_bytes() # type: ignore[no-any-return]
871
+ encoded = memo.get(_ITEM_IPC_MEMO)
872
+ if encoded is None:
873
+ encoded = memo[_ITEM_IPC_MEMO] = info.serialize_to_bytes()
874
+ return encoded # type: ignore[no-any-return]
875
+
876
+
845
877
  def _catalog_items_response(item_type: type) -> type:
846
878
  """Generate a catalog items response class for the given `ArrowSerializableDataclass` type.
847
879
 
@@ -868,13 +900,13 @@ def _catalog_items_response(item_type: type) -> type:
868
900
 
869
901
  @staticmethod
870
902
  def from_infos(infos: list) -> _Response: # type: ignore[type-arg]
871
- return _Response(items=[info.serialize_to_bytes() for info in infos])
903
+ return _Response(items=[_item_ipc_bytes(info) for info in infos])
872
904
 
873
905
  @staticmethod
874
906
  def from_optional(info: object | None) -> _Response:
875
907
  if info is None:
876
908
  return _Response(items=[])
877
- return _Response(items=[info.serialize_to_bytes()]) # type: ignore[attr-defined]
909
+ return _Response(items=[_item_ipc_bytes(info)])
878
910
 
879
911
  def to_infos(self) -> list: # type: ignore[type-arg]
880
912
  return [item_type.deserialize_from_bytes(b) for b in self.items] # type: ignore[attr-defined]
@@ -1827,17 +1859,84 @@ class _TrackingOutputCollector:
1827
1859
  return getattr(self._inner, name)
1828
1860
 
1829
1861
 
1830
- def _apply_filter_delta_bytes(current: Any, filter_bytes: bytes) -> Any:
1831
- """Decode one IPC delta and apply it to existing immutable filter state."""
1832
- if current is None:
1833
- raise ValueError("dynamic filter delta received without an initial snapshot")
1862
+ def _decode_filter_delta(filter_bytes: bytes) -> pa.RecordBatch:
1863
+ """Decode one IPC dynamic-filter delta into its single document batch."""
1834
1864
  if len(filter_bytes) > (17 << 20):
1835
1865
  raise ValueError("dynamic filter IPC payload exceeds the encoded-size limit")
1836
1866
  table = pa.ipc.open_stream(filter_bytes).read_all()
1837
1867
  batches = table.to_batches()
1838
1868
  if len(batches) != 1:
1839
1869
  raise ValueError("dynamic filter metadata must contain exactly one RecordBatch")
1840
- return current.apply_delta(batches[0])
1870
+ return batches[0]
1871
+
1872
+
1873
+ def _apply_filter_delta_bytes(current: Any, filter_bytes: bytes) -> Any:
1874
+ """Decode one IPC delta and apply it to existing immutable filter state."""
1875
+ if current is None:
1876
+ raise ValueError("dynamic filter delta received without an initial snapshot")
1877
+ return current.apply_delta(_decode_filter_delta(filter_bytes))
1878
+
1879
+
1880
+ def _record_filter_delta(current: Any, history: list[bytes], filter_bytes: bytes) -> tuple[Any, list[bytes], list[str]]:
1881
+ """Apply one tick's dynamic-filter delta; return the new filters and what a cursor must carry.
1882
+
1883
+ An HTTP stream cannot keep parsed filters between turns, so its cursor
1884
+ carries the deltas it has applied and :func:`_replay_filter_history` rebuilds
1885
+ the state from them on the next turn. Carrying *every* delta made each turn
1886
+ replay all of its predecessors -- quadratic in the tick count, and a Top-N
1887
+ scan gets a delta on nearly every tick (``dynamic_filter.test`` took ~200 s).
1888
+
1889
+ So the history is compacted to the deltas that installed some predicate's
1890
+ *current* revision, tombstones included: for each ``(id, revision)`` of the
1891
+ live state, the first delta carrying it. Replaying just those reproduces the
1892
+ same predicates, values and revisions -- earlier updates to an ID are
1893
+ overwritten by its current revision, later ones were stale and stay stale --
1894
+ and the history stays bounded by the number of predicate IDs instead of
1895
+ growing per tick. Replay cannot always reproduce predicate *order* (an ID
1896
+ removed and later re-added moves to the end), so the order is returned
1897
+ alongside to be restored explicitly.
1898
+
1899
+ Args:
1900
+ current: The stream's current ``PushdownFilters``.
1901
+ history: The compacted deltas applied so far.
1902
+ filter_bytes: The IPC bytes of the delta this tick carried.
1903
+
1904
+ Returns:
1905
+ ``(filters, history, order)``: the updated filters, the compacted
1906
+ history including this delta if it is still needed, and the live
1907
+ predicate IDs in order.
1908
+
1909
+ """
1910
+ updated = _apply_filter_delta_bytes(current, filter_bytes)
1911
+ wanted = set(updated._revision_map().items())
1912
+ kept: list[bytes] = []
1913
+ for delta in [*history, filter_bytes]:
1914
+ carried = set(delta_revisions(_decode_filter_delta(delta))) & wanted
1915
+ if carried:
1916
+ kept.append(delta)
1917
+ wanted -= carried
1918
+ return updated, kept, [p.id for p in updated.predicates]
1919
+
1920
+
1921
+ def _replay_filter_history(filters: Any, history: list[bytes], order: list[str]) -> Any:
1922
+ """Rebuild a turn's filters from the init snapshot plus the carried deltas.
1923
+
1924
+ Runs inside :func:`vgi.filter_v2.replaying_accepted_state` (see the call
1925
+ sites): every document here was validated when it first arrived, and both
1926
+ tokens are sealed by the worker.
1927
+
1928
+ Args:
1929
+ filters: The ``PushdownFilters`` parsed from the init snapshot.
1930
+ history: The compacted deltas from :func:`_record_filter_delta`.
1931
+ order: The live predicate order recorded alongside them.
1932
+
1933
+ Returns:
1934
+ The stream's current ``PushdownFilters``.
1935
+
1936
+ """
1937
+ for filter_bytes in history:
1938
+ filters = _apply_filter_delta_bytes(filters, filter_bytes)
1939
+ return filters._with_predicate_order(order) if history else filters
1841
1940
 
1842
1941
 
1843
1942
  @dataclass
@@ -1868,7 +1967,10 @@ class TableProducerState(_VgiCallStateHolder, ProducerState):
1868
1967
  _user_state: Annotated[Any, Transient()] = field(default=None, repr=False)
1869
1968
  _pushdown_filters: Annotated[Any, Transient()] = field(default=None, repr=False) # PushdownFilters | None
1870
1969
  _auto_apply: Annotated[bool, Transient()] = field(default=False, repr=False)
1970
+ # The dynamic-filter deltas this stream must replay on an HTTP turn, compacted by
1971
+ # _record_filter_delta, and the live predicate order they rebuild.
1871
1972
  _filter_delta_history: list[bytes] = field(default_factory=list, repr=False)
1973
+ _filter_predicate_order: list[str] = field(default_factory=list, repr=False)
1872
1974
  _vgi_tracer: Annotated[VgiTracer, Transient()] = field(default_factory=get_noop_tracer, repr=False)
1873
1975
  # Conditional-revalidation validators read off the first tick's custom_metadata
1874
1976
  # and surfaced to the generator via ProcessParams (M6). None on a normal call.
@@ -1935,13 +2037,17 @@ class TableProducerState(_VgiCallStateHolder, ProducerState):
1935
2037
  # manual consumers see the same required/current state after HTTP rehydrate.
1936
2038
  self._auto_apply = func_cls._should_auto_apply_filters()
1937
2039
  if self._call.init_call.pushdown_filters is not None:
1938
- self._pushdown_filters = func_cls.pushdown_filters(
1939
- self._call.init_call.pushdown_filters,
1940
- join_keys=self._call.init_call.join_keys,
1941
- output_schema=self._call.init_call.output_schema,
1942
- )
1943
- for filter_bytes in self._filter_delta_history:
1944
- self._pushdown_filters = _apply_filter_delta_bytes(self._pushdown_filters, filter_bytes)
2040
+ # Everything re-parsed here was validated when it first arrived.
2041
+ with replaying_accepted_state():
2042
+ self._pushdown_filters = _replay_filter_history(
2043
+ func_cls.pushdown_filters(
2044
+ self._call.init_call.pushdown_filters,
2045
+ join_keys=self._call.init_call.join_keys,
2046
+ output_schema=self._call.init_call.output_schema,
2047
+ ),
2048
+ self._filter_delta_history,
2049
+ self._filter_predicate_order,
2050
+ )
1945
2051
  self._params = dataclasses.replace(self._params, current_pushdown_filters=self._pushdown_filters)
1946
2052
  # Restore _user_state from serialized bytes if available
1947
2053
  if self._user_state_bytes is not None:
@@ -1978,8 +2084,9 @@ class TableProducerState(_VgiCallStateHolder, ProducerState):
1978
2084
  if len(encoded_filters) > (24 << 20):
1979
2085
  raise ValueError("base64 dynamic filter metadata exceeds the encoded-size limit")
1980
2086
  filter_bytes = base64.b64decode(encoded_filters, validate=True)
1981
- self._pushdown_filters = _apply_filter_delta_bytes(self._pushdown_filters, filter_bytes)
1982
- self._filter_delta_history.append(filter_bytes)
2087
+ self._pushdown_filters, self._filter_delta_history, self._filter_predicate_order = _record_filter_delta(
2088
+ self._pushdown_filters, self._filter_delta_history, filter_bytes
2089
+ )
1983
2090
 
1984
2091
  def produce(self, out: OutputCollector, ctx: CallContext) -> None:
1985
2092
  """Produce the next output batch from the table function."""
@@ -2052,7 +2159,10 @@ class TableInOutExchangeState(_VgiCallStateHolder, ExchangeState):
2052
2159
  _user_state: Annotated[Any, Transient()] = field(default=None, repr=False)
2053
2160
  _pushdown_filters: Annotated[Any, Transient()] = field(default=None, repr=False) # PushdownFilters | None
2054
2161
  _auto_apply: Annotated[bool, Transient()] = field(default=False, repr=False)
2162
+ # The dynamic-filter deltas this stream must replay on an HTTP turn, compacted by
2163
+ # _record_filter_delta, and the live predicate order they rebuild.
2055
2164
  _filter_delta_history: list[bytes] = field(default_factory=list, repr=False)
2165
+ _filter_predicate_order: list[str] = field(default_factory=list, repr=False)
2056
2166
  _vgi_tracer: Annotated[VgiTracer, Transient()] = field(default_factory=get_noop_tracer, repr=False)
2057
2167
 
2058
2168
  def __post_init__(self) -> None:
@@ -2113,13 +2223,17 @@ class TableInOutExchangeState(_VgiCallStateHolder, ExchangeState):
2113
2223
  )
2114
2224
  self._auto_apply = func_cls._should_auto_apply_filters()
2115
2225
  if self._call.init_call.pushdown_filters is not None:
2116
- self._pushdown_filters = func_cls.pushdown_filters(
2117
- self._call.init_call.pushdown_filters,
2118
- join_keys=self._call.init_call.join_keys,
2119
- output_schema=self._call.init_call.output_schema,
2120
- )
2121
- for filter_bytes in self._filter_delta_history:
2122
- self._pushdown_filters = _apply_filter_delta_bytes(self._pushdown_filters, filter_bytes)
2226
+ # Everything re-parsed here was validated when it first arrived.
2227
+ with replaying_accepted_state():
2228
+ self._pushdown_filters = _replay_filter_history(
2229
+ func_cls.pushdown_filters(
2230
+ self._call.init_call.pushdown_filters,
2231
+ join_keys=self._call.init_call.join_keys,
2232
+ output_schema=self._call.init_call.output_schema,
2233
+ ),
2234
+ self._filter_delta_history,
2235
+ self._filter_predicate_order,
2236
+ )
2123
2237
  self._params = dataclasses.replace(self._params, current_pushdown_filters=self._pushdown_filters)
2124
2238
  # Restore _user_state from serialized bytes if available
2125
2239
  if self._user_state_bytes is not None:
@@ -2141,8 +2255,9 @@ class TableInOutExchangeState(_VgiCallStateHolder, ExchangeState):
2141
2255
  if len(encoded) > (24 << 20):
2142
2256
  raise ValueError("base64 dynamic filter metadata exceeds the encoded-size limit")
2143
2257
  filter_bytes = base64.b64decode(encoded, validate=True)
2144
- self._pushdown_filters = _apply_filter_delta_bytes(self._pushdown_filters, filter_bytes)
2145
- self._filter_delta_history.append(filter_bytes)
2258
+ self._pushdown_filters, self._filter_delta_history, self._filter_predicate_order = _record_filter_delta(
2259
+ self._pushdown_filters, self._filter_delta_history, filter_bytes
2260
+ )
2146
2261
  params = dataclasses.replace(
2147
2262
  self._params,
2148
2263
  auth_context=ctx.auth,
vgi/scalar_function.py CHANGED
@@ -755,7 +755,8 @@ class ScalarFunctionGenerator(vgi.function.Function):
755
755
  """Initialize the function during the init API call.
756
756
 
757
757
  Override to perform one-time setup that should happen after bind
758
- but before processing batches. The default returns max_processes=1.
758
+ but before processing batches. The default leaves ``max_workers``
759
+ unbounded, so the caller decides how many connections to use.
759
760
 
760
761
  Args:
761
762
  bind_call: The original BindCall with arguments and schema.
@@ -769,7 +770,7 @@ class ScalarFunctionGenerator(vgi.function.Function):
769
770
  storage: [`BoundStorage`][] for storing data across calls.
770
771
 
771
772
  Returns:
772
- [`GlobalInitResponse`][] with max_processes and optional opaque data.
773
+ [`GlobalInitResponse`][] with ``max_workers`` and optional opaque data.
773
774
 
774
775
  """
775
776
  return GlobalInitResponse()
@@ -1131,6 +1131,42 @@ class PushdownFilters:
1131
1131
  raise FilterDeserializationError(f"Failed to parse filter delta: {exc}") from exc
1132
1132
  return _pushdown_from_v2_state(state)
1133
1133
 
1134
+ def _revision_map(self) -> dict[str, int]:
1135
+ """Map every predicate ID seen in this scan, tombstones included, to its last applied revision.
1136
+
1137
+ Returns:
1138
+ ``{predicate_id: revision}``; empty for filters not decoded from v2.
1139
+
1140
+ """
1141
+ return dict(self._v2_state.revisions) if self._v2_state is not None else {}
1142
+
1143
+ def _with_predicate_order(self, order: list[str]) -> PushdownFilters:
1144
+ """Return these filters with their v2 predicates arranged in ``order``.
1145
+
1146
+ Only the order changes: ``order`` must name exactly the live predicate IDs.
1147
+ Used to restore the order a state had before its delta history was
1148
+ compacted (an ID removed and later re-added moves to the end, which a
1149
+ shorter history does not reproduce by itself).
1150
+
1151
+ Args:
1152
+ order: The live predicate IDs in the order to restore.
1153
+
1154
+ Returns:
1155
+ Filters with the same predicates, arranged in ``order``.
1156
+
1157
+ Raises:
1158
+ FilterDeserializationError: If ``order`` does not name exactly the
1159
+ live predicates.
1160
+
1161
+ """
1162
+ state = self._v2_state
1163
+ if state is None or [p.id for p in state.predicates] == order:
1164
+ return self
1165
+ by_id = {p.id: p for p in state.predicates}
1166
+ if len(order) != len(by_id) or set(order) != set(by_id):
1167
+ raise FilterDeserializationError("recorded predicate order does not match the replayed filter state")
1168
+ return _pushdown_from_v2_state(dataclasses.replace(state, predicates=tuple(by_id[i] for i in order)))
1169
+
1134
1170
  def get_join_keys_batch(self) -> pa.RecordBatch | None:
1135
1171
  """Return a merged join keys batch for temp table registration.
1136
1172
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: vgi-python
3
- Version: 0.35.0
3
+ Version: 0.36.0
4
4
  Summary: Vector Gateway Interface - Connect DuckDB to external programs via Apache Arrow
5
5
  Project-URL: Homepage, https://query.farm
6
6
  Project-URL: Repository, https://github.com/Query-farm/vgi-python
@@ -5,12 +5,12 @@ vgi/aggregate_function.py,sha256=v03wBSy2Rrl9F2pMYvu8GnGrTCQrAVm01HXIFNdjnm4,268
5
5
  vgi/argument_spec.py,sha256=SADH97VlFu3asI8CWZ_I7XTuSxwqFciFgdupdAf-tkY,34231
6
6
  vgi/arguments.py,sha256=ncnC1CS7KN-2rDjSw9vdf1uHqV5ozebet_eZtcq_ujw,72300
7
7
  vgi/attach_header.py,sha256=sxpzCOVa5fl_-kQSCZgeUxS5I9Kuv8VCUjAOF5s1pr0,3782
8
- vgi/auth.py,sha256=TlD5bhdyXnFuAoCnoHMU9ffjewVEnpwO5auIvp9GsCA,2486
8
+ vgi/auth.py,sha256=m97hDDR8bXIQRBvqhHQZesBrNkAnKgSH6jhuXtGubjI,4688
9
9
  vgi/cache_control.py,sha256=IyBp9gZTCc4Y-4V7Rs05_o5lSEjbR2jt6cddXPU0dKw,7812
10
10
  vgi/copy_from_function.py,sha256=oroQe5qYzZTJ-ppzQTQ-OruEF1ZlwTBNLBoXYfN-P2o,7133
11
11
  vgi/copy_to_function.py,sha256=8l0v_DyEF1EnFwinStrZDQ-yNFfGJsv3Mr5o2jG30f0,9279
12
12
  vgi/exceptions.py,sha256=oX_sZc9xGWi7Xf8cJQf89fX19i3ocDEj_V_76GINgBQ,7294
13
- vgi/filter_v2.py,sha256=3cW6ukHOb0kx13-mN2UEvfib4_Z2r94zVT2E5zQZwiM,53769
13
+ vgi/filter_v2.py,sha256=pSdmZF4HgSr9ZuM_mU3NMwUuAw0X4sZ7KonTMeGnjUw,56243
14
14
  vgi/filter_v2_builder.py,sha256=8OmyzDI6GDdH4pYm2sJFMidzKusi0I8IGNipolPpjX0,5535
15
15
  vgi/function.py,sha256=KIfvKk1nzBvRDPXViHNX89tv6gvIMiDj4e8fl6ANqQg,11626
16
16
  vgi/function_storage.py,sha256=FjVwxc1BIjWo-82j8YE5qcU8w66sGjwyGsTmCUhn-Kw,67576
@@ -22,10 +22,10 @@ vgi/meta_worker.py,sha256=WggIKvDAlahSrBPaVM7ZqVWk3Xc3JNE70EgeoXdy-3w,31725
22
22
  vgi/metadata.py,sha256=M8KZObX1wwCr2lPjgDL0-79gPbeM_NSOIqYCIAAck5M,74719
23
23
  vgi/otel.py,sha256=JlUDnk2c0UXSo_j1PJdS67Y--t55sXPg054ADuvHANc,15452
24
24
  vgi/profiling.py,sha256=rK7vij3oHhXC3JHI2BsMySsXh7_hrXi0aXpCW8iFSIc,3671
25
- vgi/protocol.py,sha256=m9W00W-c9jv2kqISp_wb8E8Rtity_60_12nzE3_d9Xg,165016
25
+ vgi/protocol.py,sha256=iKuo-k6oH9nss_33LfYcekBBNhsM-yeHY0T5fwgg7iw,170333
26
26
  vgi/protocol_version.txt,sha256=wo_MpTY3vIjhJK8XJd8Ty5jGne3v1i-zzb4c22t2BiQ,6
27
27
  vgi/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
28
- vgi/scalar_function.py,sha256=zj6U1e5mXm4hk_ZB_CAjYmeO1LZBJcEMzNDzhwpXPmQ,53415
28
+ vgi/scalar_function.py,sha256=YBy5s4eEBYTMjACGPE-8-82BObyJoM9qt9cnm24b_RQ,53485
29
29
  vgi/schema_path.py,sha256=GVlkZlCUj7GbVhgiocSMC6Ff5iTJhKqiYE62hMIEKiI,2403
30
30
  vgi/schema_utils.py,sha256=zQCliUgIpO2u77Mhtgg3OFvOEDNgJopu6kj2fw8YP1I,7945
31
31
  vgi/secret_protocol.py,sha256=iCDl7nF7Nb4cJV4vpfbldunTDqDJWzCzS7uCmOACf7U,7151
@@ -33,7 +33,7 @@ vgi/secret_service.py,sha256=r7qCzfnOMirf9wSCUe-mgFe3z6v9_cDRzixuSAZG_bs,8973
33
33
  vgi/serve.py,sha256=5UmvRAGcHj5RKCOUQMXG9RFVYAbLZsHwwk1qVI9fipY,63844
34
34
  vgi/split_token.py,sha256=tutcksj-hgWrLY7UW3sSqkyoaD-cMqNTOeff40MGl_g,10850
35
35
  vgi/table_buffering_function.py,sha256=O5TcIdqk8-YViCQCkb8bOevO-zhe3IoyzLgwXDO3Cjc,19531
36
- vgi/table_filter_pushdown.py,sha256=2wJK09OkZZm-WYFsB-um4lbIMZV6gU1fcx8EsHglnA4,61861
36
+ vgi/table_filter_pushdown.py,sha256=agzbDOz_-64E2uNZdDOQs8woP1CBAE_SU5rv2N4fqWk,63435
37
37
  vgi/table_function.py,sha256=8Pti3uFnsRaPCVG-CGYl9r3GHJF146Fp_62oXXwHKKw,73502
38
38
  vgi/table_in_out_function.py,sha256=F2QEtsbGEYHDVyQ4Cf9OCW7HlWXeqekAyx607foB-bs,19393
39
39
  vgi/worker.py,sha256=MGMQmy6PPKCqo5q_9fBS3z3zf4y6-UKdJAJiHGfnk50,249553
@@ -96,7 +96,7 @@ vgi/_test_fixtures/table/batch_index.py,sha256=P5ds0xgikuEQanSEWVWKMLbdvIzUeJraI
96
96
  vgi/_test_fixtures/table/batch_index_broken.py,sha256=kZOGrLL7ZW1rmwPmNEYRmiF_vqIfHsfXioq5vKPWHk0,7314
97
97
  vgi/_test_fixtures/table/cache.py,sha256=7oohKpFJ_si7LoKE7QINkz5oIyj1sx8Kiobpwo4Sn7w,72322
98
98
  vgi/_test_fixtures/table/catalog_scans.py,sha256=5j1Sx02-HWK7bFurDu4e9HiS3Q9BmBukA3sAErH4GHE,5080
99
- vgi/_test_fixtures/table/filters.py,sha256=x-k3fOHdA65-45CZUmyGxpmI0uk9eqaNixrs97Wlv70,41918
99
+ vgi/_test_fixtures/table/filters.py,sha256=pOvuBFr0Rj7PXUWILUdfGf735sJda-iphJ_aQyiw_VY,42197
100
100
  vgi/_test_fixtures/table/late_materialization.py,sha256=BT1FOpQNeP4IXBS3hUvAmO-5jscl5iiQ11fvxUfN6DM,10355
101
101
  vgi/_test_fixtures/table/make_series.py,sha256=K-G_YNq25Kb7I5bp6XK4rCZzwMYNTxgKH3M_O-AQAlA,8812
102
102
  vgi/_test_fixtures/table/misc.py,sha256=71WOIFqk5ntnEIqsG-57rZ9DY7ShQqMKHi7yluNAlM4,16250
@@ -108,9 +108,9 @@ vgi/_test_fixtures/table/profiling_example.py,sha256=Rt3fgKxJhr7Q8QQRss2-OV13wh2
108
108
  vgi/_test_fixtures/table/required_filters.py,sha256=StJeS2tQYyXjivNs_tkcRBq4rVtIGoGQDWcggib1Rxg,7079
109
109
  vgi/_test_fixtures/table/same_name_cached.py,sha256=JMXSXTcqyO7XwWqIizgddlTvLYuUy02Qcq9ZkQnTYU0,4415
110
110
  vgi/_test_fixtures/table/same_name_schemas.py,sha256=yRti9HXO3-Y9ahiLIHoaZ31LvWjtPSbjxmPJyoQ_0mQ,4930
111
- vgi/_test_fixtures/table/sequence.py,sha256=yGn4b3Co06KgkYAUuCj27Fm8GDcHbWFq6zSvNPwHOqw,26073
111
+ vgi/_test_fixtures/table/sequence.py,sha256=0kOckCwRu19RMKgFrUcnnnDR8e0utjX_E2kD3w6NnRU,26352
112
112
  vgi/_test_fixtures/table/settings.py,sha256=C4J5OKY3vgNbB09QAiQlzwhI3Tc5wtewHHLFhO5xvo4,16689
113
- vgi/_test_fixtures/table/splits.py,sha256=XvN-oAmLg3suVXSsCEuvVjDNSeVuvl-nc2gXWY5Q5sE,48010
113
+ vgi/_test_fixtures/table/splits.py,sha256=jN1pt_dL0f9JAJXfMzIIO2-V6qH9vS4GF5q2xejjUl4,48333
114
114
  vgi/_test_fixtures/table/transaction_storage.py,sha256=cGSaM25cE6brwbUJ0-mchocX0UfwOb7uscCATKLYeME,6086
115
115
  vgi/_test_fixtures/table/tt_pushdown.py,sha256=jMELJ2HkG5-yIuYqKRYaPSouJxs3N-tiWRPthoPZ9ko,7420
116
116
  vgi/_test_fixtures/table/typed_probe.py,sha256=JjgLeHuhkXAVnCcezgqR7ayXMp3tTTixegftCwjKOSY,5122
@@ -123,7 +123,7 @@ vgi/_test_fixtures/writable/worker.py,sha256=ti-YFqVdYTMPIvDC8iwvaZ_CYUFTIf9xod2
123
123
  vgi/catalog/__init__.py,sha256=edc2wuvloEgY2WnnJNngV3xP0DN_p49SElYQanGzq4A,2629
124
124
  vgi/catalog/_descriptor_spec.py,sha256=fyeWOC3eN04UXDAOnxFeOS1VwGBWoBCMOzZfRYKPFH4,9773
125
125
  vgi/catalog/attach_option.py,sha256=4cgJZjTVCnFv8TrLYnfjy-XZLAx5ux4UisJRcBeXkfo,5735
126
- vgi/catalog/catalog_interface.py,sha256=ucv4QvMwqU7xdWtSBOLn2bXo0VhyXOQCyN51zL0ILjU,157262
126
+ vgi/catalog/catalog_interface.py,sha256=-JdVTfkTOC2RYickktUiESF5UCuLVnPnLCUBQrTaN_4,158452
127
127
  vgi/catalog/descriptors.py,sha256=JvI0mqDy6DGmtnhZw-U2Z9ZROMjz4KejxGrw7XmNnGw,49362
128
128
  vgi/catalog/duckdb_statistics.py,sha256=Bzw4UGO-zdKrontaCcRV2palO8G3XLCWh-AhZozvQI8,16894
129
129
  vgi/catalog/secret_type.py,sha256=1DoEMyQun9K5z53817rUuhG1YpHSEAsazEmubkRLVL0,3745
@@ -139,7 +139,7 @@ vgi/client/cli_table.py,sha256=H1xrKZF5yYmSe55YSrLWXGIXPmuGuzQcncEBvqL43zs,26498
139
139
  vgi/client/cli_transaction.py,sha256=TDW0ZrBy8XM_eYSkYk5ewEsnozfzDJOlopd66Fm7OOE,3142
140
140
  vgi/client/cli_utils.py,sha256=_JoNiVu12NV4Pg2b1wtWFt0L0uIWJ9p9mBIfIu-QIuM,17387
141
141
  vgi/client/cli_view.py,sha256=0qhxoS91-zFMjq8V_Fb0PNW-YDxERu8-zPpxSFEE-G8,8490
142
- vgi/client/client.py,sha256=ocboi5d_h4KT1RXMEWMXAWYKSC5OH9IsJ4ah6z6C1JI,163751
142
+ vgi/client/client.py,sha256=cvBQHzM2Z2lamC-fkBAvZD4HuRxB03aUkEKFVo_E9GU,167312
143
143
  vgi/client/errors.py,sha256=cSN65cyQMk3crwZCz1IdPErK3mtlEkTNpu-nuQUDOQE,1633
144
144
  vgi/http/__init__.py,sha256=hlOwOVcoZqmEKVGk7ganOdr5ryRSpEhLBF9sRD7BkYc,608
145
145
  vgi/http/demo_storage.py,sha256=830s-H61thozC3eEuqdnwCpYFfvoJRq9vjyuXa5OURo,8769
@@ -152,8 +152,8 @@ vgi/transactor/_duckdb_compat.py,sha256=sXVZ9JLKAQyGR1BjWczSwdQEavtr-TcZPoVZZnTr
152
152
  vgi/transactor/client.py,sha256=7DTeMksogsw6ANjQjGOPpKYrV76rg4_kGjktMJf54jg,4486
153
153
  vgi/transactor/protocol.py,sha256=v59IvrKnuvwOXvn_HEcBAbCzDVcx0akgKN0R1mChXg0,5034
154
154
  vgi/transactor/server.py,sha256=nzsZQxJZdgappdYX8okrFIrjFtUbdM_Hhskl1fZ2nDY,32553
155
- vgi_python-0.35.0.dist-info/METADATA,sha256=lNsV-ESefoEMsTf3l8hEhgiqthtqmbRBHFHRoGDY7VE,25617
156
- vgi_python-0.35.0.dist-info/WHEEL,sha256=THafob7ofN-NsuMN7Mg4qZyHaQI7KkD-QlcQatYhXPo,87
157
- vgi_python-0.35.0.dist-info/entry_points.txt,sha256=3Kz1vgodw3pOL_xjtSyDB55-ZRy-U2X-X_Bdr582x0Q,165
158
- vgi_python-0.35.0.dist-info/licenses/LICENSE,sha256=pbJb4zZasP6n5ifEV81wFu017TarjydaYVmGbHcehtY,6103
159
- vgi_python-0.35.0.dist-info/RECORD,,
155
+ vgi_python-0.36.0.dist-info/METADATA,sha256=AURP4Y2G9g1qjUinaOSYm_owju8eSRHibiK_8Z0B7p8,25617
156
+ vgi_python-0.36.0.dist-info/WHEEL,sha256=THafob7ofN-NsuMN7Mg4qZyHaQI7KkD-QlcQatYhXPo,87
157
+ vgi_python-0.36.0.dist-info/entry_points.txt,sha256=3Kz1vgodw3pOL_xjtSyDB55-ZRy-U2X-X_Bdr582x0Q,165
158
+ vgi_python-0.36.0.dist-info/licenses/LICENSE,sha256=pbJb4zZasP6n5ifEV81wFu017TarjydaYVmGbHcehtY,6103
159
+ vgi_python-0.36.0.dist-info/RECORD,,