vgi-python 0.34.0__py3-none-any.whl → 0.35.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -138,9 +138,9 @@ class SubstreamPartialSumFunction(TableInOutFunction[SingleTableArguments, Subst
138
138
  their finalize outputs, so the caller re-aggregates with an outer
139
139
  ``SELECT sum(...)`` to get the global total — correct no matter how the rows
140
140
  were partitioned across substreams. Each substream's ``finish()`` reads only
141
- its OWN worker's accumulated state (keyed by the substream's execution_id;
142
- ``params.substream_id`` is the stable client-owned key available for workers
143
- that manage cross-backend state themselves). This is the per-substream
141
+ its OWN execution's accumulated state — the framework keeps one row per
142
+ ``params.substream_id`` in storage scoped to that execution, so it is found
143
+ even when a finalize lands on a different backend. This is the per-substream
144
144
  finalize contract A4 enables — it is NOT a global cross-substream combine
145
145
  (that is a ``TableBufferingFunction``; see ``SumAllColumnsSimpleDistributed``).
146
146
 
@@ -191,8 +191,9 @@ class SubstreamPartialSumFunction(TableInOutFunction[SingleTableArguments, Subst
191
191
  params: ProcessParams[SingleTableArguments],
192
192
  states: list[SubstreamPartialSumState],
193
193
  ) -> list[pa.RecordBatch]:
194
- # `states` are THIS substream's accumulated states (one per worker pid that
195
- # handled this substream's batches); their sum is this substream's partial.
194
+ # `states` are this execution's accumulated states, one per substream that
195
+ # saw input (the Python client fans one execution across several); their
196
+ # sum is this finalize's partial.
196
197
  total = sum(st.total for st in states)
197
198
  name = params.output_schema.names[0]
198
199
  return [pa.RecordBatch.from_pydict({name: [total]}, schema=params.output_schema)]
vgi/client/client.py CHANGED
@@ -78,7 +78,9 @@ import io
78
78
  import itertools
79
79
  import logging
80
80
  import os
81
+ import shlex
81
82
  import subprocess
83
+ import sys
82
84
  import threading
83
85
  from collections.abc import Callable, Generator, Iterator, Sequence
84
86
  from contextlib import AbstractContextManager
@@ -182,7 +184,25 @@ _default_pool = WorkerPool(max_idle=8, idle_timeout=30.0)
182
184
  # whether to skip the HTTP leg of the matrix.
183
185
  _HTTP_TRANSPORT_READY = True
184
186
 
187
+ # ``server_path`` prefix selecting ``transport="launch"`` — the same scheme as
188
+ # the VGI DuckDB extension's ``launch:<argv>`` LOCATION.
189
+ _LAUNCH_SCHEME = "launch:"
190
+
185
191
  _DEFAULT_ACCEPTED_MAX_RESPONSE_BYTES = 256 * 1024 * 1024
192
+
193
+
194
+ def _substream_id_for(phase: TableInOutFunctionInitPhase | None) -> bytes | None:
195
+ """Mint the ``InitRequest.substream_id`` for a connection's table-in-out INPUT stream.
196
+
197
+ Each connection a table-in-out call fans out to is its own substream of the
198
+ one execution — the worker keeps one accumulated state per substream, and
199
+ the single finalize drains them all. Sharing an id (or sending none, which
200
+ the worker keys by process) lets connections served by one process
201
+ overwrite each other's state. ``None`` for every other kind of init.
202
+ """
203
+ return os.urandom(16) if phase == TableInOutFunctionInitPhase.INPUT else None
204
+
205
+
186
206
  _MIN_ACCEPTED_MAX_RESPONSE_BYTES = 65536
187
207
  _MAX_SAFE_HTTP_BYTES = (1 << 53) - 1
188
208
 
@@ -215,6 +235,8 @@ class WorkerConnection:
215
235
  stream: The active streaming session, if any.
216
236
  proc: The worker subprocess for direct (non-pooled) subprocess transport.
217
237
  connection: The RPC connection for direct subprocess transport.
238
+ substream_id: The ``InitRequest.substream_id`` this connection's
239
+ table-in-out stream was opened with, or None.
218
240
  """
219
241
 
220
242
  proxy: VgiProtocol
@@ -223,6 +245,7 @@ class WorkerConnection:
223
245
  # Subprocess transport, direct (non-pooled).
224
246
  proc: subprocess.Popen[bytes] | None = None
225
247
  connection: RpcConnection[VgiProtocol] | None = None
248
+ substream_id: bytes | None = None
226
249
  # Subprocess transport, pooled.
227
250
  _pool_ctx: AbstractContextManager[Any] | None = field(default=None, repr=False)
228
251
  # HTTP transport: context manager from vgi_rpc.http.http_connect.
@@ -486,6 +509,11 @@ class Client(CatalogClientMixin, AggregateClientMixin):
486
509
  spaces or quotes (``[sys.executable, "-c", script]``). No shell
487
510
  is involved either way, so shell syntax — pipes, redirection,
488
511
  ``VAR=value`` prefixes, ``~`` expansion — is not interpreted.
512
+ A string of the form ``launch:<argv>`` selects
513
+ ``transport="launch"`` instead, with ``<argv>`` split the same
514
+ way into ``launch_argv`` — the VGI DuckDB extension's
515
+ ``launch:`` LOCATION scheme, so one worker string names the
516
+ same shared worker in both. ``pool`` does not apply to it.
489
517
  passthrough_stderr: Subprocess-only. If True, worker stderr is
490
518
  passed through to the parent process's stderr in real-time.
491
519
  worker_limit: Maximum number of parallel worker processes.
@@ -584,6 +612,14 @@ class Client(CatalogClientMixin, AggregateClientMixin):
584
612
  combination is inconsistent.
585
613
 
586
614
  """
615
+ if transport == "subprocess" and isinstance(server_path, str) and server_path.startswith(_LAUNCH_SCHEME):
616
+ if launch_argv is not None:
617
+ raise ValueError("pass the worker command as a 'launch:' server_path or as launch_argv, not both")
618
+ launch_argv = shlex.split(server_path[len(_LAUNCH_SCHEME) :], posix=sys.platform != "win32")
619
+ server_path = None
620
+ pool = None
621
+ transport = "launch"
622
+
587
623
  if transport == "subprocess":
588
624
  if server_path is None:
589
625
  raise ValueError("subprocess transport requires server_path")
@@ -1618,6 +1654,7 @@ class Client(CatalogClientMixin, AggregateClientMixin):
1618
1654
  finalize_state_id: bytes | None = None,
1619
1655
  split_tokens: list[bytes] | None = None,
1620
1656
  join_keys: list[pa.RecordBatch] | None = None,
1657
+ substream_id: bytes | None = None,
1621
1658
  ) -> StreamSession:
1622
1659
  """Call init on a worker proxy and return a `StreamSession`.
1623
1660
 
@@ -1642,6 +1679,8 @@ class Client(CatalogClientMixin, AggregateClientMixin):
1642
1679
  looked up by column name — see
1643
1680
  ``PushdownFilters.get_join_keys_column``), or `None` when not
1644
1681
  applicable.
1682
+ substream_id: The table-in-out substream this init opens or
1683
+ finalizes (see ``_substream_id_for``), or `None`.
1645
1684
 
1646
1685
  Returns:
1647
1686
  `StreamSession` for data exchange or production.
@@ -1662,6 +1701,7 @@ class Client(CatalogClientMixin, AggregateClientMixin):
1662
1701
  init_opaque_data=init_opaque_data,
1663
1702
  finalize_state_id=finalize_state_id,
1664
1703
  split_tokens=split_tokens,
1704
+ substream_id=substream_id,
1665
1705
  )
1666
1706
  try:
1667
1707
  stream: StreamSession = proxy.init(request=init_request) # type: ignore[assignment]
@@ -1750,6 +1790,7 @@ class Client(CatalogClientMixin, AggregateClientMixin):
1750
1790
  )
1751
1791
  bind_response = self._do_bind(self._primary.proxy, bind_request, bind_result_callback)
1752
1792
 
1793
+ substream_id = _substream_id_for(phase)
1753
1794
  stream = self._do_init(
1754
1795
  self._primary.proxy,
1755
1796
  bind_request,
@@ -1761,8 +1802,10 @@ class Client(CatalogClientMixin, AggregateClientMixin):
1761
1802
  execution_id=split_execution_id,
1762
1803
  init_opaque_data=split_init_opaque_data,
1763
1804
  join_keys=join_keys,
1805
+ substream_id=substream_id,
1764
1806
  )
1765
1807
  self._primary.stream = stream
1808
+ self._primary.substream_id = substream_id
1766
1809
 
1767
1810
  init_response = stream.typed_header(GlobalInitResponse)
1768
1811
  max_workers = 1 if split_tokens is not None else self._determine_max_workers(init_response.max_workers)
@@ -1838,6 +1881,7 @@ class Client(CatalogClientMixin, AggregateClientMixin):
1838
1881
 
1839
1882
  def do_init(worker: WorkerConnection) -> None:
1840
1883
  try:
1884
+ worker.substream_id = _substream_id_for(phase)
1841
1885
  stream = self._do_init(
1842
1886
  worker.proxy,
1843
1887
  bind_request,
@@ -1848,6 +1892,7 @@ class Client(CatalogClientMixin, AggregateClientMixin):
1848
1892
  join_keys=join_keys,
1849
1893
  execution_id=global_init_response.execution_id,
1850
1894
  init_opaque_data=global_init_response.opaque_data,
1895
+ substream_id=worker.substream_id,
1851
1896
  )
1852
1897
  worker.stream = stream
1853
1898
  except Exception as e:
@@ -2629,6 +2674,7 @@ class Client(CatalogClientMixin, AggregateClientMixin):
2629
2674
  phase=TableInOutFunctionInitPhase.FINALIZE,
2630
2675
  execution_id=init_response.execution_id,
2631
2676
  init_opaque_data=init_response.opaque_data,
2677
+ substream_id=self._primary.substream_id,
2632
2678
  )
2633
2679
 
2634
2680
  try:
vgi/function.py CHANGED
@@ -21,6 +21,7 @@ from __future__ import annotations
21
21
 
22
22
  import logging
23
23
  import os
24
+ import threading
24
25
  from abc import ABC
25
26
  from typing import (
26
27
  Annotated,
@@ -73,19 +74,32 @@ def _resolve_storage() -> FunctionStorage:
73
74
 
74
75
 
75
76
  class _DefaultStorageDescriptor:
76
- """Resolve `FunctionStorage` lazily on first attribute access.
77
+ """Resolve `FunctionStorage` lazily on first attribute access — exactly once.
77
78
 
78
79
  This avoids evaluating environment variables at import time. When a
79
80
  subclass explicitly sets ``storage = SomeStorage(...)``, the plain
80
81
  attribute shadows this descriptor — no interference.
82
+
83
+ Resolution is locked because first use is a *request* now, and requests
84
+ arrive concurrently. Until 0.34.1 the store was resolved as a side effect of
85
+ ``import vgi``, single-threaded; once that was made lazy, an unguarded
86
+ check-then-set let two first callers each build a store — and with the
87
+ ``:memory:`` backend that is two unrelated databases, so state one thread
88
+ wrote was invisible to the other. Double-checked, so every later access
89
+ stays lock-free.
81
90
  """
82
91
 
83
92
  _resolved: FunctionStorage | None = None
93
+ _lock = threading.Lock()
84
94
 
85
95
  def __get__(self, obj: object | None, objtype: type | None = None) -> FunctionStorage:
86
- if self._resolved is None:
87
- self._resolved = _resolve_storage()
88
- return self._resolved
96
+ resolved = self._resolved
97
+ if resolved is None:
98
+ with self._lock:
99
+ if self._resolved is None:
100
+ self._resolved = _resolve_storage()
101
+ resolved = self._resolved
102
+ return resolved
89
103
 
90
104
 
91
105
  # Default max_workers when not explicitly specified (effectively unlimited)
vgi/function_storage.py CHANGED
@@ -20,6 +20,7 @@ Implementations:
20
20
  import contextlib
21
21
  import enum
22
22
  import functools
23
+ import inspect
23
24
  import logging
24
25
  import os
25
26
  import sqlite3
@@ -944,9 +945,9 @@ class FunctionStorageSqlite:
944
945
  self._memory_uri = None
945
946
  self._anchor_conn = None
946
947
  self.db_path = db_path if db_path is not None else _get_default_db_path()
947
- # Shared-cache in-memory DBs need a process-local write lock; file DBs
948
- # do not. See `_write_guard`.
949
- self._write_lock: threading.Lock | None = threading.Lock() if self._memory_uri is not None else None
948
+ # Shared-cache in-memory DBs need a process-local lock around every
949
+ # operation; file DBs do not. See `_op_guard`.
950
+ self._op_lock: threading.Lock | None = threading.Lock() if self._memory_uri is not None else None
950
951
  self._tls = threading.local()
951
952
  self._ensure_tables()
952
953
 
@@ -986,8 +987,8 @@ class FunctionStorageSqlite:
986
987
  return conn
987
988
 
988
989
  @contextlib.contextmanager
989
- def _write_guard(self) -> Iterator[None]:
990
- """Serialize writers on a shared-cache in-memory DB; no-op for a file DB.
990
+ def _op_guard(self) -> Iterator[None]:
991
+ """Serialize every operation on a shared-cache in-memory DB; no-op for a file DB.
991
992
 
992
993
  `_conn` gives each thread its own connection so SQLite's own locking
993
994
  serializes writers "without a Python-level lock and without forfeiting
@@ -1008,11 +1009,25 @@ class FunctionStorageSqlite:
1008
1009
  Serializing writes costs nothing here. Measured: pops are 0.007 ms and
1009
1010
  concurrency already made them *slower* (128k/s at 1 thread → 79k/s at
1010
1011
  16), so there was never any write parallelism to lose — only the error.
1012
+
1013
+ **Readers too.** This first serialized writers only, which is not
1014
+ enough: shared-cache locks are per *table* and conflict in both
1015
+ directions. A writer cannot modify a table another connection is
1016
+ reading, and a reader cannot read one another connection has written
1017
+ and not yet committed — either is the same unretried `SQLITE_LOCKED`.
1018
+ So a `state_scan` overlapping an `execution_clear` failed exactly as
1019
+ two pops had. It stayed hidden because `vgi-fixture-http`, the one
1020
+ in-tree user of `:memory:`, silently ran on the file DB until 0.34.1:
1021
+ `import vgi` resolved `Function.storage` before the fixture's `main()`
1022
+ could ask for memory. Once it really ran here, 69 cases of the
1023
+ integration suite failed with "database table is locked". Every read
1024
+ fetches eagerly (`fetchall`/`fetchone`), so no statement outlives the
1025
+ guard, and — as above — there was no concurrency to give up.
1011
1026
  """
1012
- if self._write_lock is None:
1027
+ if self._op_lock is None:
1013
1028
  yield
1014
1029
  return
1015
- with self._write_lock:
1030
+ with self._op_lock:
1016
1031
  yield
1017
1032
 
1018
1033
  def close(self) -> None:
@@ -1477,40 +1492,37 @@ class FunctionStorageSqlite:
1477
1492
  conn.commit()
1478
1493
 
1479
1494
 
1480
- def _guard_sqlite_writes() -> None:
1481
- """Route every FunctionStorageSqlite writer through `_write_guard`.
1495
+ #: Public methods that run no statement against the shared cache, so they are
1496
+ #: left outside `_op_guard`. ``close`` only drops the calling thread's own
1497
+ #: connection.
1498
+ _UNGUARDED_OPS = frozenset({"close"})
1499
+
1482
1500
 
1483
- Applied here rather than by editing each method so no writer can be added
1484
- later and silently miss the guard — a partial fix would be worse than none,
1485
- since one unguarded writer is enough to re-introduce `SQLITE_LOCKED` for
1486
- every other one sharing the cache.
1501
+ def _guard_sqlite_ops() -> None:
1502
+ """Route every public FunctionStorageSqlite operation through `_op_guard`.
1503
+
1504
+ Derived from the class rather than listed: the previous version named its
1505
+ methods by hand, the four readers were never on the list, and one unguarded
1506
+ operation is enough to re-introduce `SQLITE_LOCKED` for every other one
1507
+ sharing the cache. A new public method is now guarded by default, and
1508
+ `_UNGUARDED_OPS` is the only way out of it.
1487
1509
  """
1488
1510
 
1489
1511
  def guarded(fn: Any) -> Any:
1490
1512
  @functools.wraps(fn)
1491
1513
  def wrapper(self: "FunctionStorageSqlite", *a: Any, **k: Any) -> Any:
1492
- with self._write_guard():
1514
+ with self._op_guard():
1493
1515
  return fn(self, *a, **k)
1494
1516
 
1495
1517
  return wrapper
1496
1518
 
1497
- for name in (
1498
- "queue_push",
1499
- "queue_pop",
1500
- "queue_clear",
1501
- "state_put_many",
1502
- "state_drain",
1503
- "state_delete",
1504
- "state_append",
1505
- "execution_clear",
1506
- "state_counter_add",
1507
- "state_counter_set",
1508
- "state_counter_delete",
1509
- ):
1510
- setattr(FunctionStorageSqlite, name, guarded(getattr(FunctionStorageSqlite, name)))
1519
+ for name, member in list(vars(FunctionStorageSqlite).items()):
1520
+ if name.startswith("_") or name in _UNGUARDED_OPS or not inspect.isfunction(member):
1521
+ continue
1522
+ setattr(FunctionStorageSqlite, name, guarded(member))
1511
1523
 
1512
1524
 
1513
- _guard_sqlite_writes()
1525
+ _guard_sqlite_ops()
1514
1526
 
1515
1527
 
1516
1528
  class ShardedSqliteStorage:
vgi/serve.py CHANGED
@@ -217,7 +217,6 @@ def create_app(
217
217
  max_stream_response_bytes: int | None = None,
218
218
  max_externalized_response_bytes: int | None = None,
219
219
  introspect_principals: Iterable[str] | None = None,
220
- introspect_rate_limit: int | None = None,
221
220
  iroh_bridge_issuer: str | None = None,
222
221
  iroh_trusted_proxy_addresses: Iterable[str] = (),
223
222
  iroh_authenticate: bool = True,
@@ -279,9 +278,6 @@ def create_app(
279
278
  ``vgi_rpc.Identity.v1``'s ``introspect_token``. Only consulted
280
279
  when the worker class overrides ``resolve_token``. ``None`` reads
281
280
  ``VGI_INTROSPECT_PRINCIPALS``.
282
- introspect_rate_limit: Introspection requests allowed per caller per
283
- second. ``None`` reads ``VGI_INTROSPECT_RATE_LIMIT``, defaulting
284
- to 20.
285
281
  iroh_bridge_issuer: Operator-controlled identity namespace for an
286
282
  ``iroh-http/2`` bridge. ``None`` disables Iroh header trust.
287
283
  iroh_trusted_proxy_addresses: Exact immediate bridge addresses.
@@ -327,7 +323,7 @@ def create_app(
327
323
  worker,
328
324
  enable_describe=describe,
329
325
  server_version=_get_vgi_version(),
330
- identity=_build_identity(worker_cls, introspect_principals, introspect_rate_limit),
326
+ identity=_build_identity(worker_cls, introspect_principals),
331
327
  )
332
328
 
333
329
  effective_peer_identity_providers = tuple(peer_identity_providers)
@@ -499,15 +495,6 @@ def main() -> None:
499
495
  "that case, with no permissive default. Env: VGI_INTROSPECT_PRINCIPALS."
500
496
  ),
501
497
  ),
502
- introspect_rate_limit: int | None = typer.Option( # noqa: B008
503
- None,
504
- "--introspect-rate-limit",
505
- help=(
506
- "Introspection requests allowed per caller per second (default 20). "
507
- "Bounds, rather than closes, the oracle an allowlisted-but-compromised "
508
- "caller has. Env: VGI_INTROSPECT_RATE_LIMIT."
509
- ),
510
- ),
511
498
  access_log_sample: float | None = typer.Option( # noqa: B008
512
499
  None,
513
500
  "--access-log-sample",
@@ -591,7 +578,6 @@ def main() -> None:
591
578
  max_stream_response_bytes=max_stream_response_bytes,
592
579
  max_externalized_response_bytes=max_externalized_response_bytes,
593
580
  introspect_principals=introspect_principals,
594
- introspect_rate_limit=introspect_rate_limit,
595
581
  server=server,
596
582
  worker_ref=worker_ref,
597
583
  http_workers=http_workers,
@@ -802,7 +788,6 @@ def _resolve_authenticate() -> Callable[..., Any] | None:
802
788
  def _build_identity(
803
789
  worker_cls: type[Worker],
804
790
  introspect_principals: Iterable[str] | None,
805
- introspect_rate_limit: int | None,
806
791
  ) -> IdentityImpl | None:
807
792
  """Build the ``vgi_rpc.Identity.v1`` implementation, or ``None``.
808
793
 
@@ -823,8 +808,6 @@ def _build_identity(
823
808
  override.
824
809
  introspect_principals: Principals permitted to introspect, or ``None``
825
810
  to read the environment.
826
- introspect_rate_limit: Per-caller, per-second ceiling, or ``None`` to
827
- read the environment.
828
811
 
829
812
  Returns:
830
813
  The implementation, or ``None`` when this worker does not resolve
@@ -842,7 +825,6 @@ def _build_identity(
842
825
  return IdentityImpl(
843
826
  resolve_token=resolver,
844
827
  introspect_principals=_resolve_introspect_principals(introspect_principals),
845
- introspect_rate_limit=_resolve_introspect_rate_limit(introspect_rate_limit),
846
828
  )
847
829
 
848
830
 
@@ -891,42 +873,6 @@ def _resolve_introspect_principals(explicit: Iterable[str] | None) -> list[str]:
891
873
  return principals
892
874
 
893
875
 
894
- def _resolve_introspect_rate_limit(explicit: int | None) -> int:
895
- """Resolve introspection requests allowed per caller per second.
896
-
897
- Env var: ``VGI_INTROSPECT_RATE_LIMIT``; defaults to 20.
898
-
899
- Args:
900
- explicit: Value passed to :func:`create_app`, or None to read the
901
- environment.
902
-
903
- Returns:
904
- A positive per-caller, per-second request ceiling.
905
-
906
- Raises:
907
- SystemExit: When the environment value is not a positive integer. A
908
- typo that silently became 0 would refuse every introspection with
909
- no diagnostic; one that became huge would remove the bound on an
910
- allowlisted-but-compromised caller's guessing rate.
911
-
912
- """
913
- if explicit is not None:
914
- value = explicit
915
- else:
916
- raw = os.environ.get("VGI_INTROSPECT_RATE_LIMIT")
917
- if not raw:
918
- return 20
919
- try:
920
- value = int(raw)
921
- except ValueError:
922
- sys.stderr.write(f"Error: VGI_INTROSPECT_RATE_LIMIT must be an integer, got {raw!r}\n")
923
- sys.exit(1)
924
- if value <= 0:
925
- sys.stderr.write(f"Error: introspection rate limit must be positive, got {value}\n")
926
- sys.exit(1)
927
- return value
928
-
929
-
930
876
  def _resolve_proxy_proof_gate() -> Any | None:
931
877
  """Build a proxy-proof gate from ``VGI_PROXY_PROOF_*`` environment variables.
932
878
 
@@ -1506,7 +1452,6 @@ def _serve_http(
1506
1452
  max_stream_response_bytes: int | None = None,
1507
1453
  max_externalized_response_bytes: int | None = None,
1508
1454
  introspect_principals: str | None = None,
1509
- introspect_rate_limit: int | None = None,
1510
1455
  server: str = "waitress",
1511
1456
  worker_ref: str | None = None,
1512
1457
  http_workers: int | None = None,
@@ -1524,8 +1469,6 @@ def _serve_http(
1524
1469
  # export_serve_config for why the environment is the channel.
1525
1470
  if introspect_principals is not None:
1526
1471
  os.environ["VGI_INTROSPECT_PRINCIPALS"] = introspect_principals
1527
- if introspect_rate_limit is not None:
1528
- os.environ["VGI_INTROSPECT_RATE_LIMIT"] = str(introspect_rate_limit)
1529
1472
 
1530
1473
  if server == "granian":
1531
1474
  _serve_http_granian(
vgi/table_function.py CHANGED
@@ -8,6 +8,7 @@ to `process()` either emits a batch via `out.emit()` or signals completion via `
8
8
 
9
9
  from __future__ import annotations
10
10
 
11
+ import inspect
11
12
  import uuid
12
13
  from abc import abstractmethod
13
14
  from collections.abc import Mapping
@@ -854,8 +855,19 @@ class TableFunctionBase[TArgs](vgi.function.Function):
854
855
  cls.FunctionArguments = type_args[0]
855
856
  break
856
857
 
857
- # Skip validation for abstract base classes
858
- is_abstract = any(getattr(getattr(cls, name, None), "__isabstractmethod__", False) for name in dir(cls))
858
+ # Skip validation for abstract base classes. Each attribute is read with
859
+ # getattr_static, which returns the raw class attribute without invoking
860
+ # descriptors: a plain getattr() here resolved the inherited
861
+ # `Function.storage` descriptor, whose first access builds the default
862
+ # SQLite store on disk. Because vgi defines subclasses of its own, that
863
+ # made `import vgi` alone create a database under the user's state
864
+ # directory, and fail outright where that directory is not writable (a
865
+ # container's non-root user, say). The raw attribute answers the same
866
+ # question: classmethod, staticmethod and property objects all expose
867
+ # the `__isabstractmethod__` of what they wrap.
868
+ is_abstract = any(
869
+ getattr(inspect.getattr_static(cls, name, None), "__isabstractmethod__", False) for name in dir(cls)
870
+ )
859
871
  if is_abstract:
860
872
  cls._setting_params = {}
861
873
  cls._secret_params = {}
@@ -360,7 +360,8 @@ class TableInOutFunction[
360
360
 
361
361
  Args:
362
362
  params: The process parameters — function args, settings, secrets.
363
- states: The accumulated per-partition states from ``transform()``.
363
+ states: The accumulated states from ``transform()``, one per
364
+ substream of this execution that saw input.
364
365
 
365
366
  Returns:
366
367
  List of pa.RecordBatch to emit as final output.
@@ -412,11 +413,17 @@ class TableInOutFunction[
412
413
  """
413
414
  result = cls.transform(batch, params, state)
414
415
 
415
- # Save state for distributed processing (upsert semantics)
416
+ # Save state for distributed processing (upsert semantics), one row per
417
+ # substream. Storage is scoped to the execution, which every connection
418
+ # of a fanned-out scan shares, and one process serves many of those
419
+ # connections under the launcher, TCP or HTTP — so a per-process key let
420
+ # them overwrite each other and finish() undercounted. The pid remains
421
+ # only for a client that sends no substream_id, which it keys as before.
416
422
  if state is not None:
417
- params.storage.state_put(
418
- FrameworkNS.TIO_STATE, BoundStorage.pack_int_key(os.getpid()), state.serialize_to_bytes()
419
- )
423
+ key = params.substream_id
424
+ if key is None:
425
+ key = BoundStorage.pack_int_key(os.getpid())
426
+ params.storage.state_put(FrameworkNS.TIO_STATE, key, state.serialize_to_bytes())
420
427
 
421
428
  # Handle single batch or list of batches — exchange must emit exactly one
422
429
  if isinstance(result, list):
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: vgi-python
3
- Version: 0.34.0
3
+ Version: 0.35.0
4
4
  Summary: Vector Gateway Interface - Connect DuckDB to external programs via Apache Arrow
5
5
  Project-URL: Homepage, https://query.farm
6
6
  Project-URL: Repository, https://github.com/Query-farm/vgi-python
@@ -162,7 +162,7 @@ Requires-Dist: httpx2>=2.9.1
162
162
  Requires-Dist: platformdirs
163
163
  Requires-Dist: pyarrow
164
164
  Requires-Dist: typer>=0.9
165
- Requires-Dist: vgi-rpc>=0.46.0
165
+ Requires-Dist: vgi-rpc>=0.47.0
166
166
  Provides-Extra: azure
167
167
  Requires-Dist: azure-identity>=1.16.0; extra == 'azure'
168
168
  Requires-Dist: pymssql>=2.3.0; extra == 'azure'
@@ -12,8 +12,8 @@ vgi/copy_to_function.py,sha256=8l0v_DyEF1EnFwinStrZDQ-yNFfGJsv3Mr5o2jG30f0,9279
12
12
  vgi/exceptions.py,sha256=oX_sZc9xGWi7Xf8cJQf89fX19i3ocDEj_V_76GINgBQ,7294
13
13
  vgi/filter_v2.py,sha256=3cW6ukHOb0kx13-mN2UEvfib4_Z2r94zVT2E5zQZwiM,53769
14
14
  vgi/filter_v2_builder.py,sha256=8OmyzDI6GDdH4pYm2sJFMidzKusi0I8IGNipolPpjX0,5535
15
- vgi/function.py,sha256=J8XlHjYsJ-Wg5ePeTosTpnGfGa-YTyYb6Yf00XP1zPM,10934
16
- vgi/function_storage.py,sha256=ZghNlKndrLjMu200_P_qzUTSk-uOQDp0ikenp5C53m8,66434
15
+ vgi/function.py,sha256=KIfvKk1nzBvRDPXViHNX89tv6gvIMiDj4e8fl6ANqQg,11626
16
+ vgi/function_storage.py,sha256=FjVwxc1BIjWo-82j8YE5qcU8w66sGjwyGsTmCUhn-Kw,67576
17
17
  vgi/function_storage_azure_sql.py,sha256=lALZK6FBU-dW4g-PEMtw-zysEfK1o_NNvTX3r6YZu-k,36158
18
18
  vgi/function_storage_cf_do.py,sha256=s5LRBLZPC6XOhLkKiXiF_OhW8GiN5zsZesb6rR_g6wY,29425
19
19
  vgi/invocation.py,sha256=GQ82v0TlryS0yNfhDjzBi5vNpRo1ApIYO0GwkUSxx_U,5713
@@ -30,12 +30,12 @@ vgi/schema_path.py,sha256=GVlkZlCUj7GbVhgiocSMC6Ff5iTJhKqiYE62hMIEKiI,2403
30
30
  vgi/schema_utils.py,sha256=zQCliUgIpO2u77Mhtgg3OFvOEDNgJopu6kj2fw8YP1I,7945
31
31
  vgi/secret_protocol.py,sha256=iCDl7nF7Nb4cJV4vpfbldunTDqDJWzCzS7uCmOACf7U,7151
32
32
  vgi/secret_service.py,sha256=r7qCzfnOMirf9wSCUe-mgFe3z6v9_cDRzixuSAZG_bs,8973
33
- vgi/serve.py,sha256=dyLHzMFdu1AUBtte2mwwYjtQeSF0fOLha1v9MBZbuD0,66163
33
+ vgi/serve.py,sha256=5UmvRAGcHj5RKCOUQMXG9RFVYAbLZsHwwk1qVI9fipY,63844
34
34
  vgi/split_token.py,sha256=tutcksj-hgWrLY7UW3sSqkyoaD-cMqNTOeff40MGl_g,10850
35
35
  vgi/table_buffering_function.py,sha256=O5TcIdqk8-YViCQCkb8bOevO-zhe3IoyzLgwXDO3Cjc,19531
36
36
  vgi/table_filter_pushdown.py,sha256=2wJK09OkZZm-WYFsB-um4lbIMZV6gU1fcx8EsHglnA4,61861
37
- vgi/table_function.py,sha256=lAxWnjBcZBpLoR87Pe-QTK7c8WK_JQh1r_OJ9amI4kk,72746
38
- vgi/table_in_out_function.py,sha256=UGEeXsGkcx0BLt9RyKP6jQEbK6eFTUG6QIGEhxo_VxA,18865
37
+ vgi/table_function.py,sha256=8Pti3uFnsRaPCVG-CGYl9r3GHJF146Fp_62oXXwHKKw,73502
38
+ vgi/table_in_out_function.py,sha256=F2QEtsbGEYHDVyQ4Cf9OCW7HlWXeqekAyx607foB-bs,19393
39
39
  vgi/worker.py,sha256=MGMQmy6PPKCqo5q_9fBS3z3zf4y6-UKdJAJiHGfnk50,249553
40
40
  vgi/write_results.py,sha256=-OrUcz6E9ahhC_pnKYAwIMR2LpapwQafLC4IOnPTmec,2851
41
41
  vgi/_test_fixtures/__init__.py,sha256=xQq-QQLk8USQ6TZHsPiPcZldNNUdcJ2gToXMPGzScLI,444
@@ -52,7 +52,7 @@ vgi/_test_fixtures/http_server.py,sha256=AviomIEClCEuA9WPgQOzrMtQ484N04rQupK8dWO
52
52
  vgi/_test_fixtures/nest_tensor.py,sha256=UjjW9UwipYG3ehQpZOm4hdzsafMyvVc0ZKJCrqiZTPU,24471
53
53
  vgi/_test_fixtures/orchard_catalog.py,sha256=8Uv-mKTIgl-OYU9JLgG23owlHGmgniyF2chPoKXGtuI,1643
54
54
  vgi/_test_fixtures/simple_writable.py,sha256=OmyENVnsgJmYH0_5sNWMt_Xgj9JoPaOu7uI5_ROZDKg,32020
55
- vgi/_test_fixtures/table_in_out.py,sha256=sWTSXwH-BUaSNOQsDmDqlFZxZ5GGtD-_xmysD2g7aV8,93148
55
+ vgi/_test_fixtures/table_in_out.py,sha256=AlBBsbwsMfZkCjGxoCub--OJqmQAim5PgWxAdpbO9Ik,93190
56
56
  vgi/_test_fixtures/table_in_out_same_name.py,sha256=jdsQNdKn64T54W_TcLB-rhydI69Y0Uzr5n6FU7v9rIU,8649
57
57
  vgi/_test_fixtures/twin_catalogs.py,sha256=UFsFuZ_nFpe62HatGnOMru36vk35Xt3Ta1Hkr1aWN9w,4450
58
58
  vgi/_test_fixtures/versioned.py,sha256=gnxNejc4cNGPKl_DxJk1SU8Ielj-rwmQYGkiShXzxK4,5786
@@ -139,7 +139,7 @@ vgi/client/cli_table.py,sha256=H1xrKZF5yYmSe55YSrLWXGIXPmuGuzQcncEBvqL43zs,26498
139
139
  vgi/client/cli_transaction.py,sha256=TDW0ZrBy8XM_eYSkYk5ewEsnozfzDJOlopd66Fm7OOE,3142
140
140
  vgi/client/cli_utils.py,sha256=_JoNiVu12NV4Pg2b1wtWFt0L0uIWJ9p9mBIfIu-QIuM,17387
141
141
  vgi/client/cli_view.py,sha256=0qhxoS91-zFMjq8V_Fb0PNW-YDxERu8-zPpxSFEE-G8,8490
142
- vgi/client/client.py,sha256=iGXmOUN1xl6gdnFZV6pfmgf70S7AgMjC9lnvzMjLZp0,161386
142
+ vgi/client/client.py,sha256=ocboi5d_h4KT1RXMEWMXAWYKSC5OH9IsJ4ah6z6C1JI,163751
143
143
  vgi/client/errors.py,sha256=cSN65cyQMk3crwZCz1IdPErK3mtlEkTNpu-nuQUDOQE,1633
144
144
  vgi/http/__init__.py,sha256=hlOwOVcoZqmEKVGk7ganOdr5ryRSpEhLBF9sRD7BkYc,608
145
145
  vgi/http/demo_storage.py,sha256=830s-H61thozC3eEuqdnwCpYFfvoJRq9vjyuXa5OURo,8769
@@ -152,8 +152,8 @@ vgi/transactor/_duckdb_compat.py,sha256=sXVZ9JLKAQyGR1BjWczSwdQEavtr-TcZPoVZZnTr
152
152
  vgi/transactor/client.py,sha256=7DTeMksogsw6ANjQjGOPpKYrV76rg4_kGjktMJf54jg,4486
153
153
  vgi/transactor/protocol.py,sha256=v59IvrKnuvwOXvn_HEcBAbCzDVcx0akgKN0R1mChXg0,5034
154
154
  vgi/transactor/server.py,sha256=nzsZQxJZdgappdYX8okrFIrjFtUbdM_Hhskl1fZ2nDY,32553
155
- vgi_python-0.34.0.dist-info/METADATA,sha256=4-WKG1yNlJmW1IRUWKY1kssQ_QLYhOrSlZqQs_7dz28,25617
156
- vgi_python-0.34.0.dist-info/WHEEL,sha256=THafob7ofN-NsuMN7Mg4qZyHaQI7KkD-QlcQatYhXPo,87
157
- vgi_python-0.34.0.dist-info/entry_points.txt,sha256=3Kz1vgodw3pOL_xjtSyDB55-ZRy-U2X-X_Bdr582x0Q,165
158
- vgi_python-0.34.0.dist-info/licenses/LICENSE,sha256=pbJb4zZasP6n5ifEV81wFu017TarjydaYVmGbHcehtY,6103
159
- vgi_python-0.34.0.dist-info/RECORD,,
155
+ vgi_python-0.35.0.dist-info/METADATA,sha256=lNsV-ESefoEMsTf3l8hEhgiqthtqmbRBHFHRoGDY7VE,25617
156
+ vgi_python-0.35.0.dist-info/WHEEL,sha256=THafob7ofN-NsuMN7Mg4qZyHaQI7KkD-QlcQatYhXPo,87
157
+ vgi_python-0.35.0.dist-info/entry_points.txt,sha256=3Kz1vgodw3pOL_xjtSyDB55-ZRy-U2X-X_Bdr582x0Q,165
158
+ vgi_python-0.35.0.dist-info/licenses/LICENSE,sha256=pbJb4zZasP6n5ifEV81wFu017TarjydaYVmGbHcehtY,6103
159
+ vgi_python-0.35.0.dist-info/RECORD,,