continuo-postgres-adapter 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
File without changes
@@ -0,0 +1,700 @@
1
+ """Postgres implementation of the WarehouseAdapter port.
2
+
3
+ One class covers both roles Continuo asks of an engine:
4
+
5
+ * **Validation DDL** — ``ensure_schema``/``drop_schema`` manage the candidate
6
+ schema; empty builds go through ``CREATE TABLE … AS (…) WITH NO DATA``, empty
7
+ clones through ``CTAS … WHERE 1=0``, and ``check_binds`` EXPLAINs a read
8
+ without scanning data. Schema creation is serialized on a session advisory
9
+ lock because parallel root validation nodes race on ``CREATE SCHEMA``.
10
+ * **Python-node data-plane I/O** — ``fetch`` executes one declared read and
11
+ returns an Arrow table; ``ensure_table``/``load`` build and atomically
12
+ replace a table's contents.
13
+
14
+ The connection runs with ``autocommit = True``: every method owns its
15
+ transaction explicitly rather than relying on psycopg2's implicit
16
+ per-connection transaction. A single-statement method (``drop_schema``,
17
+ ``build_empty_from_sql``, ``clone_empty_from_prod``) needs nothing further —
18
+ under autocommit each statement commits as it runs. A method that must apply
19
+ several statements atomically (``build_empty_from_columns``, ``check_binds``,
20
+ ``ensure_table``, ``load``) wraps them in its own explicit
21
+ ``BEGIN``/``COMMIT``, with a ``ROLLBACK`` on the error path so a mid-sequence
22
+ failure leaves no partial change and no aborted transaction behind for the
23
+ next call on the same connection.
24
+ """
25
+ import hashlib
26
+ import logging
27
+ import os
28
+
29
+ from typing import Any
30
+
31
+ import psycopg2 # type: ignore[import-untyped]
32
+ import pyarrow as pa # type: ignore[import-untyped]
33
+
34
+ from continuo_engine_contract.config import ensure_known_keys # type: ignore[import-untyped]
35
+ from continuo_engine_contract.port import WarehouseAdapter # type: ignore[import-untyped]
36
+ from continuo_engine_contract.sql import ensure_single_read # type: ignore[import-untyped]
37
+ from continuo_engine_contract.types import validate_column_type # type: ignore[import-untyped]
38
+ from psycopg2 import errors as pg_errors # type: ignore[import-untyped]
39
+ from psycopg2 import sql as pg_sql # type: ignore[import-untyped]
40
+ from psycopg2.extras import execute_values # type: ignore[import-untyped]
41
+
42
+ logger = logging.getLogger("continuo_postgres_adapter")
43
+
44
+ # The postgres physical-layout vocabulary, mirroring dbt-postgres's own `indexes`
45
+ # config so the graph reads a python node's layout the way it reads a dbt model's.
46
+ # This adapter is the sole owner and enforcer of the vocabulary — the contract
47
+ # loader validates `config` as shape only and stays engine-blind (see
48
+ # continuo_python_runtime/contract/loader.py).
49
+ _KNOWN_CONFIG_KEYS: tuple[str, ...] = ("indexes",)
50
+ _KNOWN_INDEX_KEYS: tuple[str, ...] = ("columns", "unique", "type", "name")
51
+ # Access methods that can back a plain CREATE INDEX. The chosen value is
52
+ # interpolated into DDL, so this allowlist is an injection guard as much as a
53
+ # spelling check — the same discipline types.validate_column_type applies to types.
54
+ _INDEX_TYPES: tuple[str, ...] = ("brin", "btree", "gin", "gist", "hash", "spgist")
55
+ # Structural restrictions postgres places on the allowlisted access methods.
56
+ # Both were confirmed against postgres:16: btree is the only one that can enforce
57
+ # uniqueness, and hash is the only one that refuses a multicolumn index (brin,
58
+ # gin, gist and spgist all accept several columns). Checking them here turns a
59
+ # raw psycopg2 error into a message naming the config key at fault.
60
+ _UNIQUE_INDEX_TYPE = "btree"
61
+ _SINGLE_COLUMN_INDEX_TYPES = frozenset({"hash"})
62
+ # Postgres's own default access method, spelled out so the emitted DDL always
63
+ # carries an explicit USING clause. `USING btree` and a bare CREATE INDEX build
64
+ # the identical index, so naming it changes nothing at the warehouse while
65
+ # keeping one code path for every access method.
66
+ _DEFAULT_INDEX_TYPE = "btree"
67
+ # NAMEDATALEN - 1. Postgres silently truncates longer identifiers, so any
68
+ # index name over this limit -- derived default or explicit -- must be
69
+ # truncated here too, or the emitted DDL's name would not match the name
70
+ # postgres actually stores.
71
+ _MAX_IDENTIFIER_BYTES = 63
72
+
73
+
74
+ def _truncate_explicit_identifier(name: str) -> str:
75
+ """Truncate an explicit, author-given identifier to at most 63 bytes.
76
+
77
+ Truncates on encoded UTF-8 bytes, not characters: postgres's limit is
78
+ bytes, and a multibyte identifier would slip past a character-wise
79
+ slice. This is a *plain* byte-for-byte cut -- no digest suffix -- so it
80
+ matches postgres's own NAMEDATALEN truncation exactly: the normalized
81
+ name this function returns is always what postgres will actually store,
82
+ never something it silently truncates further behind our back.
83
+
84
+ Unlike :func:`_index_name`'s derived defaults, an explicit ``name:`` is
85
+ something the author chose, so this deliberately does NOT inject a
86
+ disambiguating digest the way ``_index_name`` does. Two different
87
+ explicit names that happen to truncate to the same 63-byte prefix
88
+ collide here on purpose: the caller's duplicate-name check (over these
89
+ already-truncated names) surfaces that as a rejection instead of the two
90
+ entries silently colliding at the warehouse, where ``CREATE INDEX IF NOT
91
+ EXISTS`` would skip every index but the first that resolves to the same
92
+ stored identifier.
93
+ """
94
+ encoded = name.encode("utf-8")
95
+ if len(encoded) <= _MAX_IDENTIFIER_BYTES:
96
+ return name
97
+ return encoded[:_MAX_IDENTIFIER_BYTES].decode("utf-8", "ignore")
98
+
99
+
100
+ def _index_name(table: str, columns: list[str]) -> str:
101
+ """Return the default index name for *columns* on *table*, within 63 bytes.
102
+
103
+ Truncates on encoded UTF-8 bytes, not characters: postgres's limit is
104
+ bytes, and a multibyte identifier would slip past a character-wise slice.
105
+
106
+ A name that overflows the limit keeps a uniqueness-preserving suffix --
107
+ ``"_"`` plus the leading 8 hex characters of
108
+ ``sha256(<full untruncated name>)`` -- rather than a bare truncation: a
109
+ long enough table name leaves ``ix_<table>_`` alone at or past 63 bytes,
110
+ so every index on it would otherwise truncate to the *same* default name,
111
+ and ``CREATE INDEX IF NOT EXISTS`` would then silently skip every index
112
+ after the first. The digest is taken over the full name, so two
113
+ different column lists that truncate to the same prefix still get
114
+ different suffixes. Unlike :func:`_truncate_explicit_identifier`, nobody
115
+ chose this literal string, so injecting a digest to keep it collision-free
116
+ is a service rather than a surprise.
117
+
118
+ The name deliberately covers only the table and the column list, NOT the
119
+ access method or uniqueness. It names indexes on PRODUCTION tables under
120
+ ``CREATE INDEX IF NOT EXISTS``, so folding more of the definition into it
121
+ would rename every index that already exists and build a second copy
122
+ alongside each one. Two entries over the same columns therefore resolve to
123
+ one name and are rejected as duplicates by :func:`_validated_indexes`; an
124
+ author who genuinely wants two access methods over one column set gives
125
+ one of them an explicit ``name:``.
126
+ """
127
+ name = f"ix_{table}_{'_'.join(columns)}"
128
+ encoded = name.encode("utf-8")
129
+ if len(encoded) <= _MAX_IDENTIFIER_BYTES:
130
+ return name
131
+ digest = hashlib.sha256(encoded).hexdigest()[:8]
132
+ truncated = encoded[: _MAX_IDENTIFIER_BYTES - 9].decode("utf-8", "ignore")
133
+ return f"{truncated}_{digest}"
134
+
135
+
136
+ def _validated_indexes(
137
+ config: dict[str, Any] | None, table: str, column_names: list[str]
138
+ ) -> list[dict[str, Any]]:
139
+ """Validate *config* against the postgres 'indexes' vocabulary; return normalized entries.
140
+
141
+ Every key — top level and per index entry — is checked before the caller
142
+ emits a single statement, so a malformed config never leaves a half-built
143
+ table behind (fail closed). A ``None`` *config* returns ``[]``. Each
144
+ returned entry carries ``columns`` (list[str]), ``unique`` (bool), ``type``
145
+ (str, the access method, defaulting to ``btree``), and ``name`` (str,
146
+ defaulted via :func:`_index_name` when not given explicitly).
147
+
148
+ Index columns are checked against the node's own declared output columns:
149
+ an index on a column the node does not produce is an authoring error, and
150
+ pre-checking also keeps every identifier that reaches DDL to a name the
151
+ spec itself declared.
152
+
153
+ The structural access-method restrictions (``_UNIQUE_INDEX_TYPE``,
154
+ ``_SINGLE_COLUMN_INDEX_TYPES``) are checked for the error message rather
155
+ than for safety — postgres rejects those combinations itself — so the
156
+ author reads "index 'unique' requires type 'btree'" instead of a raw
157
+ psycopg2 traceback.
158
+
159
+ Raises:
160
+ ValueError: Naming the offending key, for any of: an unrecognized
161
+ top-level key; ``indexes`` not a list, or a non-mapping element;
162
+ an unrecognized index key; ``columns`` missing, not a list, empty,
163
+ or containing a non-string or an empty string; an index column not
164
+ present in *column_names*; ``unique`` present and not a bool;
165
+ ``type`` outside :data:`_INDEX_TYPES`, or structurally incompatible
166
+ with ``unique``/several columns; ``name`` present and not a
167
+ non-empty string; or two entries resolving to the same index name
168
+ -- letting that through would silently drop every colliding index
169
+ but the first under ``CREATE INDEX IF NOT EXISTS``. An explicit
170
+ ``name`` over 63 bytes is truncated the same way postgres itself
171
+ would truncate it (:func:`_truncate_explicit_identifier`) *before*
172
+ this comparison runs, so two over-long explicit names that share
173
+ only their first 63 bytes are caught here too, not just exact
174
+ duplicates.
175
+ """
176
+ if config is None:
177
+ return []
178
+ ensure_known_keys(config, _KNOWN_CONFIG_KEYS, "postgres")
179
+
180
+ # .get, not a bare subscript: this function's whole job is failing closed
181
+ # with a named message, and `indexes` is only guaranteed present while
182
+ # _KNOWN_CONFIG_KEYS has exactly one member. A second postgres key would
183
+ # otherwise turn `{"newkey": ...}` into a bare KeyError right here.
184
+ raw_indexes = config.get("indexes", [])
185
+ if not isinstance(raw_indexes, list):
186
+ raise ValueError(f"config 'indexes' must be a list, got {type(raw_indexes).__name__}")
187
+
188
+ declared = set(column_names)
189
+ normalized: list[dict[str, Any]] = []
190
+ for entry in raw_indexes:
191
+ ensure_known_keys(
192
+ entry, _KNOWN_INDEX_KEYS, "postgres", where="config 'indexes' entry"
193
+ )
194
+
195
+ columns = entry.get("columns")
196
+ if (
197
+ not isinstance(columns, list)
198
+ or not columns
199
+ or not all(isinstance(c, str) and c for c in columns)
200
+ ):
201
+ raise ValueError(
202
+ f"index 'columns' must be a non-empty list of column names, got {columns!r}"
203
+ )
204
+ missing = [c for c in columns if c not in declared]
205
+ if missing:
206
+ raise ValueError(
207
+ f"index on undeclared column(s) {missing!r}; "
208
+ f"declared columns: {sorted(declared)!r}"
209
+ )
210
+
211
+ unique = entry.get("unique", False)
212
+ if not isinstance(unique, bool):
213
+ raise ValueError(f"index 'unique' must be a boolean, got {unique!r}")
214
+
215
+ index_type = entry.get("type", _DEFAULT_INDEX_TYPE)
216
+ if index_type not in _INDEX_TYPES:
217
+ raise ValueError(
218
+ f"unsupported index 'type' {index_type!r}; supported: {', '.join(_INDEX_TYPES)}"
219
+ )
220
+ if unique and index_type != _UNIQUE_INDEX_TYPE:
221
+ raise ValueError(
222
+ f"index 'unique' requires type {_UNIQUE_INDEX_TYPE!r}; postgres cannot "
223
+ f"enforce uniqueness with access method {index_type!r}"
224
+ )
225
+ if index_type in _SINGLE_COLUMN_INDEX_TYPES and len(columns) > 1:
226
+ raise ValueError(
227
+ f"index 'type' {index_type!r} indexes a single column; "
228
+ f"got {len(columns)}: {', '.join(repr(c) for c in columns)}"
229
+ )
230
+
231
+ name = entry.get("name")
232
+ if name is not None and (not isinstance(name, str) or not name):
233
+ raise ValueError(f"index 'name' must be a non-empty string, got {name!r}")
234
+
235
+ normalized.append({
236
+ "columns": list(columns),
237
+ "unique": unique,
238
+ "type": index_type,
239
+ "name": (
240
+ _truncate_explicit_identifier(name)
241
+ if name is not None
242
+ else _index_name(table, columns)
243
+ ),
244
+ })
245
+
246
+ # Reject before returning -- this is the fail-closed gate that runs
247
+ # before any DDL is emitted, so a collision (explicit `name` vs. another
248
+ # explicit `name`, or vs. a derived default) must surface here rather
249
+ # than let CREATE INDEX IF NOT EXISTS silently skip every index but the
250
+ # first that resolves to the same name.
251
+ seen_names: set[str] = set()
252
+ for index in normalized:
253
+ index_name = index["name"]
254
+ if index_name in seen_names:
255
+ raise ValueError(f"duplicate index name: {index_name!r}")
256
+ seen_names.add(index_name)
257
+
258
+ return normalized
259
+
260
+
261
+ def _index_ddl(
262
+ schema: str, table: str, index: dict[str, Any], *, if_not_exists: bool
263
+ ) -> "pg_sql.Composed":
264
+ """Build one ``CREATE [UNIQUE] INDEX`` statement for a validated *index*.
265
+
266
+ Every identifier — index name, schema, table, access method and each
267
+ column — goes through ``pg_sql.Identifier``; never raw interpolation. The
268
+ ``UNIQUE`` keyword and the ``IF NOT EXISTS`` clause are constants from this
269
+ module, and the access method comes from this module's own allowlist, not
270
+ from author input.
271
+
272
+ *if_not_exists* is True on the ``ensure_table`` path, which creates a table
273
+ that may already exist and already carry its indexes, and False on the
274
+ ``build_empty_from_columns`` path, which has just dropped and recreated the
275
+ table so no index of that name can survive.
276
+ """
277
+ return pg_sql.SQL("CREATE {}INDEX {}{} ON {}.{} USING {} ({})").format(
278
+ pg_sql.SQL("UNIQUE ") if index["unique"] else pg_sql.SQL(""),
279
+ pg_sql.SQL("IF NOT EXISTS " if if_not_exists else ""),
280
+ pg_sql.Identifier(index["name"]),
281
+ pg_sql.Identifier(schema),
282
+ pg_sql.Identifier(table),
283
+ pg_sql.Identifier(index["type"]),
284
+ pg_sql.SQL(", ").join(pg_sql.Identifier(c) for c in index["columns"]),
285
+ )
286
+
287
+
288
+ def _column_ddl(columns: list[dict[str, Any]]) -> "pg_sql.Composed":
289
+ """Compile declared typed *columns* into a CREATE TABLE column list.
290
+
291
+ The single typed-DDL compiler behind both build paths — the validation
292
+ rebuild (:meth:`PostgresAdapter.build_empty_from_columns`) and the runtime
293
+ create (:meth:`PostgresAdapter.ensure_table`) — so the two cannot drift into
294
+ emitting different shapes for the same declared columns. Callers must have
295
+ run :func:`~continuo_engine_contract.types.validate_column_type` over every
296
+ ``type`` first: the type text is interpolated as SQL, only the name is
297
+ identifier-quoted. ``nullable`` absent means True, i.e. no constraint.
298
+ """
299
+ col_defs = []
300
+ for col in columns:
301
+ parts = [pg_sql.Identifier(col["name"]), pg_sql.SQL(col["type"])]
302
+ if not col.get("nullable", True):
303
+ parts.append(pg_sql.SQL("NOT NULL"))
304
+ col_defs.append(pg_sql.SQL(" ").join(parts))
305
+ return pg_sql.SQL(", ").join(col_defs)
306
+
307
+
308
+ def _arrow_table_from_rows(colnames: list[str], rows: list[tuple[Any, ...]]) -> "pa.Table":
309
+ """Build a column-wise Arrow table from cursor description names and fetched rows.
310
+
311
+ Type inference is left to pyarrow over the Python values psycopg2 yields
312
+ (Decimal -> decimal128, date -> date32, datetime -> timestamp, bool, int,
313
+ float, str). An empty result produces a 0-row table whose columns are typed
314
+ ``null`` (``pa.nulls(0)`` per column) rather than inferred — the script and
315
+ ``conform()`` define the output shape, so this is acceptable.
316
+
317
+ Raises
318
+ ------
319
+ ValueError
320
+ If *colnames* contains duplicates (e.g. ``SELECT 1 AS id, 2 AS id``):
321
+ building a dict column-wise would otherwise silently drop all but the
322
+ last occurrence, corrupting the result instead of surfacing an error.
323
+ """
324
+ seen: set[str] = set()
325
+ duplicates: set[str] = set()
326
+ for name in colnames:
327
+ if name in seen:
328
+ duplicates.add(name)
329
+ seen.add(name)
330
+ if duplicates:
331
+ raise ValueError(
332
+ f"duplicate column name(s) in SELECT result: {sorted(duplicates)!r}"
333
+ )
334
+ if not rows:
335
+ return pa.table({name: pa.nulls(0) for name in colnames})
336
+ by_column = list(zip(*rows))
337
+ return pa.table({name: pa.array(values) for name, values in zip(colnames, by_column)})
338
+
339
+
340
+ class PostgresAdapter(WarehouseAdapter):
341
+ """WarehouseAdapter speaking postgres over a psycopg2 connection."""
342
+
343
+ def __init__(self, conn: "psycopg2.extensions.connection") -> None:
344
+ self._conn = conn
345
+ self._conn.autocommit = True
346
+
347
+ @classmethod
348
+ def required_env(cls) -> list[str]:
349
+ """Vars that must be non-empty before connecting."""
350
+ return ["POSTGRES_HOST", "POSTGRES_DB", "POSTGRES_USER"]
351
+
352
+ @classmethod
353
+ def from_env(cls) -> "PostgresAdapter":
354
+ """Connect from POSTGRES_* env (port defaults 5432, password empty)."""
355
+ conn = psycopg2.connect(
356
+ host=os.environ["POSTGRES_HOST"],
357
+ port=os.environ.get("POSTGRES_PORT", "5432"),
358
+ dbname=os.environ["POSTGRES_DB"],
359
+ user=os.environ["POSTGRES_USER"],
360
+ password=os.environ.get("POSTGRES_PASSWORD", ""),
361
+ )
362
+ return cls(conn)
363
+
364
+ # --- Schema lifecycle ---------------------------------------------------
365
+
366
+ def ensure_schema(self, schema: str) -> None:
367
+ """Idempotently create *schema*; safe under concurrent callers.
368
+
369
+ Race-safe: root validation nodes dispatch in parallel and can collide
370
+ on CREATE SCHEMA. Serialize on a session advisory lock keyed by schema
371
+ name; tolerate DuplicateSchema/UniqueViolation as a second line of
372
+ defense. The session advisory lock is held independently of any
373
+ transaction, and under ``autocommit = True`` each statement here runs
374
+ (and commits, or aborts on failure) as its own single-statement
375
+ transaction, so no statement ever finds the connection sitting in a
376
+ transaction left open by a previous call.
377
+
378
+ The explicit ``self._conn.commit()``/``self._conn.rollback()`` calls
379
+ below are therefore no-ops against the server — there is never an
380
+ open multi-statement transaction for them to act on under autocommit
381
+ — but they are kept as a defensive no-op rather than removed: they
382
+ cost nothing, and they keep this method's shape identical whether or
383
+ not a future caller ever constructs this adapter over a
384
+ non-autocommit connection.
385
+ """
386
+ with self._conn.cursor() as cur:
387
+ cur.execute("SELECT pg_advisory_lock(hashtext(%s))", (schema,))
388
+ self._conn.commit()
389
+ try:
390
+ stmt = pg_sql.SQL("CREATE SCHEMA IF NOT EXISTS {}").format(
391
+ pg_sql.Identifier(schema)
392
+ )
393
+ logger.info("ensuring schema %s exists", schema)
394
+ try:
395
+ cur.execute(stmt)
396
+ self._conn.commit()
397
+ except (pg_errors.DuplicateSchema, pg_errors.UniqueViolation):
398
+ self._conn.rollback()
399
+ logger.info(
400
+ "schema %s already exists (concurrent create); continuing", schema
401
+ )
402
+ except Exception:
403
+ self._conn.rollback()
404
+ raise
405
+ finally:
406
+ cur.execute("SELECT pg_advisory_unlock(hashtext(%s))", (schema,))
407
+ self._conn.commit()
408
+
409
+ def drop_schema(self, schema: str) -> None:
410
+ """Idempotently drop *schema* and everything in it; no-op if absent."""
411
+ with self._conn.cursor() as cur:
412
+ stmt = pg_sql.SQL("DROP SCHEMA IF EXISTS {} CASCADE").format(
413
+ pg_sql.Identifier(schema)
414
+ )
415
+ logger.info("dropping candidate schema %s", schema)
416
+ cur.execute(stmt)
417
+
418
+ # --- Validation builds --------------------------------------------------
419
+
420
+ def build_empty_from_sql(self, schema: str, table: str, compiled_sql: str) -> None:
421
+ """Create ``schema.table`` empty, shaped by the compiled SELECT."""
422
+ # Strip any trailing terminator so the SELECT nests cleanly inside AS ( ... ).
423
+ inner = compiled_sql.strip().rstrip(";").strip()
424
+ with self._conn.cursor() as cur:
425
+ cur.execute(
426
+ pg_sql.SQL("DROP TABLE IF EXISTS {}.{}").format(
427
+ pg_sql.Identifier(schema), pg_sql.Identifier(table)
428
+ )
429
+ )
430
+ cur.execute(
431
+ pg_sql.SQL("CREATE TABLE {}.{} AS ({}) WITH NO DATA").format(
432
+ pg_sql.Identifier(schema),
433
+ pg_sql.Identifier(table),
434
+ pg_sql.SQL(inner),
435
+ )
436
+ )
437
+
438
+ def clone_empty_from_prod(self, candidate_schema: str, prod_schema: str, table: str) -> None:
439
+ """Create ``candidate_schema.table`` empty, shaped like ``prod_schema.table``."""
440
+ with self._conn.cursor() as cur:
441
+ cur.execute(
442
+ pg_sql.SQL("DROP TABLE IF EXISTS {}.{}").format(
443
+ pg_sql.Identifier(candidate_schema), pg_sql.Identifier(table)
444
+ )
445
+ )
446
+ cur.execute(
447
+ pg_sql.SQL(
448
+ "CREATE TABLE {}.{} AS SELECT * FROM {}.{} WHERE 1=0"
449
+ ).format(
450
+ pg_sql.Identifier(candidate_schema),
451
+ pg_sql.Identifier(table),
452
+ pg_sql.Identifier(prod_schema),
453
+ pg_sql.Identifier(table),
454
+ )
455
+ )
456
+
457
+ def build_empty_from_columns(
458
+ self, schema: str, table: str, columns: list[dict], config: dict
459
+ ) -> None:
460
+ """Create ``schema.table`` empty from declared typed columns (drop-then-create).
461
+
462
+ *config* carries this engine's physical-layout vocabulary: ``indexes``,
463
+ a list of ``{"columns": [...], "unique": bool, "type": str, "name":
464
+ str}`` mirroring dbt-postgres's own config. Every key — top level and
465
+ per entry — is validated before the first statement runs, and the
466
+ rebuild itself runs in one transaction, so a bad block fails the
467
+ release gate without leaving a half-built table behind. An empty
468
+ *config* emits exactly the statements contract 0.4.0 emitted, plus the
469
+ transaction around them.
470
+ """
471
+ for col in columns:
472
+ validate_column_type(col["type"])
473
+ col_ddl = _column_ddl(columns)
474
+ indexes = _validated_indexes(config, table, [col["name"] for col in columns])
475
+ with self._conn.cursor() as cur:
476
+ # Postgres DDL is transactional, so the drop, the create and every
477
+ # index either all land or none do. Without this, an index postgres
478
+ # refuses — a missing operator class for the column's type, say,
479
+ # which no allowlist here can predict — would leave the table
480
+ # rebuilt and partially indexed. BEGIN it explicitly for the same
481
+ # reason check_binds does.
482
+ cur.execute("BEGIN")
483
+ try:
484
+ cur.execute(
485
+ pg_sql.SQL("DROP TABLE IF EXISTS {}.{}").format(
486
+ pg_sql.Identifier(schema), pg_sql.Identifier(table)
487
+ )
488
+ )
489
+ cur.execute(
490
+ pg_sql.SQL("CREATE TABLE {}.{} ({})").format(
491
+ pg_sql.Identifier(schema),
492
+ pg_sql.Identifier(table),
493
+ col_ddl,
494
+ )
495
+ )
496
+ for index in indexes:
497
+ logger.info(
498
+ "creating %sindex on %s.%s (%s) using %s",
499
+ "unique " if index["unique"] else "",
500
+ schema, table, ", ".join(index["columns"]), index["type"],
501
+ )
502
+ cur.execute(_index_ddl(schema, table, index, if_not_exists=False))
503
+ except Exception:
504
+ # The failure itself is what the caller must see, so a rollback
505
+ # that also fails only logs. Skipping it would strand the
506
+ # session in an aborted transaction and fail every later
507
+ # statement in the same Job on a poisoned connection.
508
+ try:
509
+ cur.execute("ROLLBACK")
510
+ except psycopg2.Error as rollback_exc:
511
+ logger.error("build rollback failed: %s", rollback_exc)
512
+ raise
513
+ cur.execute("COMMIT")
514
+
515
+ def check_binds(self, sql: str) -> None:
516
+ """Verify *sql* binds against current schema state; EXPLAIN scans no data.
517
+
518
+ *sql* must be a single read query, and that is decided by parsing it —
519
+ see :func:`continuo_engine_contract.sql.ensure_single_read` — before
520
+ any of it reaches the server. Three defences, in order: the parse gate,
521
+ an explicit read-only transaction, and the subquery wrap.
522
+ """
523
+ # Layer 1. psycopg2 sends a parameterless statement over the
524
+ # simple-query protocol, which executes EVERY ``;``-separated statement
525
+ # in the batch. The wrap below is not enough on its own: a read that
526
+ # closes the wrap's paren and reopens it after an injected statement
527
+ # (``select 1) AS x; DELETE FROM t; SELECT * FROM (SELECT 1``) leaves
528
+ # the wrapped text balanced, and the DELETE runs for real. So decide the
529
+ # shape by parsing first.
530
+ ensure_single_read(sql, dialect="postgres")
531
+ inner = sql.strip().rstrip(";").strip()
532
+ with self._conn.cursor() as cur:
533
+ logger.info("bind-checking read via EXPLAIN")
534
+ # Layer 2. Backstop: even if the gate is ever bypassed, the engine
535
+ # itself refuses to mutate.
536
+ cur.execute("BEGIN READ ONLY")
537
+ try:
538
+ # Layer 3. A ``;`` inside the parentheses is a syntax error, so a
539
+ # naively stacked statement cannot survive even the server's own
540
+ # parse, and a non-query statement is not a legal subquery — do
541
+ # not "simplify" this back to `EXPLAIN {inner}`.
542
+ cur.execute(
543
+ pg_sql.SQL("EXPLAIN SELECT * FROM (\n{}\n) AS __check_binds__").format(
544
+ pg_sql.SQL(inner)
545
+ )
546
+ )
547
+ finally:
548
+ # Must run on the error path too: a failed EXPLAIN leaves the
549
+ # session in an aborted transaction, and without this every
550
+ # later read in the same Job would fail on a poisoned
551
+ # connection. A ROLLBACK failure only logs — the EXPLAIN's own
552
+ # error, if any, is the one worth propagating.
553
+ try:
554
+ cur.execute("ROLLBACK")
555
+ except psycopg2.Error as rollback_exc:
556
+ logger.error("bind-check rollback failed: %s", rollback_exc)
557
+
558
+ # --- Python-node data plane ---------------------------------------------
559
+
560
+ def fetch(self, sql: str) -> "pa.Table":
561
+ """Execute one declared read and return the result as an Arrow table."""
562
+ with self._conn.cursor() as cur:
563
+ try:
564
+ cur.execute(sql)
565
+ colnames = [d.name for d in cur.description] if cur.description else []
566
+ rows = list(cur.fetchall()) if cur.description else []
567
+ table = _arrow_table_from_rows(colnames, rows)
568
+ self._conn.commit()
569
+ except Exception:
570
+ self._conn.rollback()
571
+ raise
572
+ return table
573
+
574
+ @classmethod
575
+ def validate_config(
576
+ cls, config: dict[str, Any] | None, column_names: list[str]
577
+ ) -> None:
578
+ """Validate *config* against this engine's vocabulary, without connecting.
579
+
580
+ The harness calls this immediately after selecting the node, so a
581
+ malformed ``config`` — a singular ``index:`` typo, an index on an
582
+ undeclared column — fails in the first second of the run instead of
583
+ after the script has already computed its whole result. It is a
584
+ tripwire, not the enforcement point: ``ensure_table`` runs the very
585
+ same check again, unchanged, and remains the thing that guarantees no
586
+ DDL is emitted for a bad config. Both go through
587
+ :func:`_validated_indexes`, so the two cannot drift apart.
588
+
589
+ The index name derived here is thrown away, so a placeholder table
590
+ name is passed: nothing about naming is being validated (any string
591
+ is accepted), and no name reaches an error message.
592
+
593
+ Like ``config`` on ``ensure_table``, this method is not declared by
594
+ the abstract ``WarehouseAdapter`` in ``continuo-engine-contract``;
595
+ this repo ships both adapters and the harness as one coordinated
596
+ release. The harness skips the call for an adapter that does not
597
+ provide it (see ``docs/boundary-contract.md`` §13.4), which costs that
598
+ adapter only earliness, never enforcement.
599
+
600
+ Raises:
601
+ ValueError: Exactly as :func:`_validated_indexes` does.
602
+ """
603
+ _validated_indexes(config, "_", column_names)
604
+
605
+ def ensure_table(
606
+ self,
607
+ schema: str,
608
+ table: str,
609
+ columns: list[dict[str, Any]],
610
+ *,
611
+ config: dict[str, Any],
612
+ ) -> None:
613
+ """CREATE TABLE IF NOT EXISTS with typed DDL compiled from *columns*.
614
+
615
+ Each column dict carries ``name``, ``type`` (validated against the
616
+ contract's SQL type grammar), ``nullable`` (bool). *config* carries
617
+ this engine's physical-layout vocabulary — ``indexes`` — validated by
618
+ :func:`_validated_indexes` before any DDL runs, so a malformed config
619
+ never leaves a half-built table behind (fail closed). Required and
620
+ keyword-only, matching the contract 0.6.0 port signature. The create
621
+ and every index either all land or none do, in the same explicit
622
+ BEGIN/COMMIT/ROLLBACK shape build_empty_from_columns uses.
623
+ """
624
+ indexes = _validated_indexes(config, table, [col["name"] for col in columns])
625
+ for col in columns:
626
+ validate_column_type(col["type"])
627
+
628
+ self.ensure_schema(schema)
629
+
630
+ stmt = pg_sql.SQL("CREATE TABLE IF NOT EXISTS {}.{} ({})").format(
631
+ pg_sql.Identifier(schema),
632
+ pg_sql.Identifier(table),
633
+ _column_ddl(columns),
634
+ )
635
+ with self._conn.cursor() as cur:
636
+ cur.execute("BEGIN")
637
+ try:
638
+ logger.info("ensuring table %s.%s exists", schema, table)
639
+ cur.execute(stmt)
640
+ for index in indexes:
641
+ logger.info(
642
+ "ensuring index %s on %s.%s exists", index["name"], schema, table
643
+ )
644
+ cur.execute(_index_ddl(schema, table, index, if_not_exists=True))
645
+ cur.execute("COMMIT")
646
+ except Exception:
647
+ try:
648
+ cur.execute("ROLLBACK")
649
+ except psycopg2.Error as rollback_exc:
650
+ logger.error("ensure_table rollback failed: %s", rollback_exc)
651
+ raise
652
+
653
+ def load(self, schema: str, table: str, data: "pa.Table") -> None:
654
+ """Atomically replace ``schema.table``'s contents with *data*.
655
+
656
+ One explicit transaction: TRUNCATE then batched inserts
657
+ (``execute_values``, page_size 1000) in the Arrow table's column
658
+ order; commits on success, rolls back and re-raises on any error. A
659
+ 0-row *data* is just a TRUNCATE. The transaction is opened explicitly
660
+ (``BEGIN``) rather than relying on an implicit one, matching every
661
+ other multi-statement method on this class under ``autocommit = True``.
662
+ """
663
+ columns = data.schema.names
664
+ with self._conn.cursor() as cur:
665
+ cur.execute("BEGIN")
666
+ try:
667
+ cur.execute(
668
+ pg_sql.SQL("TRUNCATE {}.{}").format(
669
+ pg_sql.Identifier(schema), pg_sql.Identifier(table)
670
+ )
671
+ )
672
+ if data.num_rows:
673
+ rows = data.to_pylist()
674
+ values = [tuple(row[c] for c in columns) for row in rows]
675
+ insert_prefix = pg_sql.SQL("INSERT INTO {}.{} ({})").format(
676
+ pg_sql.Identifier(schema),
677
+ pg_sql.Identifier(table),
678
+ pg_sql.SQL(", ").join(pg_sql.Identifier(c) for c in columns),
679
+ )
680
+ # execute_values reparses every percent token after composable
681
+ # identifiers have rendered. Escape literal identifier percents,
682
+ # then append its one values-list expansion token unescaped.
683
+ insert_stmt = (
684
+ insert_prefix.as_string(cur).replace("%", "%%") + " VALUES %s"
685
+ )
686
+ logger.info(
687
+ "loading %d row(s) into %s.%s", data.num_rows, schema, table
688
+ )
689
+ execute_values(cur, insert_stmt, values, page_size=1000)
690
+ cur.execute("COMMIT")
691
+ except Exception:
692
+ try:
693
+ cur.execute("ROLLBACK")
694
+ except psycopg2.Error as rollback_exc:
695
+ logger.error("load rollback failed: %s", rollback_exc)
696
+ raise
697
+
698
+ def close(self) -> None:
699
+ """Release the underlying connection."""
700
+ self._conn.close()
@@ -0,0 +1,51 @@
1
+ Metadata-Version: 2.5
2
+ Name: continuo-postgres-adapter
3
+ Version: 0.2.0
4
+ Summary: Postgres warehouse adapter for Continuo: validation and python-node runtime.
5
+ Author: Simone Carolini
6
+ Maintainer: Simone Carolini
7
+ Classifier: Development Status :: 4 - Beta
8
+ Classifier: Intended Audience :: Developers
9
+ Classifier: Programming Language :: Python :: 3.14
10
+ Requires-Python: >=3.14
11
+ Requires-Dist: continuo-engine-contract==0.7.3
12
+ Requires-Dist: psycopg2-binary==2.9.12
13
+ Requires-Dist: pyarrow==25.0.1
14
+ Description-Content-Type: text/markdown
15
+
16
+ # continuo-postgres-adapter
17
+
18
+ Postgres engine-adapter library for Continuo python nodes. `PostgresAdapter`
19
+ implements `continuo_engine_contract.port.WarehouseAdapter` (from the
20
+ `continuo-engine-contract` package) — one class covering both the data plane
21
+ (`fetch` / `ensure_table` / `load`) and release-time validation
22
+ (`ensure_schema` / `drop_schema` / `build_empty_from_sql` /
23
+ `build_empty_from_columns` / `clone_empty_from_prod` / `check_binds`) — and
24
+ registers itself under entry-point group `continuo_engine.adapters` as
25
+ `postgres`.
26
+
27
+ Connection env: `POSTGRES_HOST`, `POSTGRES_DB`, `POSTGRES_USER` (required);
28
+ `POSTGRES_PORT` (default 5432), `POSTGRES_PASSWORD` (default empty).
29
+
30
+ The connection runs with `autocommit = True`; every method owns its
31
+ transaction explicitly. `load()` and `ensure_table()` open an explicit
32
+ `BEGIN` around their statements and `COMMIT`/`ROLLBACK` themselves, so a
33
+ mid-sequence failure leaves no partial change. `fetch()` executes a single
34
+ read and commits (or rolls back on error), never leaving an open transaction
35
+ dangling.
36
+
37
+ The `config:` index vocabulary is the union of the two vocabularies this
38
+ adapter inherited from the merge: `columns`, `unique`, `type`, and `name`,
39
+ with the strictest check from each side applied. The derived index name is
40
+ built from the table and the column list only — `type` and `unique` are not
41
+ folded into it, so two index entries on the same columns that differ only in
42
+ `type` (or only in `unique`) collide on that derived name and are rejected as
43
+ a duplicate. Give one of them an explicit `name:` when that is genuinely
44
+ intended.
45
+
46
+ Verification tier: unit tests are mock-free pure-logic tests (type-grammar
47
+ validation, Arrow table construction from rows); DDL/transactional behavior
48
+ (schema/table creation, load truncate+insert+rollback, validation DDL) is
49
+ verified against a live postgres:16 engine in
50
+ `tests/test_integration_runtime_postgres.py` and
51
+ `tests/test_integration_postgres_validation.py`.
@@ -0,0 +1,6 @@
1
+ continuo_postgres_adapter/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
2
+ continuo_postgres_adapter/adapter.py,sha256=FLXmigZjVNxgpqgRa45Qx_ywqSlWOyh0MWcRc3tkuiU,34133
3
+ continuo_postgres_adapter-0.2.0.dist-info/METADATA,sha256=HK7zGQi5dvYsDNayvVHH0Rnnm-dTNxl7bXv2gQDAZlU,2498
4
+ continuo_postgres_adapter-0.2.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
5
+ continuo_postgres_adapter-0.2.0.dist-info/entry_points.txt,sha256=nrayDbSl8uI7uc7ej6rJ9-HNd6H43W80t-bkUk5_CAU,88
6
+ continuo_postgres_adapter-0.2.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [continuo_engine.adapters]
2
+ postgres = continuo_postgres_adapter.adapter:PostgresAdapter