holypipe 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
holypipe/__init__.py ADDED
@@ -0,0 +1 @@
1
+ __version__ = "1.0.0"
File without changes
holypipe/api/routes.py ADDED
@@ -0,0 +1,479 @@
1
+ """HTTP + websocket API for HolyPipe."""
2
+ from __future__ import annotations
3
+
4
+ import datetime as dt
5
+
6
+ from croniter import croniter
7
+ from fastapi import APIRouter, BackgroundTasks, Depends, HTTPException, WebSocket, WebSocketDisconnect
8
+ from sqlalchemy import select
9
+ from sqlalchemy.orm import Session
10
+
11
+ from ..bus import encode, subscribe, unsubscribe
12
+ from ..cache import discovery_cache
13
+ from ..connectors import DESTINATION_TYPES, SOURCE_TYPES, build_destination, build_source, config_spec
14
+ from ..connectors.base import ConnectorError
15
+ from ..connectors.dsn import DsnError, parse_dsn
16
+ from ..db import get_session, session_scope
17
+ from ..engine.cdc import supervisor
18
+ from ..engine.sync import run_batch_sync
19
+ from ..models import Connection, Destination, LogEntry, Source, SyncRun
20
+ from ..timeutil import utcnow
21
+ from . import schemas as sch
22
+
23
+ router = APIRouter(prefix="/api")
24
+
25
+
26
+ def _resolve_config(body: sch.ConnectorIn) -> dict:
27
+ """Config from the form fields, or parsed from a pasted connection URI."""
28
+ if body.uri:
29
+ try:
30
+ return parse_dsn(body.type, body.uri)
31
+ except DsnError as exc:
32
+ raise HTTPException(400, str(exc)) from exc
33
+ return body.config
34
+
35
+
36
+ def _next_cron_fire(expression: str, base: dt.datetime | None = None) -> dt.datetime:
37
+ try:
38
+ itr = croniter(expression, base or utcnow())
39
+ return itr.get_next(dt.datetime)
40
+ except (ValueError, KeyError) as exc:
41
+ raise HTTPException(400, f"Invalid cron expression: {exc}") from exc
42
+
43
+
44
+ # ---------------------------------------------------------------------------
45
+ # Connector metadata
46
+ # ---------------------------------------------------------------------------
47
+ @router.get("/connector-types")
48
+ def connector_types():
49
+ def describe(types: dict, kind: str) -> list[dict]:
50
+ return [{"type": t, "fields": config_spec(kind, t)} for t in types]
51
+
52
+ return {
53
+ "sources": describe(SOURCE_TYPES, "source"),
54
+ "destinations": describe(DESTINATION_TYPES, "destination"),
55
+ }
56
+
57
+
58
+ # ---------------------------------------------------------------------------
59
+ # Sources
60
+ # ---------------------------------------------------------------------------
61
+ @router.get("/sources", response_model=list[sch.ConnectorOut])
62
+ def list_sources(db: Session = Depends(get_session)):
63
+ return db.execute(select(Source).order_by(Source.created_at.desc())).scalars().all()
64
+
65
+
66
+ @router.post("/sources", response_model=sch.ConnectorOut)
67
+ def create_source(body: sch.ConnectorIn, db: Session = Depends(get_session)):
68
+ if body.type not in SOURCE_TYPES:
69
+ raise HTTPException(400, f"Unknown source type '{body.type}'")
70
+ if db.execute(select(Source).where(Source.name == body.name)).scalar_one_or_none():
71
+ raise HTTPException(409, "A source with this name already exists")
72
+ src = Source(name=body.name, type=body.type, config=_resolve_config(body))
73
+ db.add(src)
74
+ db.commit()
75
+ db.refresh(src)
76
+ return src
77
+
78
+
79
+ @router.get("/sources/{source_id}", response_model=sch.ConnectorOut)
80
+ def get_source(source_id: str, db: Session = Depends(get_session)):
81
+ src = db.get(Source, source_id)
82
+ if not src:
83
+ raise HTTPException(404, "Source not found")
84
+ return src
85
+
86
+
87
+ @router.put("/sources/{source_id}", response_model=sch.ConnectorOut)
88
+ def update_source(source_id: str, body: sch.ConnectorIn, db: Session = Depends(get_session)):
89
+ src = db.get(Source, source_id)
90
+ if not src:
91
+ raise HTTPException(404, "Source not found")
92
+ src.name, src.type, src.config = body.name, body.type, _resolve_config(body)
93
+ db.commit()
94
+ db.refresh(src)
95
+ discovery_cache.invalidate(source_id)
96
+ return src
97
+
98
+
99
+ @router.delete("/sources/{source_id}")
100
+ def delete_source(source_id: str, db: Session = Depends(get_session)):
101
+ src = db.get(Source, source_id)
102
+ if not src:
103
+ raise HTTPException(404, "Source not found")
104
+ if src.connections:
105
+ raise HTTPException(409, "Source is used by an existing connection")
106
+ db.delete(src)
107
+ db.commit()
108
+ discovery_cache.invalidate(source_id)
109
+ return {"ok": True}
110
+
111
+
112
+ @router.post("/sources/{source_id}/test", response_model=sch.TestResult)
113
+ def test_source(source_id: str, db: Session = Depends(get_session)):
114
+ src = db.get(Source, source_id)
115
+ if not src:
116
+ raise HTTPException(404, "Source not found")
117
+ try:
118
+ conn = build_source(src.type, src.config)
119
+ message = conn.test()
120
+ conn.close()
121
+ return sch.TestResult(ok=True, message=message)
122
+ except ConnectorError as exc:
123
+ return sch.TestResult(ok=False, message=str(exc))
124
+ except Exception as exc: # noqa: BLE001
125
+ return sch.TestResult(ok=False, message=str(exc))
126
+
127
+
128
+ @router.post("/sources/test-config", response_model=sch.TestResult)
129
+ def test_source_config(body: sch.ConnectorIn):
130
+ if body.type not in SOURCE_TYPES:
131
+ raise HTTPException(400, f"Unknown source type '{body.type}'")
132
+ try:
133
+ conn = build_source(body.type, _resolve_config(body))
134
+ message = conn.test()
135
+ conn.close()
136
+ return sch.TestResult(ok=True, message=message)
137
+ except Exception as exc: # noqa: BLE001
138
+ return sch.TestResult(ok=False, message=str(exc))
139
+
140
+
141
+ @router.get("/sources/{source_id}/discover", response_model=list[sch.DiscoveredStream])
142
+ def discover_source(source_id: str, refresh: bool = False, db: Session = Depends(get_session)):
143
+ src = db.get(Source, source_id)
144
+ if not src:
145
+ raise HTTPException(404, "Source not found")
146
+
147
+ if refresh:
148
+ discovery_cache.invalidate(source_id)
149
+
150
+ cached = discovery_cache.get(source_id)
151
+ if cached is not None:
152
+ return cached
153
+
154
+ try:
155
+ conn = build_source(src.type, src.config)
156
+ streams = conn.discover()
157
+ conn.close()
158
+ except ConnectorError as exc:
159
+ raise HTTPException(400, str(exc)) from exc
160
+
161
+ result = [
162
+ sch.DiscoveredStream(
163
+ name=s.name, table=s.table, namespace=s.namespace,
164
+ columns=[sch.DiscoveredColumn(**c.as_dict()) for c in s.columns],
165
+ primary_key=s.primary_key,
166
+ )
167
+ for s in streams
168
+ ]
169
+ # Schema discovery scans information_schema/sqlite_master, which can be
170
+ # slow on databases with thousands of tables — cache briefly so opening
171
+ # the connection wizard repeatedly doesn't re-scan every time.
172
+ discovery_cache.set(source_id, result)
173
+ return result
174
+
175
+
176
+ # ---------------------------------------------------------------------------
177
+ # Destinations
178
+ # ---------------------------------------------------------------------------
179
+ @router.get("/destinations", response_model=list[sch.ConnectorOut])
180
+ def list_destinations(db: Session = Depends(get_session)):
181
+ return db.execute(select(Destination).order_by(Destination.created_at.desc())).scalars().all()
182
+
183
+
184
+ @router.post("/destinations", response_model=sch.ConnectorOut)
185
+ def create_destination(body: sch.ConnectorIn, db: Session = Depends(get_session)):
186
+ if body.type not in DESTINATION_TYPES:
187
+ raise HTTPException(400, f"Unknown destination type '{body.type}'")
188
+ if db.execute(select(Destination).where(Destination.name == body.name)).scalar_one_or_none():
189
+ raise HTTPException(409, "A destination with this name already exists")
190
+ dst = Destination(name=body.name, type=body.type, config=_resolve_config(body))
191
+ db.add(dst)
192
+ db.commit()
193
+ db.refresh(dst)
194
+ return dst
195
+
196
+
197
+ @router.get("/destinations/{destination_id}", response_model=sch.ConnectorOut)
198
+ def get_destination(destination_id: str, db: Session = Depends(get_session)):
199
+ dst = db.get(Destination, destination_id)
200
+ if not dst:
201
+ raise HTTPException(404, "Destination not found")
202
+ return dst
203
+
204
+
205
+ @router.put("/destinations/{destination_id}", response_model=sch.ConnectorOut)
206
+ def update_destination(destination_id: str, body: sch.ConnectorIn, db: Session = Depends(get_session)):
207
+ dst = db.get(Destination, destination_id)
208
+ if not dst:
209
+ raise HTTPException(404, "Destination not found")
210
+ dst.name, dst.type, dst.config = body.name, body.type, _resolve_config(body)
211
+ db.commit()
212
+ db.refresh(dst)
213
+ return dst
214
+
215
+
216
+ @router.delete("/destinations/{destination_id}")
217
+ def delete_destination(destination_id: str, db: Session = Depends(get_session)):
218
+ dst = db.get(Destination, destination_id)
219
+ if not dst:
220
+ raise HTTPException(404, "Destination not found")
221
+ if dst.connections:
222
+ raise HTTPException(409, "Destination is used by an existing connection")
223
+ db.delete(dst)
224
+ db.commit()
225
+ return {"ok": True}
226
+
227
+
228
+ @router.post("/destinations/{destination_id}/test", response_model=sch.TestResult)
229
+ def test_destination(destination_id: str, db: Session = Depends(get_session)):
230
+ dst = db.get(Destination, destination_id)
231
+ if not dst:
232
+ raise HTTPException(404, "Destination not found")
233
+ try:
234
+ conn = build_destination(dst.type, dst.config)
235
+ message = conn.test()
236
+ conn.close()
237
+ return sch.TestResult(ok=True, message=message)
238
+ except Exception as exc: # noqa: BLE001
239
+ return sch.TestResult(ok=False, message=str(exc))
240
+
241
+
242
+ @router.post("/destinations/test-config", response_model=sch.TestResult)
243
+ def test_destination_config(body: sch.ConnectorIn):
244
+ if body.type not in DESTINATION_TYPES:
245
+ raise HTTPException(400, f"Unknown destination type '{body.type}'")
246
+ try:
247
+ conn = build_destination(body.type, _resolve_config(body))
248
+ message = conn.test()
249
+ conn.close()
250
+ return sch.TestResult(ok=True, message=message)
251
+ except Exception as exc: # noqa: BLE001
252
+ return sch.TestResult(ok=False, message=str(exc))
253
+
254
+
255
+ # ---------------------------------------------------------------------------
256
+ # Connections
257
+ # ---------------------------------------------------------------------------
258
+ @router.get("/connections", response_model=list[sch.ConnectionOut])
259
+ def list_connections(db: Session = Depends(get_session)):
260
+ return db.execute(select(Connection).order_by(Connection.created_at.desc())).scalars().all()
261
+
262
+
263
+ @router.post("/connections", response_model=sch.ConnectionOut)
264
+ def create_connection(body: sch.ConnectionIn, db: Session = Depends(get_session)):
265
+ source = db.get(Source, body.source_id)
266
+ destination = db.get(Destination, body.destination_id)
267
+ if not source or not destination:
268
+ raise HTTPException(404, "Source or destination not found")
269
+ if body.mode not in ("batch", "cdc"):
270
+ raise HTTPException(400, "mode must be 'batch' or 'cdc'")
271
+ if body.mode == "cdc" and not SOURCE_TYPES[source.type].supports_cdc:
272
+ raise HTTPException(400, f"{source.type} source does not support CDC")
273
+ if body.schedule_type not in ("interval", "cron"):
274
+ raise HTTPException(400, "schedule_type must be 'interval' or 'cron'")
275
+ for stream in body.streams:
276
+ if stream.get("sync_mode") == "xmin" and source.type != "postgres":
277
+ raise HTTPException(400, "xmin update method is only available for PostgreSQL sources")
278
+ if db.execute(select(Connection).where(Connection.name == body.name)).scalar_one_or_none():
279
+ raise HTTPException(409, "A connection with this name already exists")
280
+
281
+ now = utcnow()
282
+ next_run = now
283
+ if body.mode == "batch" and body.schedule_type == "cron":
284
+ next_run = _next_cron_fire(body.cron_expression or "", now)
285
+
286
+ conn = Connection(
287
+ name=body.name, source_id=body.source_id, destination_id=body.destination_id,
288
+ mode=body.mode, schedule_type=body.schedule_type, interval_seconds=body.interval_seconds,
289
+ cron_expression=body.cron_expression, enabled=body.enabled,
290
+ destination_namespace=body.destination_namespace, table_prefix=body.table_prefix,
291
+ streams=body.streams, state={}, next_run_at=next_run,
292
+ )
293
+ db.add(conn)
294
+ db.commit()
295
+ db.refresh(conn)
296
+ return conn
297
+
298
+
299
+ @router.get("/connections/{connection_id}", response_model=sch.ConnectionOut)
300
+ def get_connection(connection_id: str, db: Session = Depends(get_session)):
301
+ conn = db.get(Connection, connection_id)
302
+ if not conn:
303
+ raise HTTPException(404, "Connection not found")
304
+ return conn
305
+
306
+
307
+ @router.patch("/connections/{connection_id}", response_model=sch.ConnectionOut)
308
+ def patch_connection(connection_id: str, body: sch.ConnectionPatch, db: Session = Depends(get_session)):
309
+ conn = db.get(Connection, connection_id)
310
+ if not conn:
311
+ raise HTTPException(404, "Connection not found")
312
+ data = body.model_dump(exclude_unset=True)
313
+ if data.get("schedule_type") not in (None, "interval", "cron"):
314
+ raise HTTPException(400, "schedule_type must be 'interval' or 'cron'")
315
+ new_mode = data.get("mode", conn.mode)
316
+ if new_mode not in ("batch", "cdc"):
317
+ raise HTTPException(400, "mode must be 'batch' or 'cdc'")
318
+ if new_mode == "cdc" and not SOURCE_TYPES[conn.source.type].supports_cdc:
319
+ raise HTTPException(400, f"{conn.source.type} source does not support CDC")
320
+ for stream in data.get("streams", []):
321
+ if stream.get("sync_mode") == "xmin" and conn.source.type != "postgres":
322
+ raise HTTPException(400, "xmin update method is only available for PostgreSQL sources")
323
+ mode_changed = "mode" in data and data["mode"] != conn.mode
324
+ schedule_changed = "schedule_type" in data or "cron_expression" in data
325
+ for k, v in data.items():
326
+ setattr(conn, k, v)
327
+ if conn.schedule_type == "cron" and not conn.cron_expression:
328
+ raise HTTPException(400, "cron_expression is required when schedule_type is 'cron'")
329
+ if mode_changed and conn.mode == "batch":
330
+ conn.next_run_at = utcnow()
331
+ elif schedule_changed and conn.mode == "batch" and conn.schedule_type == "cron":
332
+ conn.next_run_at = _next_cron_fire(conn.cron_expression)
333
+ db.commit()
334
+ db.refresh(conn)
335
+ if mode_changed and conn.mode == "batch":
336
+ supervisor.stop(connection_id)
337
+ return conn
338
+
339
+
340
+ @router.delete("/connections/{connection_id}")
341
+ def delete_connection(connection_id: str, db: Session = Depends(get_session)):
342
+ conn = db.get(Connection, connection_id)
343
+ if not conn:
344
+ raise HTTPException(404, "Connection not found")
345
+ supervisor.stop(connection_id)
346
+ db.delete(conn)
347
+ db.commit()
348
+ return {"ok": True}
349
+
350
+
351
+ @router.post("/connections/{connection_id}/run")
352
+ def trigger_run(connection_id: str, background_tasks: BackgroundTasks, db: Session = Depends(get_session)):
353
+ conn = db.get(Connection, connection_id)
354
+ if not conn:
355
+ raise HTTPException(404, "Connection not found")
356
+ if conn.mode == "cdc":
357
+ raise HTTPException(400, "CDC connections stream continuously; use /start instead")
358
+ if conn.status == "running":
359
+ raise HTTPException(409, "A sync is already running for this connection")
360
+
361
+ background_tasks.add_task(run_batch_sync, connection_id, trigger="manual")
362
+ return {"ok": True, "status": "started"}
363
+
364
+
365
+ @router.post("/connections/{connection_id}/resync")
366
+ def resync_connection(connection_id: str, body: sch.ResyncRequest, background_tasks: BackgroundTasks,
367
+ db: Session = Depends(get_session)):
368
+ """Forces a full reload — for incremental/xmin streams this drops the
369
+ stored cursor first so the next run re-reads everything instead of just
370
+ the delta, unlike a plain /run. For CDC, restarts the worker with the
371
+ initial-snapshot flag cleared so it re-copies every current row before
372
+ resuming streaming. `stream_names` (optional) scopes it to specific
373
+ tables instead of the whole connection — CDC doesn't support this since
374
+ its snapshot is connection-wide, not per-table."""
375
+ conn = db.get(Connection, connection_id)
376
+ if not conn:
377
+ raise HTTPException(404, "Connection not found")
378
+
379
+ valid_names = {s["schema"]["name"] for s in conn.streams}
380
+ if body.stream_names:
381
+ unknown = set(body.stream_names) - valid_names
382
+ if unknown:
383
+ raise HTTPException(400, f"Unknown stream(s): {', '.join(sorted(unknown))}")
384
+
385
+ if conn.mode == "cdc":
386
+ if body.stream_names:
387
+ raise HTTPException(400, "Per-table resync isn't available for CDC connections "
388
+ "(the initial snapshot covers the whole connection) — "
389
+ "resync without stream_names to redo it for every table.")
390
+ state = dict(conn.state or {})
391
+ state.pop("snapshot_done", None)
392
+ conn.state = state
393
+ conn.enabled = True
394
+ db.commit()
395
+ supervisor.stop(connection_id)
396
+ supervisor.start(connection_id)
397
+ return {"ok": True, "mode": "cdc"}
398
+
399
+ if conn.status == "running":
400
+ raise HTTPException(409, "A sync is already running for this connection")
401
+
402
+ targets = body.stream_names or sorted(valid_names)
403
+ state = dict(conn.state or {})
404
+ cursors = dict(state.get("cursors") or {})
405
+ for name in targets:
406
+ cursors.pop(name, None)
407
+ state["cursors"] = cursors
408
+ conn.state = state
409
+ conn.enabled = True # an explicit resync click should work even while paused
410
+ db.commit()
411
+
412
+ background_tasks.add_task(run_batch_sync, connection_id, trigger="resync", only_streams=targets)
413
+ return {"ok": True, "mode": "batch", "streams": targets}
414
+
415
+
416
+ @router.post("/connections/{connection_id}/start")
417
+ def start_cdc(connection_id: str, db: Session = Depends(get_session)):
418
+ conn = db.get(Connection, connection_id)
419
+ if not conn:
420
+ raise HTTPException(404, "Connection not found")
421
+ if conn.mode != "cdc":
422
+ raise HTTPException(400, "Connection is not in CDC mode")
423
+ conn.enabled = True
424
+ db.commit()
425
+ supervisor.start(connection_id)
426
+ return {"ok": True}
427
+
428
+
429
+ @router.post("/connections/{connection_id}/stop")
430
+ def stop_cdc(connection_id: str, db: Session = Depends(get_session)):
431
+ conn = db.get(Connection, connection_id)
432
+ if not conn:
433
+ raise HTTPException(404, "Connection not found")
434
+ # Must clear `enabled`, not just kill the worker — the scheduler treats
435
+ # every enabled CDC connection as "should be running" and would restart
436
+ # this on its very next tick otherwise.
437
+ conn.enabled = False
438
+ conn.status = "idle"
439
+ db.commit()
440
+ supervisor.stop(connection_id)
441
+ return {"ok": True}
442
+
443
+
444
+ @router.get("/connections/{connection_id}/runs", response_model=list[sch.SyncRunOut])
445
+ def list_runs(connection_id: str, limit: int = 50, db: Session = Depends(get_session)):
446
+ q = (select(SyncRun).where(SyncRun.connection_id == connection_id)
447
+ .order_by(SyncRun.started_at.desc()).limit(min(limit, 200)))
448
+ return db.execute(q).scalars().all()
449
+
450
+
451
+ # ---------------------------------------------------------------------------
452
+ # Logs
453
+ # ---------------------------------------------------------------------------
454
+ @router.get("/logs", response_model=list[sch.LogOut])
455
+ def get_logs(connection_id: str | None = None, limit: int = 200, db: Session = Depends(get_session)):
456
+ q = select(LogEntry)
457
+ if connection_id:
458
+ q = q.where(LogEntry.connection_id == connection_id)
459
+ q = q.order_by(LogEntry.id.desc()).limit(min(limit, 1000))
460
+ rows = list(db.execute(q).scalars().all())
461
+ rows.reverse()
462
+ return rows
463
+
464
+
465
+ # ---------------------------------------------------------------------------
466
+ # Live events
467
+ # ---------------------------------------------------------------------------
468
+ @router.websocket("/ws")
469
+ async def ws_events(websocket: WebSocket):
470
+ await websocket.accept()
471
+ queue = subscribe()
472
+ try:
473
+ while True:
474
+ event = await queue.get()
475
+ await websocket.send_text(encode(event))
476
+ except WebSocketDisconnect:
477
+ pass
478
+ finally:
479
+ unsubscribe(queue)
@@ -0,0 +1,135 @@
1
+ """Pydantic request/response models for the HolyPipe API."""
2
+ from __future__ import annotations
3
+
4
+ import datetime as dt
5
+ from typing import Any
6
+
7
+ from pydantic import BaseModel, Field
8
+
9
+
10
+ class ResyncRequest(BaseModel):
11
+ stream_names: list[str] | None = None # None = every selected stream
12
+
13
+
14
+ class ConnectorIn(BaseModel):
15
+ name: str
16
+ type: str
17
+ config: dict = Field(default_factory=dict)
18
+ uri: str | None = None # optional connection-string alternative to `config`
19
+
20
+
21
+ class ConnectorOut(BaseModel):
22
+ id: str
23
+ name: str
24
+ type: str
25
+ config: dict
26
+ created_at: dt.datetime
27
+
28
+ model_config = {"from_attributes": True}
29
+
30
+
31
+ class TestResult(BaseModel):
32
+ ok: bool
33
+ message: str
34
+
35
+
36
+ class DiscoveredColumn(BaseModel):
37
+ name: str
38
+ type: str
39
+ nullable: bool
40
+ native_type: str
41
+
42
+
43
+ class DiscoveredStream(BaseModel):
44
+ name: str
45
+ table: str
46
+ namespace: str | None
47
+ columns: list[DiscoveredColumn]
48
+ primary_key: list[str]
49
+
50
+
51
+ class StreamConfig(BaseModel):
52
+ schema_: dict = Field(alias="schema")
53
+ selected: bool = True
54
+ sync_mode: str = "full_refresh" # full_refresh | incremental | xmin
55
+ cursor_field: str | None = None
56
+ primary_key: list[str] = Field(default_factory=list)
57
+ destination_table: str | None = None
58
+ columns: list[str] | None = None # None = all columns; otherwise an explicit subset
59
+
60
+ model_config = {"populate_by_name": True}
61
+
62
+
63
+ class ConnectionIn(BaseModel):
64
+ name: str
65
+ source_id: str
66
+ destination_id: str
67
+ mode: str = "batch" # batch | cdc
68
+ schedule_type: str = "interval" # interval | cron (batch mode only)
69
+ interval_seconds: int = 60
70
+ cron_expression: str | None = None
71
+ enabled: bool = True
72
+ destination_namespace: str | None = None
73
+ table_prefix: str = ""
74
+ streams: list[dict] = Field(default_factory=list)
75
+
76
+
77
+ class ConnectionPatch(BaseModel):
78
+ name: str | None = None
79
+ mode: str | None = None
80
+ schedule_type: str | None = None
81
+ interval_seconds: int | None = None
82
+ cron_expression: str | None = None
83
+ enabled: bool | None = None
84
+ destination_namespace: str | None = None
85
+ table_prefix: str | None = None
86
+ streams: list[dict] | None = None
87
+
88
+
89
+ class ConnectionOut(BaseModel):
90
+ id: str
91
+ name: str
92
+ source_id: str
93
+ destination_id: str
94
+ mode: str
95
+ schedule_type: str
96
+ interval_seconds: int
97
+ cron_expression: str | None
98
+ enabled: bool
99
+ destination_namespace: str | None
100
+ table_prefix: str
101
+ streams: list
102
+ status: str
103
+ status_detail: str | None
104
+ last_run_at: dt.datetime | None
105
+ next_run_at: dt.datetime | None
106
+ created_at: dt.datetime
107
+
108
+ model_config = {"from_attributes": True}
109
+
110
+
111
+ class SyncRunOut(BaseModel):
112
+ id: str
113
+ connection_id: str
114
+ trigger: str
115
+ status: str
116
+ started_at: dt.datetime
117
+ finished_at: dt.datetime | None
118
+ records_read: int
119
+ records_written: int
120
+ records_deleted: int
121
+ error: str | None
122
+ stream_stats: dict
123
+
124
+ model_config = {"from_attributes": True}
125
+
126
+
127
+ class LogOut(BaseModel):
128
+ id: int
129
+ connection_id: str | None
130
+ run_id: str | None
131
+ ts: dt.datetime
132
+ level: str
133
+ message: str
134
+
135
+ model_config = {"from_attributes": True}
holypipe/bus.py ADDED
@@ -0,0 +1,68 @@
1
+ """Thread-safe event bus bridging sync worker threads to the asyncio websocket layer."""
2
+ from __future__ import annotations
3
+
4
+ import asyncio
5
+ import datetime as dt
6
+ import json
7
+ import queue
8
+ import threading
9
+ from typing import Any
10
+
11
+ _events: "queue.Queue[dict]" = queue.Queue(maxsize=10000)
12
+ _subscribers: set[asyncio.Queue] = set()
13
+ _lock = threading.Lock()
14
+ _loop: asyncio.AbstractEventLoop | None = None
15
+
16
+
17
+ def publish(event_type: str, **payload: Any) -> None:
18
+ """Called from any thread. Never blocks the caller."""
19
+ event = {"type": event_type, "ts": dt.datetime.now(dt.timezone.utc).isoformat(), **payload}
20
+ try:
21
+ _events.put_nowait(event)
22
+ except queue.Full: # pragma: no cover - drop oldest under pressure
23
+ try:
24
+ _events.get_nowait()
25
+ _events.put_nowait(event)
26
+ except queue.Empty:
27
+ pass
28
+
29
+
30
+ def subscribe() -> asyncio.Queue:
31
+ q: asyncio.Queue = asyncio.Queue(maxsize=500)
32
+ with _lock:
33
+ _subscribers.add(q)
34
+ return q
35
+
36
+
37
+ def unsubscribe(q: asyncio.Queue) -> None:
38
+ with _lock:
39
+ _subscribers.discard(q)
40
+
41
+
42
+ def _fanout(event: dict) -> None:
43
+ with _lock:
44
+ targets = list(_subscribers)
45
+ for q in targets:
46
+ try:
47
+ q.put_nowait(event)
48
+ except asyncio.QueueFull:
49
+ pass
50
+
51
+
52
+ async def pump() -> None:
53
+ """Background asyncio task: drains the thread queue onto websocket subscribers."""
54
+ global _loop
55
+ _loop = asyncio.get_running_loop()
56
+ while True:
57
+ try:
58
+ event = await asyncio.to_thread(_events.get, True, 1.0)
59
+ except queue.Empty:
60
+ continue
61
+ except Exception: # pragma: no cover
62
+ await asyncio.sleep(0.2)
63
+ continue
64
+ _fanout(event)
65
+
66
+
67
+ def encode(event: dict) -> str:
68
+ return json.dumps(event, default=str)