tkati-dashboard 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tkati_dashboard/__init__.py +9 -0
- tkati_dashboard/__main__.py +4 -0
- tkati_dashboard/_kafka_metadata.py +31 -0
- tkati_dashboard/app.py +155 -0
- tkati_dashboard/dataflow.py +168 -0
- tkati_dashboard/flows.py +54 -0
- tkati_dashboard/lag.py +78 -0
- tkati_dashboard/main.py +76 -0
- tkati_dashboard/py.typed +0 -0
- tkati_dashboard/snapshot.py +111 -0
- tkati_dashboard/static/index.html +1410 -0
- tkati_dashboard/topic_stats.py +95 -0
- tkati_dashboard-0.4.0.dist-info/METADATA +172 -0
- tkati_dashboard-0.4.0.dist-info/RECORD +17 -0
- tkati_dashboard-0.4.0.dist-info/WHEEL +5 -0
- tkati_dashboard-0.4.0.dist-info/entry_points.txt +2 -0
- tkati_dashboard-0.4.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""Shared "resolve a live topic's metadata" helpers for snapshot.py, lag.py, and topic_stats.py."""
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from confluent_kafka import Consumer
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def resolve_topic_metadata(
|
|
9
|
+
consumer: Consumer, broker: str, topic: str, timeout_sec: float
|
|
10
|
+
) -> Any:
|
|
11
|
+
"""Return `topic`'s TopicMetadata (partitions, each with leader/replicas/isrs), or raise
|
|
12
|
+
RuntimeError with a message naming `broker`/`topic` if the broker is unreachable or the
|
|
13
|
+
topic doesn't exist.
|
|
14
|
+
"""
|
|
15
|
+
try:
|
|
16
|
+
metadata = consumer.list_topics(topic, timeout=timeout_sec)
|
|
17
|
+
except Exception as e:
|
|
18
|
+
raise RuntimeError(f"Could not reach broker {broker!r}: {e}") from e
|
|
19
|
+
|
|
20
|
+
topic_metadata = metadata.topics.get(topic)
|
|
21
|
+
if topic_metadata is None or topic_metadata.error is not None:
|
|
22
|
+
raise RuntimeError(f"Topic {topic!r} not found on {broker!r}")
|
|
23
|
+
|
|
24
|
+
return topic_metadata
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def resolve_partitions(
|
|
28
|
+
consumer: Consumer, broker: str, topic: str, timeout_sec: float
|
|
29
|
+
) -> list[int]:
|
|
30
|
+
"""Return the partition ids of `topic`. See resolve_topic_metadata for error behavior."""
|
|
31
|
+
return list(resolve_topic_metadata(consumer, broker, topic, timeout_sec).partitions)
|
tkati_dashboard/app.py
ADDED
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
from collections.abc import Callable
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from fastapi import FastAPI, HTTPException
|
|
6
|
+
from fastapi.responses import FileResponse
|
|
7
|
+
|
|
8
|
+
from tkati_dashboard import lag, snapshot, topic_stats
|
|
9
|
+
from tkati_dashboard.dataflow import (
|
|
10
|
+
SOURCE_SINK_TYPES,
|
|
11
|
+
DataflowValidationError,
|
|
12
|
+
NodeDef,
|
|
13
|
+
load_dataflow,
|
|
14
|
+
)
|
|
15
|
+
from tkati_dashboard.flows import FlowConfigError
|
|
16
|
+
|
|
17
|
+
STATIC_DIR = Path(__file__).parent / "static"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _list_flows(list_flows: Callable[[], dict[str, Path]]) -> dict[str, Path]:
|
|
21
|
+
try:
|
|
22
|
+
return list_flows()
|
|
23
|
+
except FlowConfigError as e:
|
|
24
|
+
raise HTTPException(status_code=500, detail=str(e)) from e
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _resolve_flow_dir(list_flows: Callable[[], dict[str, Path]], flow_id: str) -> Path:
|
|
28
|
+
directory = _list_flows(list_flows).get(flow_id)
|
|
29
|
+
if directory is None:
|
|
30
|
+
raise HTTPException(status_code=404, detail=f"Unknown flow {flow_id!r}")
|
|
31
|
+
return directory
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _graph_json(directory: Path) -> dict[str, Any]:
|
|
35
|
+
dataflow = load_dataflow(directory)
|
|
36
|
+
|
|
37
|
+
nodes = [
|
|
38
|
+
{
|
|
39
|
+
"id": node_id,
|
|
40
|
+
"label": node.name or node_id,
|
|
41
|
+
"type": node.type,
|
|
42
|
+
"group": "source-sink"
|
|
43
|
+
if node.type in SOURCE_SINK_TYPES
|
|
44
|
+
else "processing-node",
|
|
45
|
+
"schema": node.schema,
|
|
46
|
+
"connection": node.connection,
|
|
47
|
+
"config": node.config,
|
|
48
|
+
}
|
|
49
|
+
for node_id, node in dataflow.nodes.items()
|
|
50
|
+
]
|
|
51
|
+
edges = [
|
|
52
|
+
{
|
|
53
|
+
"from": edge.from_,
|
|
54
|
+
"to": edge.to,
|
|
55
|
+
"kind": edge.kind,
|
|
56
|
+
"consumer": edge.consumer,
|
|
57
|
+
}
|
|
58
|
+
for edge in dataflow.edges
|
|
59
|
+
]
|
|
60
|
+
return {"name": dataflow.name, "nodes": nodes, "edges": edges}
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _get_node(directory: Path, node_id: str) -> NodeDef:
|
|
64
|
+
dataflow = load_dataflow(directory)
|
|
65
|
+
node = dataflow.nodes.get(node_id)
|
|
66
|
+
if node is None:
|
|
67
|
+
raise HTTPException(status_code=404, detail=f"Unknown node {node_id!r}")
|
|
68
|
+
return node
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _require_kafka_connection(
|
|
72
|
+
node_id: str, node: NodeDef, feature: str
|
|
73
|
+
) -> tuple[str, str]:
|
|
74
|
+
"""Common gate for the two live-Kafka endpoints: node must be a kafka-topic with a
|
|
75
|
+
broker/topic to connect to. Returns (broker, topic) or raises HTTPException."""
|
|
76
|
+
if node.type != "kafka-topic":
|
|
77
|
+
raise HTTPException(
|
|
78
|
+
status_code=404,
|
|
79
|
+
detail=f"No {feature} available for node type {node.type!r}",
|
|
80
|
+
)
|
|
81
|
+
connection = node.connection or {}
|
|
82
|
+
broker, topic = connection.get("broker"), connection.get("topic")
|
|
83
|
+
if not broker or not topic:
|
|
84
|
+
raise HTTPException(
|
|
85
|
+
status_code=422,
|
|
86
|
+
detail=f"Node {node_id!r} is missing connection.broker/connection.topic",
|
|
87
|
+
)
|
|
88
|
+
return broker, topic
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def create_app(list_flows: Callable[[], dict[str, Path]]) -> FastAPI:
|
|
92
|
+
app = FastAPI(title="tkati-dashboard")
|
|
93
|
+
|
|
94
|
+
@app.get("/")
|
|
95
|
+
def index() -> FileResponse:
|
|
96
|
+
return FileResponse(STATIC_DIR / "index.html")
|
|
97
|
+
|
|
98
|
+
@app.get("/api/flows")
|
|
99
|
+
def flows() -> list[dict[str, str]]:
|
|
100
|
+
resolved = _list_flows(list_flows)
|
|
101
|
+
return [{"id": flow_id, "name": flow_id} for flow_id in sorted(resolved, key=str.lower)]
|
|
102
|
+
|
|
103
|
+
@app.get("/api/flows/{flow_id}/graph")
|
|
104
|
+
def graph(flow_id: str) -> dict[str, Any]:
|
|
105
|
+
directory = _resolve_flow_dir(list_flows, flow_id)
|
|
106
|
+
try:
|
|
107
|
+
return _graph_json(directory)
|
|
108
|
+
except DataflowValidationError as e:
|
|
109
|
+
raise HTTPException(status_code=422, detail=str(e)) from e
|
|
110
|
+
|
|
111
|
+
@app.get("/api/flows/{flow_id}/nodes/{node_id}/snapshot")
|
|
112
|
+
def node_snapshot(flow_id: str, node_id: str) -> dict[str, Any]:
|
|
113
|
+
directory = _resolve_flow_dir(list_flows, flow_id)
|
|
114
|
+
try:
|
|
115
|
+
node = _get_node(directory, node_id)
|
|
116
|
+
except DataflowValidationError as e:
|
|
117
|
+
raise HTTPException(status_code=422, detail=str(e)) from e
|
|
118
|
+
broker, topic = _require_kafka_connection(node_id, node, "live snapshot")
|
|
119
|
+
|
|
120
|
+
try:
|
|
121
|
+
events = snapshot.fetch_kafka_snapshot(broker, topic)
|
|
122
|
+
except snapshot.SnapshotError as e:
|
|
123
|
+
raise HTTPException(status_code=502, detail=str(e)) from e
|
|
124
|
+
|
|
125
|
+
return {"events": events}
|
|
126
|
+
|
|
127
|
+
@app.get("/api/flows/{flow_id}/nodes/{node_id}/consumer-lag")
|
|
128
|
+
def node_consumer_lag(flow_id: str, node_id: str, group_id: str) -> dict[str, Any]:
|
|
129
|
+
directory = _resolve_flow_dir(list_flows, flow_id)
|
|
130
|
+
try:
|
|
131
|
+
node = _get_node(directory, node_id)
|
|
132
|
+
except DataflowValidationError as e:
|
|
133
|
+
raise HTTPException(status_code=422, detail=str(e)) from e
|
|
134
|
+
broker, topic = _require_kafka_connection(node_id, node, "consumer lag")
|
|
135
|
+
|
|
136
|
+
try:
|
|
137
|
+
return lag.fetch_consumer_lag(broker, topic, group_id)
|
|
138
|
+
except lag.LagError as e:
|
|
139
|
+
raise HTTPException(status_code=502, detail=str(e)) from e
|
|
140
|
+
|
|
141
|
+
@app.get("/api/flows/{flow_id}/nodes/{node_id}/topic-stats")
|
|
142
|
+
def node_topic_stats(flow_id: str, node_id: str) -> dict[str, Any]:
|
|
143
|
+
directory = _resolve_flow_dir(list_flows, flow_id)
|
|
144
|
+
try:
|
|
145
|
+
node = _get_node(directory, node_id)
|
|
146
|
+
except DataflowValidationError as e:
|
|
147
|
+
raise HTTPException(status_code=422, detail=str(e)) from e
|
|
148
|
+
broker, topic = _require_kafka_connection(node_id, node, "topic stats")
|
|
149
|
+
|
|
150
|
+
try:
|
|
151
|
+
return topic_stats.fetch_topic_stats(broker, topic)
|
|
152
|
+
except topic_stats.TopicStatsError as e:
|
|
153
|
+
raise HTTPException(status_code=502, detail=str(e)) from e
|
|
154
|
+
|
|
155
|
+
return app
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""Load and validate a serialized tkati dataflow directory.
|
|
2
|
+
|
|
3
|
+
See docs/dataflow-serialization.md for the format this module implements: a directory of JSON or
|
|
4
|
+
YAML fragments merged into one graph of nodes and edges. There is no manifest file — every
|
|
5
|
+
`*.json`/`*.yaml`/`*.yml` file directly inside the directory is a fragment.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
import yaml
|
|
13
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
14
|
+
from tkati_core.type_mapping import TYPE_MAPPING
|
|
15
|
+
|
|
16
|
+
# Node types that represent data at rest (as opposed to a processing step) and therefore require
|
|
17
|
+
# a `schema`. Kept as a heuristic, not a closed registry: an unrecognized type is still accepted,
|
|
18
|
+
# it just isn't schema-checked.
|
|
19
|
+
SOURCE_SINK_TYPES = {"kafka-topic", "clickhouse-table"}
|
|
20
|
+
|
|
21
|
+
FRAGMENT_GLOBS = ("*.json", "*.yaml", "*.yml")
|
|
22
|
+
YAML_SUFFIXES = (".yaml", ".yml")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class DataflowValidationError(ValueError):
|
|
26
|
+
"""A serialized dataflow directory failed validation."""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class NodeDef(BaseModel):
|
|
30
|
+
model_config = ConfigDict(extra="allow")
|
|
31
|
+
|
|
32
|
+
type: str
|
|
33
|
+
name: str | None = None
|
|
34
|
+
schema: dict[str, str] | None = None
|
|
35
|
+
connection: dict[str, Any] | None = None
|
|
36
|
+
config: dict[str, Any] | None = None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class EdgeDef(BaseModel):
|
|
40
|
+
model_config = ConfigDict(populate_by_name=True)
|
|
41
|
+
|
|
42
|
+
from_: str = Field(alias="from")
|
|
43
|
+
to: str
|
|
44
|
+
kind: str = "stream"
|
|
45
|
+
consumer: dict[str, Any] | None = None
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class Dataflow(BaseModel):
|
|
49
|
+
name: str
|
|
50
|
+
nodes: dict[str, NodeDef]
|
|
51
|
+
edges: list[EdgeDef]
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def find_fragment_paths(directory: Path) -> list[Path]:
|
|
55
|
+
"""Return every `*.json`/`*.yaml`/`*.yml` file directly inside `directory`, sorted by name
|
|
56
|
+
(mixing extensions in one alphabetical list, so merge order is purely filename-driven)."""
|
|
57
|
+
return sorted(p for pattern in FRAGMENT_GLOBS for p in directory.glob(pattern))
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _read_fragment(path: Path) -> dict[str, Any]:
|
|
61
|
+
try:
|
|
62
|
+
text = path.read_text()
|
|
63
|
+
except FileNotFoundError as e:
|
|
64
|
+
raise DataflowValidationError(f"Missing dataflow file: {path}") from e
|
|
65
|
+
|
|
66
|
+
if path.suffix in YAML_SUFFIXES:
|
|
67
|
+
try:
|
|
68
|
+
data = yaml.safe_load(text)
|
|
69
|
+
except yaml.YAMLError as e:
|
|
70
|
+
raise DataflowValidationError(f"Invalid YAML in {path}: {e}") from e
|
|
71
|
+
else:
|
|
72
|
+
try:
|
|
73
|
+
data = json.loads(text)
|
|
74
|
+
except json.JSONDecodeError as e:
|
|
75
|
+
raise DataflowValidationError(f"Invalid JSON in {path}: {e}") from e
|
|
76
|
+
|
|
77
|
+
if data is None:
|
|
78
|
+
return {}
|
|
79
|
+
if not isinstance(data, dict):
|
|
80
|
+
raise DataflowValidationError(
|
|
81
|
+
f"{path}: fragment must be a JSON/YAML object, got {type(data).__name__}"
|
|
82
|
+
)
|
|
83
|
+
return data
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _validate_node(node_id: str, node: NodeDef) -> None:
|
|
87
|
+
# `schema` is optional even for source/sink nodes: a real-world fragment (e.g. one written
|
|
88
|
+
# by hand, or discovered from a live cluster without introspecting its columns) may not
|
|
89
|
+
# have one on hand. When it is present, its field types are still checked.
|
|
90
|
+
if node.schema is not None:
|
|
91
|
+
for field_name, field_type in node.schema.items():
|
|
92
|
+
if field_type not in TYPE_MAPPING:
|
|
93
|
+
raise DataflowValidationError(
|
|
94
|
+
f"Node {node_id!r} field {field_name!r} has unknown schema type {field_type!r}"
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _merge_node(
|
|
99
|
+
nodes: dict[str, NodeDef],
|
|
100
|
+
node_sources: dict[str, str],
|
|
101
|
+
fragment_name: str,
|
|
102
|
+
node_id: str,
|
|
103
|
+
raw_node: dict[str, Any],
|
|
104
|
+
) -> None:
|
|
105
|
+
node = NodeDef.model_validate(raw_node)
|
|
106
|
+
if node_id in nodes:
|
|
107
|
+
if nodes[node_id] != node:
|
|
108
|
+
raise DataflowValidationError(
|
|
109
|
+
f"Node {node_id!r} is defined differently in "
|
|
110
|
+
f"{node_sources[node_id]!r} and {fragment_name!r}"
|
|
111
|
+
)
|
|
112
|
+
return
|
|
113
|
+
nodes[node_id] = node
|
|
114
|
+
node_sources[node_id] = fragment_name
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def load_dataflow(directory: Path) -> Dataflow:
|
|
118
|
+
"""Read every JSON/YAML fragment directly inside `directory`, merge, and validate them.
|
|
119
|
+
|
|
120
|
+
There is no manifest: any `*.json`/`*.yaml`/`*.yml` file in the directory is a fragment
|
|
121
|
+
contributing to the graph. The dataflow's name is the directory's own name.
|
|
122
|
+
"""
|
|
123
|
+
if not directory.is_dir():
|
|
124
|
+
raise DataflowValidationError(f"Not a directory: {directory}")
|
|
125
|
+
|
|
126
|
+
fragment_paths = find_fragment_paths(directory)
|
|
127
|
+
if not fragment_paths:
|
|
128
|
+
raise DataflowValidationError(
|
|
129
|
+
f"No dataflow fragments ({'/'.join(FRAGMENT_GLOBS)}) found in {directory}"
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
nodes: dict[str, NodeDef] = {}
|
|
133
|
+
node_sources: dict[
|
|
134
|
+
str, str
|
|
135
|
+
] = {} # node id -> fragment it was first seen in, for error messages
|
|
136
|
+
edges: list[EdgeDef] = []
|
|
137
|
+
|
|
138
|
+
for fragment_path in fragment_paths:
|
|
139
|
+
fragment = _read_fragment(fragment_path)
|
|
140
|
+
|
|
141
|
+
for node_id, raw_node in fragment.get("nodes", {}).items():
|
|
142
|
+
_merge_node(nodes, node_sources, fragment_path.name, node_id, raw_node)
|
|
143
|
+
|
|
144
|
+
# A fragment may also declare a single node via a top-level "node" object carrying
|
|
145
|
+
# its own "id", instead of keying it under "nodes" — e.g. one file per node.
|
|
146
|
+
if "node" in fragment:
|
|
147
|
+
raw_node = fragment["node"]
|
|
148
|
+
node_id = raw_node.get("id")
|
|
149
|
+
if not node_id:
|
|
150
|
+
raise DataflowValidationError(
|
|
151
|
+
f"{fragment_path.name!r} has a 'node' object with no 'id'"
|
|
152
|
+
)
|
|
153
|
+
_merge_node(nodes, node_sources, fragment_path.name, node_id, raw_node)
|
|
154
|
+
|
|
155
|
+
for raw_edge in fragment.get("edges", []):
|
|
156
|
+
edges.append(EdgeDef.model_validate(raw_edge))
|
|
157
|
+
|
|
158
|
+
for node_id, node in nodes.items():
|
|
159
|
+
_validate_node(node_id, node)
|
|
160
|
+
|
|
161
|
+
for edge in edges:
|
|
162
|
+
for endpoint in (edge.from_, edge.to):
|
|
163
|
+
if endpoint not in nodes:
|
|
164
|
+
raise DataflowValidationError(
|
|
165
|
+
f"Edge {edge.from_!r} -> {edge.to!r} references unknown node {endpoint!r}"
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
return Dataflow(name=directory.name, nodes=nodes, edges=edges)
|
tkati_dashboard/flows.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Resolve the set of dataflow directories a dashboard instance observes.
|
|
2
|
+
|
|
3
|
+
A dashboard instance can watch more than one dataflow at once: some directories are named
|
|
4
|
+
explicitly on the command line, others are discovered as immediate subdirectories of a
|
|
5
|
+
`--flows-root`. Either way, a flow's id is simply its directory's own basename — the same
|
|
6
|
+
identity `dataflow.load_dataflow` already gives a `Dataflow.name`.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from collections.abc import Callable
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from tkati_dashboard.dataflow import find_fragment_paths
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class FlowConfigError(ValueError):
|
|
16
|
+
"""The configured dataflow directories/roots don't resolve to an unambiguous flow set."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def discover_flows_root(root: Path) -> dict[str, Path]:
|
|
20
|
+
"""Every immediate subdirectory of `root` that contains at least one dataflow fragment is a
|
|
21
|
+
flow, keyed by its own basename. A subdirectory with no fragments (e.g. a stray `README` or
|
|
22
|
+
`.git`) is silently skipped rather than treated as an error."""
|
|
23
|
+
return {
|
|
24
|
+
path.name: path
|
|
25
|
+
for path in sorted(root.iterdir())
|
|
26
|
+
if path.is_dir() and find_fragment_paths(path)
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def make_flow_lister(
|
|
31
|
+
dataflow_dirs: list[Path], flows_roots: list[Path]
|
|
32
|
+
) -> Callable[[], dict[str, Path]]:
|
|
33
|
+
"""Build a `list_flows()` callable returning the current flow id -> directory map.
|
|
34
|
+
|
|
35
|
+
Explicit `dataflow_dirs` are fixed once, at build time. Each `flows_roots` entry is
|
|
36
|
+
re-scanned on every call, so adding or removing a flow subdirectory there is picked up
|
|
37
|
+
without restarting the server — the same "always read fresh, never cached" philosophy
|
|
38
|
+
`load_dataflow` already applies to a single flow's fragments.
|
|
39
|
+
"""
|
|
40
|
+
fixed = {directory.name: directory for directory in dataflow_dirs}
|
|
41
|
+
|
|
42
|
+
def list_flows() -> dict[str, Path]:
|
|
43
|
+
flows = dict(fixed)
|
|
44
|
+
for root in flows_roots:
|
|
45
|
+
for flow_id, path in discover_flows_root(root).items():
|
|
46
|
+
if flow_id in flows and flows[flow_id] != path:
|
|
47
|
+
raise FlowConfigError(
|
|
48
|
+
f"Flow id {flow_id!r} is ambiguous: {flows[flow_id]} and {path} both "
|
|
49
|
+
"resolve to it — rename one of the directories"
|
|
50
|
+
)
|
|
51
|
+
flows[flow_id] = path
|
|
52
|
+
return flows
|
|
53
|
+
|
|
54
|
+
return list_flows
|
tkati_dashboard/lag.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Live consumer-group lag lookup for a dataflow edge, for the dashboard's edge labels.
|
|
2
|
+
|
|
3
|
+
Like tkati_dashboard.snapshot, this is a best-effort, on-demand look at a live broker — it never
|
|
4
|
+
subscribes or polls as the consumer group, only reads its committed offsets via `.committed()`,
|
|
5
|
+
so it can never join the group, trigger a rebalance, or otherwise disrupt a real pipeline.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from confluent_kafka import Consumer, TopicPartition
|
|
11
|
+
|
|
12
|
+
from tkati_dashboard._kafka_metadata import resolve_partitions
|
|
13
|
+
|
|
14
|
+
DEFAULT_TIMEOUT_SEC = 5.0
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class LagError(RuntimeError):
|
|
18
|
+
"""Consumer lag could not be computed."""
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def fetch_consumer_lag(
|
|
22
|
+
broker: str,
|
|
23
|
+
topic: str,
|
|
24
|
+
group_id: str,
|
|
25
|
+
timeout_sec: float = DEFAULT_TIMEOUT_SEC,
|
|
26
|
+
) -> dict[str, Any]:
|
|
27
|
+
"""Return per-partition and total lag of `group_id` on `topic`.
|
|
28
|
+
|
|
29
|
+
A partition `group_id` has never committed an offset on counts as fully behind (lag =
|
|
30
|
+
partition size), since that's the backlog the group would have to process from scratch.
|
|
31
|
+
"""
|
|
32
|
+
consumer = Consumer(
|
|
33
|
+
{
|
|
34
|
+
"bootstrap.servers": broker,
|
|
35
|
+
"group.id": group_id,
|
|
36
|
+
"enable.auto.commit": False,
|
|
37
|
+
}
|
|
38
|
+
)
|
|
39
|
+
try:
|
|
40
|
+
try:
|
|
41
|
+
partition_ids = resolve_partitions(consumer, broker, topic, timeout_sec)
|
|
42
|
+
except RuntimeError as e:
|
|
43
|
+
raise LagError(str(e)) from e
|
|
44
|
+
|
|
45
|
+
if not partition_ids:
|
|
46
|
+
return {"total_lag": 0, "partitions": []}
|
|
47
|
+
|
|
48
|
+
try:
|
|
49
|
+
committed = consumer.committed(
|
|
50
|
+
[TopicPartition(topic, pid) for pid in partition_ids],
|
|
51
|
+
timeout=timeout_sec,
|
|
52
|
+
)
|
|
53
|
+
except Exception as e:
|
|
54
|
+
raise LagError(
|
|
55
|
+
f"Could not fetch committed offsets for group {group_id!r}: {e}"
|
|
56
|
+
) from e
|
|
57
|
+
|
|
58
|
+
partitions = []
|
|
59
|
+
total_lag = 0
|
|
60
|
+
for tp in committed:
|
|
61
|
+
low, high = consumer.get_watermark_offsets(
|
|
62
|
+
TopicPartition(topic, tp.partition), timeout=timeout_sec, cached=False
|
|
63
|
+
)
|
|
64
|
+
has_committed = tp.offset is not None and tp.offset >= 0
|
|
65
|
+
current = tp.offset if has_committed else low
|
|
66
|
+
partition_lag = max(high - current, 0)
|
|
67
|
+
total_lag += partition_lag
|
|
68
|
+
partitions.append(
|
|
69
|
+
{
|
|
70
|
+
"partition": tp.partition,
|
|
71
|
+
"committed_offset": tp.offset if has_committed else None,
|
|
72
|
+
"high_watermark": high,
|
|
73
|
+
"lag": partition_lag,
|
|
74
|
+
}
|
|
75
|
+
)
|
|
76
|
+
return {"total_lag": total_lag, "partitions": partitions}
|
|
77
|
+
finally:
|
|
78
|
+
consumer.close()
|
tkati_dashboard/main.py
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
import uvicorn
|
|
5
|
+
|
|
6
|
+
from tkati_dashboard.app import create_app
|
|
7
|
+
from tkati_dashboard.dataflow import FRAGMENT_GLOBS, find_fragment_paths
|
|
8
|
+
from tkati_dashboard.flows import FlowConfigError, make_flow_lister
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def main() -> None:
|
|
12
|
+
parser = argparse.ArgumentParser(
|
|
13
|
+
prog="tkati-dashboard",
|
|
14
|
+
description="Serve a graph view of one or more serialized tkati dataflow directories",
|
|
15
|
+
)
|
|
16
|
+
parser.add_argument(
|
|
17
|
+
"dataflow_dir",
|
|
18
|
+
type=Path,
|
|
19
|
+
nargs="*",
|
|
20
|
+
default=[],
|
|
21
|
+
help=(
|
|
22
|
+
"Path to a dataflow directory (a directory of *.json/*.yaml/*.yml fragments). Pass "
|
|
23
|
+
"more than one to observe multiple flows from one dashboard; see --flows-root to "
|
|
24
|
+
"auto-discover a whole directory of them instead of naming each one"
|
|
25
|
+
),
|
|
26
|
+
)
|
|
27
|
+
parser.add_argument(
|
|
28
|
+
"--flows-root",
|
|
29
|
+
type=Path,
|
|
30
|
+
action="append",
|
|
31
|
+
default=[],
|
|
32
|
+
metavar="DIR",
|
|
33
|
+
help=(
|
|
34
|
+
"Treat every immediate subdirectory of DIR that contains dataflow fragments as its "
|
|
35
|
+
"own flow (id = subdirectory name). Re-scanned on every request, so adding or "
|
|
36
|
+
"removing a flow directory there is picked up without restarting. Repeatable."
|
|
37
|
+
),
|
|
38
|
+
)
|
|
39
|
+
parser.add_argument("--host", default="127.0.0.1")
|
|
40
|
+
parser.add_argument("--port", type=int, default=8000)
|
|
41
|
+
args = parser.parse_args()
|
|
42
|
+
|
|
43
|
+
if not args.dataflow_dir and not args.flows_root:
|
|
44
|
+
parser.error("pass at least one dataflow_dir or --flows-root")
|
|
45
|
+
|
|
46
|
+
for directory in args.dataflow_dir:
|
|
47
|
+
if not directory.is_dir():
|
|
48
|
+
parser.error(f"{directory} is not a directory")
|
|
49
|
+
if not find_fragment_paths(directory):
|
|
50
|
+
parser.error(
|
|
51
|
+
f"{directory} contains no dataflow fragments ({'/'.join(FRAGMENT_GLOBS)})"
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
seen: dict[str, Path] = {}
|
|
55
|
+
for directory in args.dataflow_dir:
|
|
56
|
+
flow_id = directory.name
|
|
57
|
+
if flow_id in seen and seen[flow_id] != directory:
|
|
58
|
+
parser.error(
|
|
59
|
+
f"flow id {flow_id!r} is ambiguous: {seen[flow_id]} and {directory} both have "
|
|
60
|
+
"this directory name — rename one of them"
|
|
61
|
+
)
|
|
62
|
+
seen[flow_id] = directory
|
|
63
|
+
|
|
64
|
+
for root in args.flows_root:
|
|
65
|
+
if not root.is_dir():
|
|
66
|
+
parser.error(f"{root} is not a directory")
|
|
67
|
+
|
|
68
|
+
list_flows = make_flow_lister(args.dataflow_dir, args.flows_root)
|
|
69
|
+
try:
|
|
70
|
+
if not list_flows():
|
|
71
|
+
parser.error("no flows found: check the given dataflow_dir(s)/--flows-root")
|
|
72
|
+
except FlowConfigError as e:
|
|
73
|
+
parser.error(str(e))
|
|
74
|
+
|
|
75
|
+
app = create_app(list_flows)
|
|
76
|
+
uvicorn.run(app, host=args.host, port=args.port)
|
tkati_dashboard/py.typed
ADDED
|
File without changes
|