lfx-confluent 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,29 @@
1
+ """lfx-confluent: IBM Confluent bundle (data-in-motion side of the IBM streamhouse).
2
+
3
+ This package is the distribution unit ``lfx-confluent``. At runtime
4
+ Langflow's loader discovers ``extension.json`` shipped alongside this
5
+ ``__init__.py`` and registers the components under the namespaced IDs:
6
+
7
+ * ``ext:confluent:ConfluentContextEngineComponent@official``
8
+ * ``ext:confluent:ConfluentKafkaConsumerComponent@official``
9
+ * ``ext:confluent:ConfluentKafkaProducerComponent@official``
10
+ * ``ext:confluent:ConfluentTableflowReaderComponent@official``
11
+
12
+ Everything talks to Confluent over open protocols only -- the Kafka wire
13
+ protocol (``confluent-kafka``), the Tableflow Iceberg REST catalog
14
+ (``pyiceberg``), and MCP over Streamable HTTP (Langflow's own MCP client
15
+ engine) -- so the same components work against Confluent Cloud, Confluent
16
+ Platform, and WarpStream wherever the corresponding surface is exposed.
17
+ """
18
+
19
+ from lfx_confluent.components.confluent.context_engine import ConfluentContextEngineComponent
20
+ from lfx_confluent.components.confluent.kafka_consumer import ConfluentKafkaConsumerComponent
21
+ from lfx_confluent.components.confluent.kafka_producer import ConfluentKafkaProducerComponent
22
+ from lfx_confluent.components.confluent.tableflow_reader import ConfluentTableflowReaderComponent
23
+
24
+ __all__ = [
25
+ "ConfluentContextEngineComponent",
26
+ "ConfluentKafkaConsumerComponent",
27
+ "ConfluentKafkaProducerComponent",
28
+ "ConfluentTableflowReaderComponent",
29
+ ]
@@ -0,0 +1,11 @@
1
+ from .context_engine import ConfluentContextEngineComponent
2
+ from .kafka_consumer import ConfluentKafkaConsumerComponent
3
+ from .kafka_producer import ConfluentKafkaProducerComponent
4
+ from .tableflow_reader import ConfluentTableflowReaderComponent
5
+
6
+ __all__ = [
7
+ "ConfluentContextEngineComponent",
8
+ "ConfluentKafkaConsumerComponent",
9
+ "ConfluentKafkaProducerComponent",
10
+ "ConfluentTableflowReaderComponent",
11
+ ]
@@ -0,0 +1,206 @@
1
+ """Shared helpers for the IBM Confluent bundle.
2
+
3
+ Small, dependency-free utilities used by every component in the bundle:
4
+ endpoint templating for the Confluent Cloud regional services, Basic-auth
5
+ header construction, Kafka client configuration for Confluent Cloud
6
+ (SASL_SSL / PLAIN with an API key + secret), and SSRF gating of every
7
+ tenant-supplied host through Langflow's connector policy.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import base64
13
+ import re
14
+
15
+ from lfx.utils.ssrf_protection import validate_connector_url_for_ssrf
16
+
17
+ DEFAULT_CLOUD = "aws"
18
+ DEFAULT_REGION = "us-east-1"
19
+ KAFKA_DEFAULT_PORT = 9092
20
+ MAX_PORT = 65535
21
+
22
+ # Confluent resource IDs (org / env / cluster / compute pool) and cloud
23
+ # region names are simple tokens. Restricting them keeps user input from
24
+ # escaping the URL path segment it is interpolated into.
25
+ _TOKEN_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]*$")
26
+
27
+
28
+ def require_token(value: str | None, label: str) -> str:
29
+ """Return ``value`` stripped, or raise ``ValueError`` if it is empty or not a plain token."""
30
+ text = (value or "").strip()
31
+ if not text:
32
+ msg = f"{label} is required."
33
+ raise ValueError(msg)
34
+ if not _TOKEN_RE.match(text):
35
+ msg = f"{label} contains unsupported characters: {text!r}"
36
+ raise ValueError(msg)
37
+ return text
38
+
39
+
40
+ def basic_auth_header(api_key: str, api_secret: str) -> str:
41
+ """Build the ``Authorization: Basic ...`` value Confluent Cloud expects for API-key auth."""
42
+ key = (api_key or "").strip()
43
+ secret = (api_secret or "").strip()
44
+ if not key or not secret:
45
+ msg = "Both the API key and the API secret are required."
46
+ raise ValueError(msg)
47
+ token = base64.b64encode(f"{key}:{secret}".encode()).decode("ascii")
48
+ return f"Basic {token}"
49
+
50
+
51
+ def ensure_url_allowed(url: str) -> str:
52
+ """SSRF-validate a tenant-supplied http(s) URL and return it stripped."""
53
+ text = (url or "").strip()
54
+ if not text:
55
+ msg = "An endpoint URL is required."
56
+ raise ValueError(msg)
57
+ validate_connector_url_for_ssrf(text)
58
+ return text
59
+
60
+
61
+ def context_engine_url(
62
+ region: str,
63
+ organization_id: str,
64
+ environment_id: str,
65
+ kafka_cluster_id: str,
66
+ cloud: str = DEFAULT_CLOUD,
67
+ ) -> str:
68
+ """Return the Real-Time Context Engine MCP endpoint for a Kafka cluster.
69
+
70
+ Pattern documented by Confluent::
71
+
72
+ https://mcp.<REGION>.<CLOUD>.confluent.cloud/mcp/v1/context-engine/
73
+ organizations/<ORG>/environments/<ENV>/kafka-clusters/<LKC>
74
+ """
75
+ region_ = require_token(region, "Region")
76
+ cloud_ = require_token(cloud, "Cloud")
77
+ org = require_token(organization_id, "Organization ID")
78
+ env = require_token(environment_id, "Environment ID")
79
+ lkc = require_token(kafka_cluster_id, "Kafka cluster ID")
80
+ return (
81
+ f"https://mcp.{region_}.{cloud_}.confluent.cloud/mcp/v1/context-engine/"
82
+ f"organizations/{org}/environments/{env}/kafka-clusters/{lkc}"
83
+ )
84
+
85
+
86
+ def tableflow_catalog_url(
87
+ region: str,
88
+ organization_id: str,
89
+ environment_id: str,
90
+ cloud: str = DEFAULT_CLOUD,
91
+ ) -> str:
92
+ """Return the Tableflow Iceberg REST catalog endpoint for an environment.
93
+
94
+ Pattern documented by Confluent::
95
+
96
+ https://tableflow.<REGION>.<CLOUD>.confluent.cloud/iceberg/catalog/
97
+ organizations/<ORG>/environments/<ENV>
98
+ """
99
+ region_ = require_token(region, "Region")
100
+ cloud_ = require_token(cloud, "Cloud")
101
+ org = require_token(organization_id, "Organization ID")
102
+ env = require_token(environment_id, "Environment ID")
103
+ return (
104
+ f"https://tableflow.{region_}.{cloud_}.confluent.cloud/iceberg/catalog/organizations/{org}/environments/{env}"
105
+ )
106
+
107
+
108
+ # ``extra`` client settings are tenant-supplied (and reachable from Tool Mode), so they
109
+ # must not be able to re-point the client at another broker after the bootstrap list has
110
+ # been SSRF-gated, nor downgrade the transport or swap the SASL credentials.
111
+ PROTECTED_KAFKA_CONFIG_KEYS = frozenset(
112
+ {
113
+ "bootstrap.servers",
114
+ "metadata.broker.list",
115
+ "security.protocol",
116
+ }
117
+ )
118
+ _PROTECTED_KAFKA_CONFIG_PREFIXES = ("sasl.", "ssl.")
119
+
120
+
121
+ def _is_protected_kafka_key(key: str) -> bool:
122
+ """Return True for connection / transport / credential settings ``extra`` may not set."""
123
+ name = (key or "").strip().lower()
124
+ return name in PROTECTED_KAFKA_CONFIG_KEYS or name.startswith(_PROTECTED_KAFKA_CONFIG_PREFIXES)
125
+
126
+
127
+ def parse_bootstrap_servers(bootstrap_servers: str) -> list[tuple[str, int]]:
128
+ """Split ``host:port,host:port`` into validated ``(host, port)`` pairs."""
129
+ text = (bootstrap_servers or "").strip()
130
+ if not text:
131
+ msg = "Bootstrap servers are required (for example 'pkc-xxxxx.us-east-1.aws.confluent.cloud:9092')."
132
+ raise ValueError(msg)
133
+ pairs: list[tuple[str, int]] = []
134
+ for raw in text.split(","):
135
+ item = raw.strip()
136
+ if not item:
137
+ continue
138
+ host, sep, port_text = item.rpartition(":")
139
+ if not sep or not host:
140
+ host, port = item, KAFKA_DEFAULT_PORT
141
+ else:
142
+ try:
143
+ port = int(port_text)
144
+ except ValueError as exc:
145
+ msg = f"Invalid bootstrap server port in {item!r}."
146
+ raise ValueError(msg) from exc
147
+ if not (0 < port <= MAX_PORT):
148
+ msg = f"Invalid bootstrap server port in {item!r}."
149
+ raise ValueError(msg)
150
+ pairs.append((host, port))
151
+ if not pairs:
152
+ msg = "Bootstrap servers are required."
153
+ raise ValueError(msg)
154
+ return pairs
155
+
156
+
157
+ def validate_bootstrap_servers(bootstrap_servers: str) -> str:
158
+ """SSRF-gate every bootstrap host and return the normalized ``host:port`` list.
159
+
160
+ Kafka bootstrap strings have no URL scheme, so each host is validated as
161
+ an ``https://host:port`` URL -- the connector policy only looks at the host,
162
+ which is what matters (cloud-metadata endpoints, RFC1918 literals, ...).
163
+ """
164
+ pairs = parse_bootstrap_servers(bootstrap_servers)
165
+ for host, port in pairs:
166
+ validate_connector_url_for_ssrf(f"https://{host}:{port}")
167
+ return ",".join(f"{host}:{port}" for host, port in pairs)
168
+
169
+
170
+ def kafka_client_config(
171
+ bootstrap_servers: str,
172
+ api_key: str,
173
+ api_secret: str,
174
+ extra: dict | None = None,
175
+ ) -> dict:
176
+ """Return a ``confluent_kafka`` client config for Confluent Cloud (SASL_SSL / PLAIN).
177
+
178
+ ``bootstrap_servers`` must already have passed :func:`validate_bootstrap_servers`.
179
+ ``extra`` may tune the client (batching, timeouts, ...) but never the connection,
180
+ transport or credentials -- see :data:`PROTECTED_KAFKA_CONFIG_KEYS`.
181
+ """
182
+ key = (api_key or "").strip()
183
+ secret = (api_secret or "").strip()
184
+ config: dict = {"bootstrap.servers": bootstrap_servers}
185
+ if key or secret:
186
+ if not key or not secret:
187
+ msg = "Both the Kafka API key and the API secret are required for SASL authentication."
188
+ raise ValueError(msg)
189
+ config.update(
190
+ {
191
+ "security.protocol": "SASL_SSL",
192
+ "sasl.mechanisms": "PLAIN",
193
+ "sasl.username": key,
194
+ "sasl.password": secret,
195
+ }
196
+ )
197
+ if extra:
198
+ protected = sorted(k for k in extra if _is_protected_kafka_key(k))
199
+ if protected:
200
+ msg = (
201
+ "Extra Client Config cannot override the connection, transport or credential settings: "
202
+ f"{', '.join(protected)}."
203
+ )
204
+ raise ValueError(msg)
205
+ config.update({k: v for k, v in extra.items() if k and v is not None})
206
+ return config
@@ -0,0 +1,127 @@
1
+ """Confluent Real-Time Context Engine as an Agent toolset."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from lfx.base.mcp.preset import MCPPresetComponent, preset_control_inputs
8
+ from lfx.io import SecretStrInput, StrInput
9
+ from lfx_confluent.components.confluent._common import (
10
+ DEFAULT_CLOUD,
11
+ DEFAULT_REGION,
12
+ basic_auth_header,
13
+ context_engine_url,
14
+ ensure_url_allowed,
15
+ )
16
+
17
+ # Tool names documented by Confluent for the Real-Time Context Engine MCP server.
18
+ CONTEXT_ENGINE_TOOLS = ["list_topics", "get_metadata", "query_data"]
19
+
20
+
21
+ class ConfluentContextEngineComponent(MCPPresetComponent):
22
+ """Serve fresh, governed Kafka topic data to an Agent through the Real-Time Context Engine.
23
+
24
+ Confluent's Real-Time Context Engine materializes schema'd Kafka topics into
25
+ a low-latency serving layer and exposes it as an MCP server with three
26
+ tools -- ``list_topics``, ``get_metadata`` and ``query_data`` (key lookups,
27
+ filters, ranges and compound predicates). This component templates the
28
+ regional endpoint from your Confluent Cloud IDs, authenticates with a
29
+ Confluent Cloud API key, and hands the tools to an Agent.
30
+ """
31
+
32
+ display_name = "Confluent Real-Time Context Engine"
33
+ description = (
34
+ "Give an Agent live, governed context from Kafka topics through Confluent's "
35
+ "Real-Time Context Engine (MCP): list topics, inspect schemas, and query the latest data."
36
+ )
37
+ documentation: str = "https://docs.langflow.org/bundles-confluent"
38
+ icon = "Confluent"
39
+ name = "ConfluentContextEngine"
40
+ metadata = {"keywords": ["confluent", "kafka", "context engine", "mcp", "real-time", "streaming", "ibm"]}
41
+
42
+ inputs = [
43
+ StrInput(
44
+ name="region",
45
+ display_name="Cloud Region",
46
+ info="Confluent Cloud region of the Kafka cluster (for example us-east-1). AWS regions only today.",
47
+ value=DEFAULT_REGION,
48
+ required=True,
49
+ ),
50
+ StrInput(
51
+ name="organization_id",
52
+ display_name="Organization ID",
53
+ info="Confluent Cloud organization ID (from Organization settings).",
54
+ required=True,
55
+ ),
56
+ StrInput(
57
+ name="environment_id",
58
+ display_name="Environment ID",
59
+ info="Confluent Cloud environment ID (for example env-abc123).",
60
+ required=True,
61
+ ),
62
+ StrInput(
63
+ name="kafka_cluster_id",
64
+ display_name="Kafka Cluster ID",
65
+ info="Kafka cluster ID (for example lkc-abc123) whose topics have the Context Engine enabled.",
66
+ required=True,
67
+ ),
68
+ SecretStrInput(
69
+ name="api_key",
70
+ display_name="API Key",
71
+ info="Confluent Cloud Global API key with read access to the cluster and Schema Registry.",
72
+ required=True,
73
+ ),
74
+ SecretStrInput(
75
+ name="api_secret",
76
+ display_name="API Secret",
77
+ info="Secret paired with the API key.",
78
+ required=True,
79
+ ),
80
+ StrInput(
81
+ name="endpoint_override",
82
+ display_name="Endpoint Override",
83
+ info=(
84
+ "Full MCP endpoint URL. Leave empty to build it from the region and IDs; set it when "
85
+ "Confluent publishes a different host for your cloud provider or region."
86
+ ),
87
+ advanced=True,
88
+ ),
89
+ StrInput(
90
+ name="cloud",
91
+ display_name="Cloud Provider",
92
+ info="Cloud provider segment of the endpoint host. The Context Engine is available on AWS today.",
93
+ value=DEFAULT_CLOUD,
94
+ advanced=True,
95
+ ),
96
+ *preset_control_inputs(
97
+ CONTEXT_ENGINE_TOOLS,
98
+ tool_info=(
99
+ "Tool to run for the Response output. In Tool Mode all three tools are exposed to the "
100
+ "Agent. Use the refresh button to re-read the tool list from the server."
101
+ ),
102
+ ),
103
+ ]
104
+
105
+ def endpoint_url(self) -> str:
106
+ override = (getattr(self, "endpoint_override", "") or "").strip()
107
+ if override:
108
+ return ensure_url_allowed(override)
109
+ url = context_engine_url(
110
+ self.region,
111
+ self.organization_id,
112
+ self.environment_id,
113
+ self.kafka_cluster_id,
114
+ cloud=getattr(self, "cloud", DEFAULT_CLOUD) or DEFAULT_CLOUD,
115
+ )
116
+ return ensure_url_allowed(url)
117
+
118
+ def _mcp_server_config(self) -> tuple[str, dict[str, Any]]:
119
+ url = self.endpoint_url()
120
+ headers = {"Authorization": basic_auth_header(self.api_key, self.api_secret)}
121
+ server_name = f"confluent-context-engine-{(self.kafka_cluster_id or '').strip() or 'cluster'}"
122
+ return server_name, {
123
+ "url": url,
124
+ "headers": headers,
125
+ "mode": "Streamable_HTTP",
126
+ "verify_ssl": bool(getattr(self, "verify_ssl", True)),
127
+ }