batch-analytics 0.3.34__tar.gz → 0.3.35__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/PKG-INFO +1 -1
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/pyproject.toml +1 -1
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/config.py +21 -5
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/job_runner.py +83 -8
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/transform.py +96 -21
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics.egg-info/PKG-INFO +1 -1
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics.egg-info/SOURCES.txt +1 -0
- batch_analytics-0.3.35/tests/test_iceberg_staging.py +85 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/README.md +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/setup.cfg +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/__init__.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/__main__.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/__init__.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/correlation.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/equipment_oee.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/gluon_autogluon_infer.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/gluon_autogluon_train.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/linear_regression.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/pca.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/pca_clustering.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/pca_core.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/pca_mvda.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/t_test.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/extract.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/log.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/modules.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/output/__init__.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/output/base.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/output/clickhouse.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/output/local.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/output/s3.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/utils/__init__.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/utils/gluon_autogluon_common.py +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics.egg-info/dependency_links.txt +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics.egg-info/entry_points.txt +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics.egg-info/requires.txt +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics.egg-info/top_level.txt +0 -0
- {batch_analytics-0.3.34 → batch_analytics-0.3.35}/tests/test_pca_mvda.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: batch-analytics
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.35
|
|
4
4
|
Summary: PySpark batch analytics: Extract, Transform, Stage, and analytical modules (linear regression, correlation, PCA, t-test, LLM classification).
|
|
5
5
|
Author: Litewave Analytics Team
|
|
6
6
|
License: MIT
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "batch-analytics"
|
|
7
|
-
version = "0.3.
|
|
7
|
+
version = "0.3.35"
|
|
8
8
|
description = "PySpark batch analytics: Extract, Transform, Stage, and analytical modules (linear regression, correlation, PCA, t-test, LLM classification)."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.8"
|
|
@@ -85,15 +85,24 @@ class TransformConfig:
|
|
|
85
85
|
"/tmp/analytics_stage",
|
|
86
86
|
)
|
|
87
87
|
)
|
|
88
|
-
# Output format for load_staged when reading (parquet/delta/clickhouse).
|
|
88
|
+
# Output format for load_staged when reading (parquet/delta/clickhouse/iceberg).
|
|
89
89
|
# Stage job always writes to ClickHouse; use clickhouse for analytics to read from staged table.
|
|
90
|
+
# Use iceberg to read Nessie Iceberg gold/silver tables (BATCH_STAGING_TABLE = catalog.ns.table).
|
|
90
91
|
staging_format: str = field(
|
|
91
92
|
default_factory=lambda: os.environ.get("BATCH_STAGING_FORMAT", "clickhouse")
|
|
92
93
|
)
|
|
93
|
-
# Staging table name
|
|
94
|
+
# Staging table: ClickHouse table name, or Iceberg identifier e.g. nessie.gold.batch_yield_features
|
|
94
95
|
staging_table: str = field(
|
|
95
96
|
default_factory=lambda: os.environ.get("BATCH_STAGING_TABLE", "analytics_staging")
|
|
96
97
|
)
|
|
98
|
+
# Optional SQL WHERE fragment when reading staging (no leading WHERE).
|
|
99
|
+
staging_filter: str = field(
|
|
100
|
+
default_factory=lambda: os.environ.get("BATCH_STAGING_FILTER", "").strip()
|
|
101
|
+
)
|
|
102
|
+
# Optional comma-separated columns when reading staging (empty = SELECT *).
|
|
103
|
+
staging_columns: str = field(
|
|
104
|
+
default_factory=lambda: os.environ.get("BATCH_STAGING_COLUMNS", "").strip()
|
|
105
|
+
)
|
|
97
106
|
# Spark save mode for ClickHouse staging (and path staging): overwrite | append
|
|
98
107
|
staging_write_mode: str = field(
|
|
99
108
|
default_factory=lambda: os.environ.get("BATCH_STAGING_WRITE_MODE", "overwrite")
|
|
@@ -211,11 +220,18 @@ class SparkK8sConfig:
|
|
|
211
220
|
executor_cores: int = int(os.environ.get("SPARK_EXECUTOR_CORES", "1"))
|
|
212
221
|
executor_memory: str = os.environ.get("SPARK_EXECUTOR_MEMORY", "512m")
|
|
213
222
|
executor_memory_overhead: str = os.environ.get("SPARK_EXECUTOR_MEMORY_OVERHEAD", "128m")
|
|
214
|
-
# S3 (optional; set for s3a:// paths)
|
|
223
|
+
# S3 (optional; set for s3a:// paths / Iceberg warehouse)
|
|
215
224
|
s3_access_key: str = os.environ.get("AWS_ACCESS_KEY_ID", "")
|
|
216
225
|
s3_secret_key: str = os.environ.get("AWS_SECRET_ACCESS_KEY", "")
|
|
217
|
-
|
|
218
|
-
|
|
226
|
+
# Prefer S3_ENDPOINT (lakehouse MinIO); fall back to AWS_ENDPOINT for legacy.
|
|
227
|
+
s3_endpoint: str = os.environ.get(
|
|
228
|
+
"S3_ENDPOINT",
|
|
229
|
+
os.environ.get("AWS_ENDPOINT", ""),
|
|
230
|
+
)
|
|
231
|
+
s3_region: str = os.environ.get(
|
|
232
|
+
"AWS_DEFAULT_REGION",
|
|
233
|
+
os.environ.get("AWS_REGION", "us-east-2"),
|
|
234
|
+
)
|
|
219
235
|
|
|
220
236
|
|
|
221
237
|
@dataclass
|
|
@@ -161,6 +161,9 @@ def create_spark_session(
|
|
|
161
161
|
read_codec,
|
|
162
162
|
)
|
|
163
163
|
|
|
164
|
+
builder = _apply_iceberg_nessie_catalog(builder, config)
|
|
165
|
+
builder = _apply_s3a_lakehouse(builder, config)
|
|
166
|
+
|
|
164
167
|
if cfg.master.startswith("k8s://"):
|
|
165
168
|
driver_host = socket.gethostbyname(socket.gethostname())
|
|
166
169
|
builder = (
|
|
@@ -188,19 +191,91 @@ def create_spark_session(
|
|
|
188
191
|
.config("spark.kubernetes.executor.serviceAccountName", cfg.service_account)
|
|
189
192
|
.config("spark.kubernetes.container.image.pullPolicy", "IfNotPresent")
|
|
190
193
|
)
|
|
191
|
-
if cfg.s3_access_key and cfg.s3_secret_key:
|
|
192
|
-
builder = (
|
|
193
|
-
builder.config("spark.hadoop.fs.s3a.impl", "org.apache.hadoop.fs.s3a.S3AFileSystem")
|
|
194
|
-
.config("spark.hadoop.fs.s3a.access.key", cfg.s3_access_key)
|
|
195
|
-
.config("spark.hadoop.fs.s3a.secret.key", cfg.s3_secret_key)
|
|
196
|
-
.config("spark.hadoop.fs.s3a.endpoint", cfg.s3_endpoint)
|
|
197
|
-
.config("spark.hadoop.fs.s3a.endpoint.region", cfg.s3_region)
|
|
198
|
-
)
|
|
199
194
|
logger.info("Spark on Kubernetes: master=%s", cfg.master)
|
|
200
195
|
|
|
201
196
|
return builder.getOrCreate()
|
|
202
197
|
|
|
203
198
|
|
|
199
|
+
def _apply_iceberg_nessie_catalog(builder, config: BatchAnalyticsConfig):
|
|
200
|
+
"""Register Nessie Iceberg catalog when lakehouse env is present."""
|
|
201
|
+
nessie_uri = (os.environ.get("NESSIE_URI") or "").strip()
|
|
202
|
+
warehouse = (os.environ.get("ICEBERG_WAREHOUSE") or "").strip()
|
|
203
|
+
if not nessie_uri and not warehouse:
|
|
204
|
+
return builder
|
|
205
|
+
catalog = (os.environ.get("ICEBERG_CATALOG") or "nessie").strip() or "nessie"
|
|
206
|
+
ref = (os.environ.get("NESSIE_REF") or "main").strip() or "main"
|
|
207
|
+
if not nessie_uri:
|
|
208
|
+
nessie_uri = "http://nessie:19120/api/v2"
|
|
209
|
+
if not warehouse:
|
|
210
|
+
warehouse = "s3a://lake/"
|
|
211
|
+
logger.info(
|
|
212
|
+
"Configuring Iceberg catalog %s uri=%s ref=%s warehouse=%s",
|
|
213
|
+
catalog,
|
|
214
|
+
nessie_uri,
|
|
215
|
+
ref,
|
|
216
|
+
warehouse,
|
|
217
|
+
)
|
|
218
|
+
return (
|
|
219
|
+
builder.config(
|
|
220
|
+
"spark.sql.extensions",
|
|
221
|
+
"org.apache.iceberg.spark.extensions.IcebergSparkSessionExtensions",
|
|
222
|
+
)
|
|
223
|
+
.config(f"spark.sql.catalog.{catalog}", "org.apache.iceberg.spark.SparkCatalog")
|
|
224
|
+
.config(
|
|
225
|
+
f"spark.sql.catalog.{catalog}.catalog-impl",
|
|
226
|
+
"org.apache.iceberg.nessie.NessieCatalog",
|
|
227
|
+
)
|
|
228
|
+
.config(f"spark.sql.catalog.{catalog}.uri", nessie_uri)
|
|
229
|
+
.config(f"spark.sql.catalog.{catalog}.ref", ref)
|
|
230
|
+
.config(f"spark.sql.catalog.{catalog}.warehouse", warehouse)
|
|
231
|
+
.config(
|
|
232
|
+
f"spark.sql.catalog.{catalog}.io-impl",
|
|
233
|
+
"org.apache.iceberg.hadoop.HadoopFileIO",
|
|
234
|
+
)
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _apply_s3a_lakehouse(builder, config: BatchAnalyticsConfig):
|
|
239
|
+
"""S3A settings for Iceberg warehouse / s3a staging (AWS or MinIO via S3_ENDPOINT)."""
|
|
240
|
+
cfg = config.spark_k8s
|
|
241
|
+
if not (cfg.s3_access_key or cfg.s3_secret_key or os.environ.get("ICEBERG_WAREHOUSE")):
|
|
242
|
+
return builder
|
|
243
|
+
|
|
244
|
+
builder = builder.config(
|
|
245
|
+
"spark.hadoop.fs.s3a.impl",
|
|
246
|
+
"org.apache.hadoop.fs.s3a.S3AFileSystem",
|
|
247
|
+
)
|
|
248
|
+
if cfg.s3_access_key:
|
|
249
|
+
builder = builder.config("spark.hadoop.fs.s3a.access.key", cfg.s3_access_key)
|
|
250
|
+
if cfg.s3_secret_key:
|
|
251
|
+
builder = builder.config("spark.hadoop.fs.s3a.secret.key", cfg.s3_secret_key)
|
|
252
|
+
|
|
253
|
+
endpoint = (cfg.s3_endpoint or "").strip()
|
|
254
|
+
region = (cfg.s3_region or "us-east-2").strip()
|
|
255
|
+
if endpoint:
|
|
256
|
+
# Custom/MinIO endpoint
|
|
257
|
+
ssl = not endpoint.startswith("http://")
|
|
258
|
+
builder = (
|
|
259
|
+
builder.config("spark.hadoop.fs.s3a.endpoint", endpoint)
|
|
260
|
+
.config("spark.hadoop.fs.s3a.path.style.access", "true")
|
|
261
|
+
.config(
|
|
262
|
+
"spark.hadoop.fs.s3a.connection.ssl.enabled",
|
|
263
|
+
"true" if ssl else "false",
|
|
264
|
+
)
|
|
265
|
+
)
|
|
266
|
+
else:
|
|
267
|
+
builder = (
|
|
268
|
+
builder.config(
|
|
269
|
+
"spark.hadoop.fs.s3a.endpoint",
|
|
270
|
+
f"https://s3.{region}.amazonaws.com",
|
|
271
|
+
)
|
|
272
|
+
.config("spark.hadoop.fs.s3a.endpoint.region", region)
|
|
273
|
+
.config("spark.hadoop.fs.s3a.path.style.access", "false")
|
|
274
|
+
.config("spark.hadoop.fs.s3a.connection.ssl.enabled", "true")
|
|
275
|
+
)
|
|
276
|
+
return builder
|
|
277
|
+
|
|
278
|
+
|
|
204
279
|
def run_pipeline(
|
|
205
280
|
config: Optional[BatchAnalyticsConfig] = None,
|
|
206
281
|
spark: Optional[SparkSession] = None,
|
|
@@ -534,33 +534,108 @@ def load_staged(
|
|
|
534
534
|
) -> DataFrame:
|
|
535
535
|
"""
|
|
536
536
|
Load previously staged data (e.g. when running only analytics modules).
|
|
537
|
+
|
|
538
|
+
Formats:
|
|
539
|
+
parquet / delta — path under BATCH_STAGING_PATH
|
|
540
|
+
clickhouse — BATCH_STAGING_TABLE in ClickHouse
|
|
541
|
+
iceberg — Nessie Iceberg table (BATCH_STAGING_TABLE = catalog.ns.table)
|
|
537
542
|
"""
|
|
538
543
|
staging_path = config.transform.staging_path
|
|
539
|
-
fmt = config.transform.staging_format
|
|
544
|
+
fmt = (config.transform.staging_format or "").strip().lower()
|
|
540
545
|
|
|
541
546
|
if fmt == "parquet":
|
|
542
547
|
return spark.read.parquet(staging_path)
|
|
543
548
|
if fmt == "delta":
|
|
544
549
|
return spark.read.format("delta").load(staging_path)
|
|
550
|
+
if fmt == "iceberg":
|
|
551
|
+
return _load_staged_iceberg(spark, config)
|
|
545
552
|
if fmt == "clickhouse":
|
|
546
|
-
|
|
547
|
-
tbl = config.transform.staging_table
|
|
548
|
-
cat = os.environ.get("BATCH_CLICKHOUSE_CATALOG", "batch_ch").strip()
|
|
549
|
-
if cat:
|
|
550
|
-
try:
|
|
551
|
-
return spark.table(f"{cat}.{ch.database}.{tbl}")
|
|
552
|
-
except Exception as e:
|
|
553
|
-
logger.warning(
|
|
554
|
-
"load_staged: catalog table %s.%s.%s failed (%s), using JDBC",
|
|
555
|
-
cat,
|
|
556
|
-
ch.database,
|
|
557
|
-
tbl,
|
|
558
|
-
e,
|
|
559
|
-
)
|
|
560
|
-
dbtable = f"(SELECT * FROM `{ch.database}`.`{tbl}`) AS _stg"
|
|
561
|
-
return spark.read.jdbc(
|
|
562
|
-
ch.jdbc_url,
|
|
563
|
-
dbtable,
|
|
564
|
-
properties=ch.jdbc_properties,
|
|
565
|
-
)
|
|
553
|
+
return _load_staged_clickhouse(spark, config)
|
|
566
554
|
return spark.read.format(fmt).load(staging_path)
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
def _staging_select_list(config: BatchAnalyticsConfig) -> str:
|
|
558
|
+
cols = (config.transform.staging_columns or "").strip()
|
|
559
|
+
if not cols:
|
|
560
|
+
return "*"
|
|
561
|
+
return ", ".join(c.strip() for c in cols.split(",") if c.strip())
|
|
562
|
+
|
|
563
|
+
|
|
564
|
+
def _staging_where_clause(config: BatchAnalyticsConfig) -> str:
|
|
565
|
+
filt = (config.transform.staging_filter or "").strip()
|
|
566
|
+
if not filt:
|
|
567
|
+
return ""
|
|
568
|
+
if filt.lower().startswith("where "):
|
|
569
|
+
filt = filt[6:].strip()
|
|
570
|
+
return f" WHERE {filt}" if filt else ""
|
|
571
|
+
|
|
572
|
+
|
|
573
|
+
def _resolve_iceberg_table_name(config: BatchAnalyticsConfig) -> str:
|
|
574
|
+
"""Return fully-qualified Iceberg table id (catalog.namespace.table)."""
|
|
575
|
+
raw = (config.transform.staging_table or "").strip()
|
|
576
|
+
if not raw:
|
|
577
|
+
raise ValueError(
|
|
578
|
+
"BATCH_STAGING_TABLE is required when BATCH_STAGING_FORMAT=iceberg "
|
|
579
|
+
"(e.g. nessie.gold.batch_yield_features)"
|
|
580
|
+
)
|
|
581
|
+
catalog = (os.environ.get("ICEBERG_CATALOG") or "nessie").strip()
|
|
582
|
+
parts = [p for p in raw.replace("/", ".").split(".") if p]
|
|
583
|
+
if len(parts) == 1:
|
|
584
|
+
return f"{catalog}.gold.{parts[0]}"
|
|
585
|
+
if len(parts) == 2:
|
|
586
|
+
return f"{catalog}.{parts[0]}.{parts[1]}"
|
|
587
|
+
return ".".join(parts)
|
|
588
|
+
|
|
589
|
+
|
|
590
|
+
def _load_staged_iceberg(
|
|
591
|
+
spark: SparkSession,
|
|
592
|
+
config: BatchAnalyticsConfig,
|
|
593
|
+
) -> DataFrame:
|
|
594
|
+
table = _resolve_iceberg_table_name(config)
|
|
595
|
+
sql = (
|
|
596
|
+
f"SELECT {_staging_select_list(config)} FROM {table}"
|
|
597
|
+
f"{_staging_where_clause(config)}"
|
|
598
|
+
)
|
|
599
|
+
logger.info("load_staged iceberg: %s", sql)
|
|
600
|
+
return spark.sql(sql)
|
|
601
|
+
|
|
602
|
+
|
|
603
|
+
def _load_staged_clickhouse(
|
|
604
|
+
spark: SparkSession,
|
|
605
|
+
config: BatchAnalyticsConfig,
|
|
606
|
+
) -> DataFrame:
|
|
607
|
+
ch = config.clickhouse
|
|
608
|
+
tbl = config.transform.staging_table
|
|
609
|
+
where = _staging_where_clause(config)
|
|
610
|
+
select_list = _staging_select_list(config)
|
|
611
|
+
cat = os.environ.get("BATCH_CLICKHOUSE_CATALOG", "batch_ch").strip()
|
|
612
|
+
if cat and not where and select_list == "*":
|
|
613
|
+
try:
|
|
614
|
+
return spark.table(f"{cat}.{ch.database}.{tbl}")
|
|
615
|
+
except Exception as e:
|
|
616
|
+
logger.warning(
|
|
617
|
+
"load_staged: catalog table %s.%s.%s failed (%s), using JDBC",
|
|
618
|
+
cat,
|
|
619
|
+
ch.database,
|
|
620
|
+
tbl,
|
|
621
|
+
e,
|
|
622
|
+
)
|
|
623
|
+
elif cat and (where or select_list != "*"):
|
|
624
|
+
sql = f"SELECT {select_list} FROM {cat}.{ch.database}.{tbl}{where}"
|
|
625
|
+
try:
|
|
626
|
+
logger.info("load_staged clickhouse catalog: %s", sql)
|
|
627
|
+
return spark.sql(sql)
|
|
628
|
+
except Exception as e:
|
|
629
|
+
logger.warning(
|
|
630
|
+
"load_staged: catalog SQL failed (%s), using JDBC",
|
|
631
|
+
e,
|
|
632
|
+
)
|
|
633
|
+
dbtable = (
|
|
634
|
+
f"(SELECT {select_list} FROM `{ch.database}`.`{tbl}`{where}) AS _stg"
|
|
635
|
+
)
|
|
636
|
+
logger.info("load_staged clickhouse jdbc: %s", dbtable)
|
|
637
|
+
return spark.read.jdbc(
|
|
638
|
+
ch.jdbc_url,
|
|
639
|
+
dbtable,
|
|
640
|
+
properties=ch.jdbc_properties,
|
|
641
|
+
)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: batch-analytics
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.35
|
|
4
4
|
Summary: PySpark batch analytics: Extract, Transform, Stage, and analytical modules (linear regression, correlation, PCA, t-test, LLM classification).
|
|
5
5
|
Author: Litewave Analytics Team
|
|
6
6
|
License: MIT
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""Unit tests for Iceberg staging load helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
import unittest
|
|
7
|
+
from unittest.mock import MagicMock, patch
|
|
8
|
+
|
|
9
|
+
from batch_analytics.config import BatchAnalyticsConfig, TransformConfig
|
|
10
|
+
from batch_analytics.transform import _load_staged_iceberg, _resolve_iceberg_table_name
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class ResolveIcebergTableNameTests(unittest.TestCase):
|
|
14
|
+
def test_full_name_passthrough(self) -> None:
|
|
15
|
+
cfg = BatchAnalyticsConfig(
|
|
16
|
+
transform=TransformConfig(staging_table="nessie.gold.batch_yield_features")
|
|
17
|
+
)
|
|
18
|
+
self.assertEqual(
|
|
19
|
+
_resolve_iceberg_table_name(cfg),
|
|
20
|
+
"nessie.gold.batch_yield_features",
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
def test_two_part_gets_catalog(self) -> None:
|
|
24
|
+
cfg = BatchAnalyticsConfig(
|
|
25
|
+
transform=TransformConfig(staging_table="gold.batch_yield_features")
|
|
26
|
+
)
|
|
27
|
+
with patch.dict(os.environ, {"ICEBERG_CATALOG": "nessie"}, clear=False):
|
|
28
|
+
self.assertEqual(
|
|
29
|
+
_resolve_iceberg_table_name(cfg),
|
|
30
|
+
"nessie.gold.batch_yield_features",
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
def test_one_part_defaults_to_gold(self) -> None:
|
|
34
|
+
cfg = BatchAnalyticsConfig(
|
|
35
|
+
transform=TransformConfig(staging_table="batch_yield_features")
|
|
36
|
+
)
|
|
37
|
+
with patch.dict(os.environ, {"ICEBERG_CATALOG": "nessie"}, clear=False):
|
|
38
|
+
self.assertEqual(
|
|
39
|
+
_resolve_iceberg_table_name(cfg),
|
|
40
|
+
"nessie.gold.batch_yield_features",
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class LoadStagedIcebergTests(unittest.TestCase):
|
|
45
|
+
def test_filter_and_columns(self) -> None:
|
|
46
|
+
spark = MagicMock()
|
|
47
|
+
spark.sql.return_value = MagicMock(name="df")
|
|
48
|
+
cfg = BatchAnalyticsConfig(
|
|
49
|
+
transform=TransformConfig(
|
|
50
|
+
staging_table="nessie.gold.batch_yield_features",
|
|
51
|
+
staging_filter="product = 'Gabapentin'",
|
|
52
|
+
staging_columns="batch_no,actual_yield_qty,batch_seq",
|
|
53
|
+
)
|
|
54
|
+
)
|
|
55
|
+
_load_staged_iceberg(spark, cfg)
|
|
56
|
+
spark.sql.assert_called_once()
|
|
57
|
+
sql = spark.sql.call_args[0][0]
|
|
58
|
+
self.assertIn("SELECT batch_no, actual_yield_qty, batch_seq FROM", sql)
|
|
59
|
+
self.assertIn("WHERE product = 'Gabapentin'", sql)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class LoadStagedClickhouseTests(unittest.TestCase):
|
|
63
|
+
def test_filter_pushed_to_jdbc(self) -> None:
|
|
64
|
+
from batch_analytics.transform import _load_staged_clickhouse
|
|
65
|
+
|
|
66
|
+
spark = MagicMock()
|
|
67
|
+
spark.read.jdbc.return_value = MagicMock(name="df")
|
|
68
|
+
cfg = BatchAnalyticsConfig(
|
|
69
|
+
transform=TransformConfig(
|
|
70
|
+
staging_table="analytics_staging",
|
|
71
|
+
staging_filter="batch_id = 'B1'",
|
|
72
|
+
staging_columns="batch_id,qty",
|
|
73
|
+
)
|
|
74
|
+
)
|
|
75
|
+
cfg.clickhouse.database = "example_db"
|
|
76
|
+
with patch.dict(os.environ, {"BATCH_CLICKHOUSE_CATALOG": ""}, clear=False):
|
|
77
|
+
_load_staged_clickhouse(spark, cfg)
|
|
78
|
+
spark.read.jdbc.assert_called_once()
|
|
79
|
+
dbtable = spark.read.jdbc.call_args[0][1]
|
|
80
|
+
self.assertIn("SELECT batch_id, qty FROM", dbtable)
|
|
81
|
+
self.assertIn("WHERE batch_id = 'B1'", dbtable)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
if __name__ == "__main__":
|
|
85
|
+
unittest.main()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/correlation.py
RENAMED
|
File without changes
|
{batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/equipment_oee.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/linear_regression.py
RENAMED
|
File without changes
|
|
File without changes
|
{batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics/analytics/pca_clustering.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics.egg-info/entry_points.txt
RENAMED
|
File without changes
|
|
File without changes
|
{batch_analytics-0.3.34 → batch_analytics-0.3.35}/src/batch_analytics.egg-info/top_level.txt
RENAMED
|
File without changes
|
|
File without changes
|