linkmerce 0.3.2__tar.gz → 0.3.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {linkmerce-0.3.2 → linkmerce-0.3.4}/PKG-INFO +2 -1
- {linkmerce-0.3.2 → linkmerce-0.3.4}/pyproject.toml +2 -1
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/.DS_Store +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/api/searchad/manage.py +2 -2
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/common/load.py +54 -64
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/naver/openapi/search/models.sql +9 -2
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/searchad/manage/exposure/models.sql +14 -10
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/searchad/manage/exposure/transform.py +4 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/brand/sales/transform.py +1 -1
- linkmerce-0.3.4/src/linkmerce/extensions/.DS_Store +0 -0
- {linkmerce-0.3.2/src/linkmerce/utils → linkmerce-0.3.4/src/linkmerce/extensions}/__init__.py +0 -0
- linkmerce-0.3.4/src/linkmerce/extensions/bigquery.py +468 -0
- linkmerce-0.3.4/src/linkmerce/utils/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/utils/tqdm.py +23 -3
- {linkmerce-0.3.2 → linkmerce-0.3.4}/README.md +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/api/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/api/naver/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/api/naver/openapi.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/api/searchad/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/api/smartstore/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/api/smartstore/brand.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/common/.DS_Store +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/common/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/common/api.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/common/exceptions.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/common/extract.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/common/models.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/common/tasks.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/common/transform.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/.DS_Store +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/naver/.DS_Store +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/naver/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/naver/openapi/.DS_Store +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/naver/openapi/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/naver/openapi/common.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/naver/openapi/search/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/naver/openapi/search/extract.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/naver/openapi/search/transform.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/searchad/.DS_Store +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/searchad/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/searchad/manage/.DS_Store +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/searchad/manage/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/searchad/manage/common.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/searchad/manage/exposure/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/searchad/manage/exposure/extract.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/.DS_Store +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/brand/.DS_Store +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/brand/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/brand/catalog/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/brand/catalog/extract.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/brand/catalog/models.sql +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/brand/catalog/transform.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/brand/common.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/brand/sales/__init__.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/brand/sales/extract.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/brand/sales/models.sql +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/utils/.DS_Store +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/utils/cast.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/utils/date.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/utils/graphql.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/utils/headers.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/utils/jinja.py +0 -0
- {linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/utils/map.py +0 -0
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: linkmerce
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.4
|
|
4
4
|
Summary: E-commerce API integration management
|
|
5
5
|
Requires-Dist: aiohttp>=3.12.15
|
|
6
6
|
Requires-Dist: bs4>=0.0.2
|
|
7
7
|
Requires-Dist: duckdb>=1.3.2
|
|
8
8
|
Requires-Dist: jinja2>=3.1.6
|
|
9
|
+
Requires-Dist: nest-asyncio>=1.6.0
|
|
9
10
|
Requires-Dist: pytz>=2025.2
|
|
10
11
|
Requires-Dist: requests>=2.32.4
|
|
11
12
|
Requires-Dist: tqdm>=4.67.1
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "linkmerce"
|
|
3
|
-
version = "0.3.
|
|
3
|
+
version = "0.3.4"
|
|
4
4
|
description = "E-commerce API integration management"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.10"
|
|
@@ -9,6 +9,7 @@ dependencies = [
|
|
|
9
9
|
"bs4>=0.0.2",
|
|
10
10
|
"duckdb>=1.3.2",
|
|
11
11
|
"jinja2>=3.1.6",
|
|
12
|
+
"nest-asyncio>=1.6.0",
|
|
12
13
|
"pytz>=2025.2",
|
|
13
14
|
"requests>=2.32.4",
|
|
14
15
|
"tqdm>=4.67.1",
|
|
Binary file
|
|
@@ -15,7 +15,7 @@ def get_module(name: str) -> str:
|
|
|
15
15
|
return (".searchad.manage" + name) if name.startswith('.') else name
|
|
16
16
|
|
|
17
17
|
|
|
18
|
-
def
|
|
18
|
+
def diagnose_exposure(
|
|
19
19
|
customer_id: str,
|
|
20
20
|
cookies: str,
|
|
21
21
|
keyword: str | Iterable[str],
|
|
@@ -34,7 +34,7 @@ def exposure_diagnosis(
|
|
|
34
34
|
return run_with_duckdb(get_module(".exposure"), "ExposureDiagnosis", "ExposureDiagnosis", connection, "sync", table, return_type, args, **options)
|
|
35
35
|
|
|
36
36
|
|
|
37
|
-
def
|
|
37
|
+
def rank_exposure(
|
|
38
38
|
customer_id: str,
|
|
39
39
|
cookies: str,
|
|
40
40
|
keyword: str | Iterable[str],
|
|
@@ -21,6 +21,12 @@ def concat_sql(*statement: str, drop_empty: bool = True, sep=' ', terminate: boo
|
|
|
21
21
|
return query + ';' if terminate and not query.endswith(';') else query
|
|
22
22
|
|
|
23
23
|
|
|
24
|
+
def where(where_clause: str | None = None, default: str | None = None) -> str:
|
|
25
|
+
if not (where_clause or default):
|
|
26
|
+
return str()
|
|
27
|
+
return f"WHERE {where_clause or default}"
|
|
28
|
+
|
|
29
|
+
|
|
24
30
|
def csv_to_json(obj: list[tuple], header: int | list[str] = 0) -> list[dict]:
|
|
25
31
|
if isinstance(header, int):
|
|
26
32
|
header, obj = obj[header], obj[header+1:]
|
|
@@ -112,16 +118,12 @@ class Connection(metaclass=ABCMeta):
|
|
|
112
118
|
|
|
113
119
|
############################ Expression ###########################
|
|
114
120
|
|
|
115
|
-
def
|
|
116
|
-
|
|
117
|
-
if
|
|
118
|
-
|
|
119
|
-
else:
|
|
120
|
-
func = "TRY_CAST" if safe else "CAST"
|
|
121
|
-
alias = f" AS {alias}" if alias else str()
|
|
122
|
-
return f"{func}({value} AS {type})" + alias
|
|
121
|
+
def expr_cast(self, value: Any | None, type: str, alias: str = str(), safe: bool = False) -> str:
|
|
122
|
+
cast = "TRY_CAST" if safe else "CAST"
|
|
123
|
+
alias = f" AS {alias}" if alias else str()
|
|
124
|
+
return f"{cast}({self.expr_value(value)} AS {type.upper()})" + alias
|
|
123
125
|
|
|
124
|
-
def expr_create(self, option: Literal["replace",
|
|
126
|
+
def expr_create(self, option: Literal["replace","ignore"] | None = None, temp: bool = False) -> str:
|
|
125
127
|
temp = "TEMP" if temp else str()
|
|
126
128
|
if option == "replace":
|
|
127
129
|
return f"CREATE OR REPLACE {temp} TABLE"
|
|
@@ -130,20 +132,14 @@ class Connection(metaclass=ABCMeta):
|
|
|
130
132
|
else:
|
|
131
133
|
return f"CREATE {temp} TABLE"
|
|
132
134
|
|
|
133
|
-
def
|
|
134
|
-
|
|
135
|
-
if
|
|
136
|
-
return
|
|
137
|
-
|
|
138
|
-
return
|
|
139
|
-
|
|
140
|
-
def expr_interval(value: str | int | None = None) -> str:
|
|
141
|
-
if isinstance(value, str):
|
|
142
|
-
return value
|
|
143
|
-
elif isinstance(value, int):
|
|
144
|
-
return "{} INTERVAL {} DAY".format('-' if value < 0 else '+', abs(value))
|
|
135
|
+
def expr_value(self, value: Any | None) -> str:
|
|
136
|
+
import datetime as dt
|
|
137
|
+
if value is None:
|
|
138
|
+
return "NULL"
|
|
139
|
+
elif isinstance(value, (float,int)):
|
|
140
|
+
return str(value)
|
|
145
141
|
else:
|
|
146
|
-
return
|
|
142
|
+
return f"'{value}'"
|
|
147
143
|
|
|
148
144
|
def expr_now(
|
|
149
145
|
self,
|
|
@@ -173,14 +169,22 @@ class Connection(metaclass=ABCMeta):
|
|
|
173
169
|
return f"STRFTIME({expr}, '{format}')"
|
|
174
170
|
return expr if type.upper() == "DATE" else "NULL"
|
|
175
171
|
|
|
172
|
+
def expr_interval(days: str | int | None = None) -> str:
|
|
173
|
+
if isinstance(days, str):
|
|
174
|
+
return days
|
|
175
|
+
elif isinstance(days, int):
|
|
176
|
+
return "{} INTERVAL {} DAY".format('-' if days < 0 else '+', abs(days))
|
|
177
|
+
else:
|
|
178
|
+
return str()
|
|
179
|
+
|
|
176
180
|
|
|
177
181
|
###################################################################
|
|
178
182
|
############################## DuckDB #############################
|
|
179
183
|
###################################################################
|
|
180
184
|
|
|
181
185
|
class DuckDBConnection(Connection):
|
|
182
|
-
def __init__(self, **kwargs):
|
|
183
|
-
self.set_connection(**kwargs)
|
|
186
|
+
def __init__(self, tzinfo: str | None = None, **kwargs):
|
|
187
|
+
self.set_connection(tzinfo, **kwargs)
|
|
184
188
|
|
|
185
189
|
@property
|
|
186
190
|
def conn(self) -> DuckDBPyConnection:
|
|
@@ -189,9 +193,11 @@ class DuckDBConnection(Connection):
|
|
|
189
193
|
def get_connection(self) -> DuckDBPyConnection:
|
|
190
194
|
return self.__conn
|
|
191
195
|
|
|
192
|
-
def set_connection(self, **kwargs):
|
|
196
|
+
def set_connection(self, tzinfo: str | None = None, **kwargs):
|
|
193
197
|
import duckdb
|
|
194
198
|
self.__conn = duckdb.connect(**kwargs)
|
|
199
|
+
if tzinfo is not None:
|
|
200
|
+
self.conn.execute(f"SET TimeZone = '{tzinfo}';")
|
|
195
201
|
|
|
196
202
|
def close(self):
|
|
197
203
|
try:
|
|
@@ -258,7 +264,7 @@ class DuckDBConnection(Connection):
|
|
|
258
264
|
save_to: str | Path | None = None,
|
|
259
265
|
) -> list[tuple] | None:
|
|
260
266
|
relation = self.conn.execute(query, parameters=params)
|
|
261
|
-
results = [self.get_columns(relation)] + relation.fetchall()
|
|
267
|
+
results = [tuple(self.get_columns(relation))] + relation.fetchall()
|
|
262
268
|
if save_to:
|
|
263
269
|
return save_to_csv(results, save_to, delimiter=',')
|
|
264
270
|
else:
|
|
@@ -358,7 +364,7 @@ class DuckDBConnection(Connection):
|
|
|
358
364
|
table: str,
|
|
359
365
|
values: list[tuple] | list[dict] | bytes | str | Path,
|
|
360
366
|
format: Literal["csv","json","parquet"],
|
|
361
|
-
option: Literal["replace",
|
|
367
|
+
option: Literal["replace","ignore"] | None = None,
|
|
362
368
|
temp: bool = False,
|
|
363
369
|
params: dict | None = None,
|
|
364
370
|
) -> DuckDBPyConnection:
|
|
@@ -371,7 +377,7 @@ class DuckDBConnection(Connection):
|
|
|
371
377
|
self,
|
|
372
378
|
table: str,
|
|
373
379
|
values: list[tuple] | str | Path,
|
|
374
|
-
option: Literal["replace",
|
|
380
|
+
option: Literal["replace","ignore"] | None = None,
|
|
375
381
|
temp: bool = False,
|
|
376
382
|
params: dict | None = None,
|
|
377
383
|
) -> DuckDBPyConnection:
|
|
@@ -381,7 +387,7 @@ class DuckDBConnection(Connection):
|
|
|
381
387
|
self,
|
|
382
388
|
table: str,
|
|
383
389
|
values: list[dict] | str | Path,
|
|
384
|
-
option: Literal["replace",
|
|
390
|
+
option: Literal["replace","ignore"] | None = None,
|
|
385
391
|
temp: bool = False,
|
|
386
392
|
params: dict | None = None,
|
|
387
393
|
) -> DuckDBPyConnection:
|
|
@@ -391,7 +397,7 @@ class DuckDBConnection(Connection):
|
|
|
391
397
|
self,
|
|
392
398
|
table: str,
|
|
393
399
|
values: bytes | str | Path,
|
|
394
|
-
option: Literal["replace",
|
|
400
|
+
option: Literal["replace","ignore"] | None = None,
|
|
395
401
|
temp: bool = False,
|
|
396
402
|
params: dict | None = None,
|
|
397
403
|
) -> DuckDBPyConnection:
|
|
@@ -399,14 +405,14 @@ class DuckDBConnection(Connection):
|
|
|
399
405
|
|
|
400
406
|
def copy_table(
|
|
401
407
|
self,
|
|
402
|
-
|
|
403
|
-
|
|
408
|
+
source_table: str,
|
|
409
|
+
target_table: str,
|
|
404
410
|
limit: int | None = 0,
|
|
405
|
-
option: Literal["replace",
|
|
411
|
+
option: Literal["replace","ignore"] | None = None,
|
|
406
412
|
temp: bool = False,
|
|
407
413
|
) -> DuckDBPyConnection:
|
|
408
414
|
limit_ = f"LIMIT {limit}" if isinstance(limit, int) else None
|
|
409
|
-
query = concat_sql(f"{self.expr_create(option, temp)} {
|
|
415
|
+
query = concat_sql(f"{self.expr_create(option, temp)} {target_table} AS SELECT * FROM {source_table}", limit_)
|
|
410
416
|
return self.conn.execute(query)
|
|
411
417
|
|
|
412
418
|
############################## Insert #############################
|
|
@@ -478,23 +484,26 @@ class DuckDBConnection(Connection):
|
|
|
478
484
|
elif agg in {"first","last","list"}:
|
|
479
485
|
return f"{agg.upper()}({col}) FILTER (WHERE {col} IS NOT NULL)"
|
|
480
486
|
else:
|
|
481
|
-
return agg
|
|
487
|
+
return f"{agg}({col})"
|
|
482
488
|
return ", ".join([f"{render(col, agg)} AS {col}" for col, agg in func.items()])
|
|
483
489
|
else:
|
|
484
490
|
return func
|
|
485
491
|
|
|
486
492
|
############################## Utils ##############################
|
|
487
493
|
|
|
488
|
-
def
|
|
494
|
+
def table_exists(self, table: str) -> bool:
|
|
489
495
|
query = f"SELECT 1 FROM information_schema.tables WHERE table_name = '{table}' LIMIT 1;"
|
|
490
496
|
return bool(self.conn.execute(query).fetchone())
|
|
491
497
|
|
|
492
|
-
def
|
|
493
|
-
if self.
|
|
498
|
+
def table_has_rows(self, table: str) -> bool:
|
|
499
|
+
if self.table_exists(table):
|
|
494
500
|
query = f"SELECT 1 FROM {table} LIMIT 1;"
|
|
495
501
|
return bool(self.conn.execute(query).fetchone())
|
|
496
502
|
return False
|
|
497
503
|
|
|
504
|
+
def count_table(self, table: str) -> int:
|
|
505
|
+
return self.conn.execute(f"SELECT COUNT(*) FROM {table}").fetchall()[0][0]
|
|
506
|
+
|
|
498
507
|
def get_columns(self, obj: str | DuckDBPyConnection) -> list[str]:
|
|
499
508
|
if isinstance(obj, str):
|
|
500
509
|
obj = self.conn.execute(f"DESCRIBE {obj}")
|
|
@@ -505,31 +514,12 @@ class DuckDBConnection(Connection):
|
|
|
505
514
|
def has_column(self, obj: str | DuckDBPyConnection, column: str) -> bool:
|
|
506
515
|
return column in self.get_columns(obj)
|
|
507
516
|
|
|
508
|
-
def unique(self, table: str, expr: str, ascending: bool | None = None,
|
|
517
|
+
def unique(self, table: str, expr: str, ascending: bool | None = None, where_clause: str | None = None) -> list:
|
|
509
518
|
select = f"SELECT DISTINCT {expr} AS expr FROM {table}"
|
|
510
519
|
order_by = "ORDER BY expr {}".format({True:"ASC", False:"DESC"}[ascending]) if isinstance(ascending, bool) else None
|
|
511
|
-
query = concat_sql(select,
|
|
520
|
+
query = concat_sql(select, where(where_clause), order_by)
|
|
512
521
|
return [row[0] for row in self.conn.execute(query).fetchall()]
|
|
513
522
|
|
|
514
|
-
def expr_where(self, condition: str | None = None, column: str | None = None, default: str = "WHERE TRUE") -> str:
|
|
515
|
-
if condition:
|
|
516
|
-
if condition.split(' ', maxsplit=1)[0].upper() == "WHERE":
|
|
517
|
-
return condition
|
|
518
|
-
else:
|
|
519
|
-
return concat_sql("WHERE", column, condition, terminate=False)
|
|
520
|
-
else:
|
|
521
|
-
return default
|
|
522
|
-
|
|
523
|
-
def expr_value(self, value: Any) -> str:
|
|
524
|
-
import datetime as dt
|
|
525
|
-
if isinstance(value, (float,int)):
|
|
526
|
-
return str(value)
|
|
527
|
-
elif isinstance(value, dt.date):
|
|
528
|
-
dtype = "TIMESTAMP" if isinstance(value, dt.datetime) else "DATE"
|
|
529
|
-
return f"{dtype} '{value}'"
|
|
530
|
-
else:
|
|
531
|
-
return f"'{value}'"
|
|
532
|
-
|
|
533
523
|
|
|
534
524
|
###################################################################
|
|
535
525
|
############################# Iterator ############################
|
|
@@ -564,7 +554,7 @@ class DuckDBIterator(Task):
|
|
|
564
554
|
self,
|
|
565
555
|
by: str | list[str],
|
|
566
556
|
ascending: bool | None = True,
|
|
567
|
-
|
|
557
|
+
where_clause: str | None = None,
|
|
568
558
|
if_errors: Literal["ignore","raise"] = "raise",
|
|
569
559
|
) -> DuckDBIterator:
|
|
570
560
|
from linkmerce.utils.tqdm import _expand_kwargs
|
|
@@ -573,11 +563,11 @@ class DuckDBIterator(Task):
|
|
|
573
563
|
if if_errors == "ignore":
|
|
574
564
|
from duckdb import BinderException
|
|
575
565
|
try:
|
|
576
|
-
map_partitions[expr] = self.conn.unique(self.table, expr, ascending,
|
|
566
|
+
map_partitions[expr] = self.conn.unique(self.table, expr, ascending, where_clause)
|
|
577
567
|
except BinderException:
|
|
578
568
|
continue
|
|
579
569
|
else:
|
|
580
|
-
map_partitions[expr] = self.conn.unique(self.table, expr, ascending,
|
|
570
|
+
map_partitions[expr] = self.conn.unique(self.table, expr, ascending, where_clause)
|
|
581
571
|
return self.setattr("partitions", _expand_kwargs(**map_partitions))
|
|
582
572
|
|
|
583
573
|
def __iter__(self) -> DuckDBIterator:
|
|
@@ -588,8 +578,8 @@ class DuckDBIterator(Task):
|
|
|
588
578
|
if self.index >= len(self):
|
|
589
579
|
raise StopIteration
|
|
590
580
|
map_partition = self.partitions[self.index]
|
|
591
|
-
|
|
592
|
-
query = f"SELECT * FROM {self.table}
|
|
581
|
+
where_clause = " AND ".join([f"{expr} = {self.conn.expr_value(value)}" for expr, value in map_partition.items()])
|
|
582
|
+
query = concat_sql(f"SELECT * FROM {self.table}", where(where_clause))
|
|
593
583
|
results = self.conn.fetch_all(self.format, query)
|
|
594
584
|
self.index += 1
|
|
595
585
|
return results
|
|
@@ -215,10 +215,10 @@ CREATE OR REPLACE TABLE {{ table }} (
|
|
|
215
215
|
keyword VARCHAR
|
|
216
216
|
, nvMid BIGINT
|
|
217
217
|
, mallPid BIGINT
|
|
218
|
-
, productType TINYINT -- {0: "가격비교 상품", 1: "가격비교 비매칭 일반상품", 2: "가격비교 매칭 일반상품"}
|
|
218
|
+
, productType TINYINT -- {0: "가격비교 상품", 1: "가격비교 비매칭 일반상품", 2: "가격비교 매칭 일반상품", 3: "광고상품"}
|
|
219
219
|
, displayRank SMALLINT
|
|
220
220
|
, createdAt TIMESTAMP NOT NULL
|
|
221
|
-
, PRIMARY KEY (keyword,
|
|
221
|
+
, PRIMARY KEY (keyword, displayRank)
|
|
222
222
|
);
|
|
223
223
|
|
|
224
224
|
-- ShoppingRank: select_rank
|
|
@@ -239,9 +239,12 @@ INSERT INTO {{ table }} {{ values }} ON CONFLICT DO NOTHING;
|
|
|
239
239
|
CREATE OR REPLACE TABLE {{ table }} (
|
|
240
240
|
nvMid BIGINT PRIMARY KEY
|
|
241
241
|
, mallPid BIGINT
|
|
242
|
+
, productType TINYINT -- {0: "가격비교 상품", 1: "일반상품", 3: "광고상품"}
|
|
242
243
|
, productName VARCHAR
|
|
244
|
+
, wholeCategoryName VARCHAR
|
|
243
245
|
, mallName VARCHAR
|
|
244
246
|
, brandName VARCHAR
|
|
247
|
+
, salesPrice INTEGER
|
|
245
248
|
, updatedAt TIMESTAMP NOT NULL
|
|
246
249
|
);
|
|
247
250
|
|
|
@@ -249,9 +252,12 @@ CREATE OR REPLACE TABLE {{ table }} (
|
|
|
249
252
|
SELECT
|
|
250
253
|
TRY_CAST(productId AS BIGINT) AS nvMid
|
|
251
254
|
, TRY_CAST(REGEXP_EXTRACT(link, '/products/(\d+)$', 1) AS BIGINT) AS mallPid
|
|
255
|
+
, IF(link LIKE '%/catalog/%', 0, 1) AS productType
|
|
252
256
|
, REGEXP_REPLACE(title, '<[^>]+>', '', 'g') AS productName
|
|
257
|
+
, CONCAT_WS('>', category1, category2, category3, category4) AS wholeCategoryName
|
|
253
258
|
, NULLIF(mallName, '네이버') AS mallName
|
|
254
259
|
, NULLIF(brand, '') AS brandName
|
|
260
|
+
, TRY_CAST(lprice AS INTEGER) AS salesPrice
|
|
255
261
|
, CAST(DATE_TRUNC('second', CURRENT_TIMESTAMP) AS TIMESTAMP) AS updatedAt
|
|
256
262
|
FROM {{ array }}
|
|
257
263
|
WHERE TRY_CAST(productId AS BIGINT) IS NOT NULL;
|
|
@@ -261,6 +267,7 @@ INSERT INTO {{ table }} {{ values }}
|
|
|
261
267
|
ON CONFLICT DO UPDATE SET
|
|
262
268
|
mallPid = COALESCE(excluded.mallPid, mallPid)
|
|
263
269
|
, productName = COALESCE(excluded.productName, productName)
|
|
270
|
+
, wholeCategoryName = COALESCE(excluded.wholeCategoryName, wholeCategoryName)
|
|
264
271
|
, mallName = COALESCE(excluded.mallName, mallName)
|
|
265
272
|
, brandName = COALESCE(excluded.brandName, brandName)
|
|
266
273
|
, updatedAt = excluded.updatedAt;
|
|
@@ -29,7 +29,7 @@ SELECT
|
|
|
29
29
|
, NULLIF(fmpBrand, '') AS mallName
|
|
30
30
|
, NULLIF(fmpMaker, '') AS makerName
|
|
31
31
|
, imageUrl
|
|
32
|
-
, TRY_CAST(COALESCE(lowPrice, mobileLowPrice
|
|
32
|
+
, TRY_CAST(COALESCE(lowPrice, mobileLowPrice) AS INTEGER) AS salesPrice
|
|
33
33
|
FROM {{ array }}
|
|
34
34
|
WHERE ($is_own IS NULL) OR (isOwn = $is_own);
|
|
35
35
|
|
|
@@ -40,10 +40,10 @@ INSERT INTO {{ table }} {{ values }} ON CONFLICT DO NOTHING;
|
|
|
40
40
|
-- ExposureRank: create_rank
|
|
41
41
|
CREATE OR REPLACE TABLE {{ table }} (
|
|
42
42
|
keyword VARCHAR
|
|
43
|
-
,
|
|
43
|
+
, nvMid BIGINT
|
|
44
44
|
, displayRank SMALLINT
|
|
45
45
|
, createdAt TIMESTAMP NOT NULL
|
|
46
|
-
, PRIMARY KEY (keyword,
|
|
46
|
+
, PRIMARY KEY (keyword, displayRank)
|
|
47
47
|
);
|
|
48
48
|
|
|
49
49
|
-- ExposureRank: select_rank
|
|
@@ -56,24 +56,26 @@ FROM (
|
|
|
56
56
|
TRY_CAST(REGEXP_EXTRACT(imageUrl, '^https://[^/]+/main_\d+/(\d+)', 1) AS BIGINT)
|
|
57
57
|
WHEN PREFIX(imageUrl, 'https://searchad-') THEN
|
|
58
58
|
TRY_CAST(TRY_CAST(FROM_BASE64(REGEXP_EXTRACT(imageUrl, '^https://[^/]+/[^/]+/([^.]+)', 1)) AS VARCHAR) AS BIGINT)
|
|
59
|
-
ELSE NULL END) AS
|
|
59
|
+
ELSE NULL END) AS nvMid
|
|
60
60
|
, rank AS displayRank
|
|
61
61
|
, CAST(DATE_TRUNC('second', CURRENT_TIMESTAMP) AS TIMESTAMP) AS createdAt
|
|
62
62
|
FROM {{ array }}
|
|
63
63
|
WHERE ($is_own IS NULL) OR (isOwn = $is_own)
|
|
64
64
|
) AS exposure
|
|
65
|
-
WHERE exposure.
|
|
65
|
+
WHERE exposure.nvMid IS NOT NULL;
|
|
66
66
|
|
|
67
67
|
-- ExposureRank: insert_rank
|
|
68
68
|
INSERT INTO {{ table }} {{ values }} ON CONFLICT DO NOTHING;
|
|
69
69
|
|
|
70
70
|
-- ExposureRank: create_product
|
|
71
71
|
CREATE OR REPLACE TABLE {{ table }} (
|
|
72
|
-
|
|
73
|
-
,
|
|
72
|
+
nvMid BIGINT PRIMARY KEY
|
|
73
|
+
, mallPid BIGINT
|
|
74
|
+
, productType TINYINT -- {0: "가격비교 상품", 1: "일반상품", 3: "광고상품"}
|
|
74
75
|
, productName VARCHAR
|
|
75
76
|
, wholeCategoryName VARCHAR
|
|
76
77
|
, mallName VARCHAR
|
|
78
|
+
, brandName VARCHAR
|
|
77
79
|
, salesPrice INTEGER
|
|
78
80
|
, updatedAt TIMESTAMP NOT NULL
|
|
79
81
|
);
|
|
@@ -87,12 +89,14 @@ FROM (
|
|
|
87
89
|
TRY_CAST(REGEXP_EXTRACT(imageUrl, '^https://[^/]+/main_\d+/(\d+)', 1) AS BIGINT)
|
|
88
90
|
WHEN PREFIX(imageUrl, 'https://searchad-') THEN
|
|
89
91
|
TRY_CAST(TRY_CAST(FROM_BASE64(REGEXP_EXTRACT(imageUrl, '^https://[^/]+/[^/]+/([^.]+)', 1)) AS VARCHAR) AS BIGINT)
|
|
90
|
-
ELSE NULL END) AS
|
|
91
|
-
,
|
|
92
|
+
ELSE NULL END) AS nvMid
|
|
93
|
+
, NULL AS mallPid
|
|
94
|
+
, IF(PREFIX(imageUrl, 'https://shopping-'), 0, 3) AS productType
|
|
92
95
|
, productTitle AS productName
|
|
93
96
|
, categoryNames AS wholeCategoryName
|
|
94
97
|
, NULLIF(fmpBrand, '') AS mallName
|
|
95
|
-
,
|
|
98
|
+
, NULL AS brandName
|
|
99
|
+
, TRY_CAST(COALESCE(lowPrice, mobileLowPrice) AS INTEGER) AS salesPrice
|
|
96
100
|
, CAST(DATE_TRUNC('second', CURRENT_TIMESTAMP) AS TIMESTAMP) AS updatedAt
|
|
97
101
|
FROM {{ array }}
|
|
98
102
|
WHERE ($is_own IS NULL) OR (isOwn = $is_own)
|
{linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/searchad/manage/exposure/transform.py
RENAMED
|
@@ -52,8 +52,12 @@ class ExposureRank(ExposureDiagnosis):
|
|
|
52
52
|
params: dict = dict(),
|
|
53
53
|
**kwargs
|
|
54
54
|
) -> tuple[DuckDBPyRelation,DuckDBPyRelation]:
|
|
55
|
+
def reparse_object(obj: list[dict]) -> list[dict]:
|
|
56
|
+
obj[0] = dict(obj[0], lowPrice=obj[0].get("lowPrice", None), mobileLowPrice=obj[0].get("mobileLowPrice", None))
|
|
57
|
+
return obj
|
|
55
58
|
def split_params(keyword: str, is_own: bool | None = None, **kwargs) -> tuple[dict,dict]:
|
|
56
59
|
return dict(keyword=keyword, is_own=is_own), dict(is_own=is_own)
|
|
60
|
+
obj = reparse_object(obj)
|
|
57
61
|
rank_params, product_params = split_params(**params)
|
|
58
62
|
rank = super().insert_into_table(obj, key="insert_rank", table=rank_table, values=":select_rank:", params=rank_params)
|
|
59
63
|
product = super().insert_into_table(obj, key="upsert_product", table=product_table, values=":select_product:", params=product_params)
|
|
@@ -26,9 +26,9 @@ class _SalesTransformer(DuckDBTransformer):
|
|
|
26
26
|
):
|
|
27
27
|
if isinstance(obj, dict):
|
|
28
28
|
if "error" not in obj:
|
|
29
|
+
sales = obj["data"][f"{self.sales_type}Sales"]
|
|
29
30
|
params = dict(mall_seq=mall_seq, end_date=end_date)
|
|
30
31
|
if self.start_date:
|
|
31
|
-
sales = obj["data"][f"{self.sales_type}Sales"]
|
|
32
32
|
params.update(start_date=start_date)
|
|
33
33
|
return self.insert_into_table(sales, params=params) if sales else None
|
|
34
34
|
else:
|
|
Binary file
|
{linkmerce-0.3.2/src/linkmerce/utils → linkmerce-0.3.4/src/linkmerce/extensions}/__init__.py
RENAMED
|
File without changes
|
|
@@ -0,0 +1,468 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from linkmerce.common.load import Connection, concat_sql, where
|
|
4
|
+
|
|
5
|
+
from typing import Sequence, TYPE_CHECKING
|
|
6
|
+
|
|
7
|
+
if TYPE_CHECKING:
|
|
8
|
+
from typing import Any, IO, Literal, Type, TypeVar
|
|
9
|
+
from types import TracebackType
|
|
10
|
+
JsonString = TypeVar("JsonString", str)
|
|
11
|
+
Path = TypeVar("Path", str)
|
|
12
|
+
TableId = TypeVar("TableId", str)
|
|
13
|
+
|
|
14
|
+
from google.cloud.bigquery import Client
|
|
15
|
+
from google.cloud.bigquery import SchemaField, Table
|
|
16
|
+
from google.cloud.bigquery.job import LoadJob, LoadJobConfig, QueryJob
|
|
17
|
+
from google.cloud.bigquery.table import Row, RowIterator
|
|
18
|
+
|
|
19
|
+
from linkmerce.common.load import DuckDBConnection
|
|
20
|
+
DuckDBTable = TypeVar("DuckDBTable", str)
|
|
21
|
+
BigQueryTable = TypeVar("BigQueryTable", str)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
DEFAULT_ACCOUNT = "env/service_account.json"
|
|
25
|
+
|
|
26
|
+
TEMP_TABLE = "temp_table"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class ServiceAccount(dict):
|
|
30
|
+
def __init__(self, info: JsonString | Path | dict[str,str]):
|
|
31
|
+
super().__init__(self.read_account(info))
|
|
32
|
+
|
|
33
|
+
def read_account(self, info: JsonString | Path | dict[str,str]) -> dict:
|
|
34
|
+
if isinstance(info, dict):
|
|
35
|
+
return info
|
|
36
|
+
elif isinstance(info, str):
|
|
37
|
+
import json
|
|
38
|
+
if info.startswith('{') and info.endswith('}'):
|
|
39
|
+
return json.loads(info)
|
|
40
|
+
else:
|
|
41
|
+
with open(info, 'r', encoding="utf-8") as file:
|
|
42
|
+
return json.loads(file.read())
|
|
43
|
+
else:
|
|
44
|
+
raise ValueError("Unrecognized service account.")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class PartitionOptions(dict):
|
|
48
|
+
def __init__(
|
|
49
|
+
self,
|
|
50
|
+
by: str | list[str] | None = None,
|
|
51
|
+
ascending: bool | None = True,
|
|
52
|
+
where_clause: str | None = None,
|
|
53
|
+
if_errors: Literal["ignore","raise"] = "raise",
|
|
54
|
+
**kwargs
|
|
55
|
+
):
|
|
56
|
+
super().__init__(by=by, ascending=ascending, where_clause=where_clause, if_errors=if_errors)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
###################################################################
|
|
60
|
+
######################### BigQuery Client #########################
|
|
61
|
+
###################################################################
|
|
62
|
+
|
|
63
|
+
class BigQueryClient(Connection):
|
|
64
|
+
def __init__(self, account: ServiceAccount):
|
|
65
|
+
self.set_connection(account)
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def conn(self) -> Client:
|
|
69
|
+
return self.get_connection()
|
|
70
|
+
|
|
71
|
+
def get_connection(self) -> Client:
|
|
72
|
+
return self.__conn
|
|
73
|
+
|
|
74
|
+
def set_connection(self, account: ServiceAccount):
|
|
75
|
+
from google.cloud.bigquery import Client
|
|
76
|
+
account = account if isinstance(account, ServiceAccount) else ServiceAccount(account)
|
|
77
|
+
self.__conn = Client.from_service_account_info(account, project=account["project_id"])
|
|
78
|
+
self.project_id = account["project_id"]
|
|
79
|
+
|
|
80
|
+
def close(self):
|
|
81
|
+
try:
|
|
82
|
+
self.conn.close()
|
|
83
|
+
except:
|
|
84
|
+
pass
|
|
85
|
+
|
|
86
|
+
def execute(self, query: str, **kwargs) -> QueryJob:
|
|
87
|
+
return self.conn.query(query, **kwargs)
|
|
88
|
+
|
|
89
|
+
def execute_job(self, query: str, **kwargs) -> RowIterator:
|
|
90
|
+
return self.conn.query_and_wait(query, **kwargs)
|
|
91
|
+
|
|
92
|
+
def __enter__(self) -> BigQueryClient:
|
|
93
|
+
return self
|
|
94
|
+
|
|
95
|
+
def __exit__(self, type: Type[BaseException], value: BaseException, traceback: TracebackType):
|
|
96
|
+
self.close()
|
|
97
|
+
|
|
98
|
+
############################## Fetch ##############################
|
|
99
|
+
|
|
100
|
+
def fetch_all(self, format: Literal["csv","json"], query: str) -> list[dict]:
|
|
101
|
+
try:
|
|
102
|
+
return getattr(self, f"fetch_all_to_{format}")(query)
|
|
103
|
+
except AttributeError:
|
|
104
|
+
raise ValueError("Invalid value for data format. Supported formats are: csv, json.")
|
|
105
|
+
|
|
106
|
+
def fetch_all_to_csv(self, query: str) -> list[tuple]:
|
|
107
|
+
def row_keys(row: Row) -> tuple:
|
|
108
|
+
return tuple(row.values())
|
|
109
|
+
def row_values(row: Row) -> tuple:
|
|
110
|
+
return tuple(row.values())
|
|
111
|
+
rows = self.execute_job(query)
|
|
112
|
+
return ([row_keys(rows[0])] if rows else list()) + [row_values(row) for row in rows]
|
|
113
|
+
|
|
114
|
+
def fetch_all_to_json(self, query: str) -> list[dict]:
|
|
115
|
+
def row_to_dict(row: Row) -> dict[str,Any]:
|
|
116
|
+
return dict(row.items())
|
|
117
|
+
return [row_to_dict(row) for row in self.execute_job(query)]
|
|
118
|
+
|
|
119
|
+
############################ CRUD Table ###########################
|
|
120
|
+
|
|
121
|
+
def create_table(
|
|
122
|
+
self,
|
|
123
|
+
table: str,
|
|
124
|
+
schema: TableId | Sequence[dict | SchemaField],
|
|
125
|
+
exists_ok: bool = True,
|
|
126
|
+
**kwargs
|
|
127
|
+
) -> Table:
|
|
128
|
+
table_ref = self.ref_table(table, schema)
|
|
129
|
+
return self.conn.create_table(table_ref, exists_ok=exists_ok, **kwargs)
|
|
130
|
+
|
|
131
|
+
def copy_table(
|
|
132
|
+
self,
|
|
133
|
+
source_table: str,
|
|
134
|
+
target_table: str,
|
|
135
|
+
where_clause: str | None = None,
|
|
136
|
+
limit: int | None = 0,
|
|
137
|
+
option: Literal["replace","ignore"] | None = None,
|
|
138
|
+
) -> RowIterator:
|
|
139
|
+
select = f"SELECT * FROM `{self.project_id}.{source_table}`"
|
|
140
|
+
limit_ = f"LIMIT {limit}" if isinstance(limit, int) else None
|
|
141
|
+
query = concat_sql(f"{self.expr_create(option)} `{self.project_id}.{target_table}` AS", select, where(where_clause), limit_)
|
|
142
|
+
return self.execute_job(query)
|
|
143
|
+
|
|
144
|
+
def delete_table(self, table: str, where: str = "TRUE") -> RowIterator:
|
|
145
|
+
return self.execute_job(f"DELETE FROM `{self.project_id}.{table}` WHERE {where};")
|
|
146
|
+
|
|
147
|
+
def select_table_to_json(self, table: str) -> list[dict]:
|
|
148
|
+
return self.fetch_all_to_json(f"SELECT * FROM `{self.project_id}.{table}`;")
|
|
149
|
+
|
|
150
|
+
def table_exists(self, table: str) -> bool:
|
|
151
|
+
from google.api_core.exceptions import NotFound
|
|
152
|
+
try:
|
|
153
|
+
self.conn.get_table(f"{self.project_id}.{table}")
|
|
154
|
+
return True
|
|
155
|
+
except NotFound:
|
|
156
|
+
return False
|
|
157
|
+
|
|
158
|
+
def table_has_rows(self, table: str, where_clause: str | None = None) -> bool:
|
|
159
|
+
if self.table_exists(table):
|
|
160
|
+
query = concat_sql(f"SELECT COUNT(*) FROM `{self.project_id}.{table}`", where(where_clause))
|
|
161
|
+
rows = list(self.execute_job(query))
|
|
162
|
+
return bool(list(rows[0].values())[0])
|
|
163
|
+
return False
|
|
164
|
+
|
|
165
|
+
def get_table(self, table: str) -> Table:
|
|
166
|
+
return self.conn.get_table(f"{self.project_id}.{table}")
|
|
167
|
+
|
|
168
|
+
def ref_table(self, table: str, schema: TableId | Sequence[dict | SchemaField]) -> Table:
|
|
169
|
+
from google.cloud.bigquery import Table
|
|
170
|
+
if isinstance(schema, str):
|
|
171
|
+
schema = self.get_schema(schema)
|
|
172
|
+
if not isinstance(schema, Sequence):
|
|
173
|
+
raise ValueError("Invalid schema: expected sequence of schema fields.")
|
|
174
|
+
return Table(table, schema)
|
|
175
|
+
|
|
176
|
+
def get_schema(self, table: str) -> list[SchemaField]:
|
|
177
|
+
return self.conn.get_table(f"{self.project_id}.{table}").schema
|
|
178
|
+
|
|
179
|
+
############################# Load Job ############################
|
|
180
|
+
|
|
181
|
+
def load_table_from_json(
|
|
182
|
+
self,
|
|
183
|
+
table: str,
|
|
184
|
+
values: list[dict] | str | Path,
|
|
185
|
+
serialize: bool = True,
|
|
186
|
+
schema: Literal["auto"] | TableId | Sequence[dict | SchemaField] = "auto",
|
|
187
|
+
write: Literal["append","empty","truncate","truncate_data"] = "append",
|
|
188
|
+
if_not_found: Literal["create","errors","ignore"] = "errors",
|
|
189
|
+
) -> LoadJob:
|
|
190
|
+
schema = self._auto_detect_schema(table, schema)
|
|
191
|
+
if self._find_table(table, schema, if_not_found):
|
|
192
|
+
if serialize:
|
|
193
|
+
import json
|
|
194
|
+
values = json.loads(json.dumps(values, ensure_ascii=False, default=str))
|
|
195
|
+
job_config = self.build_load_job_config(schema, write_disposition=write.upper())
|
|
196
|
+
return self.conn.load_table_from_json(values, f"{self.project_id}.{table}", job_config=job_config).result()
|
|
197
|
+
|
|
198
|
+
def load_table_from_file(
|
|
199
|
+
self,
|
|
200
|
+
table: str,
|
|
201
|
+
file_obj: IO[bytes],
|
|
202
|
+
format: Literal["avgo","csv","json","orc","parquet"],
|
|
203
|
+
schema: Literal["auto"] | TableId | Sequence[dict | SchemaField] = "auto",
|
|
204
|
+
write: Literal["append","empty","truncate","truncate_data"] = "append",
|
|
205
|
+
if_not_found: Literal["create","errors","ignore"] = "errors",
|
|
206
|
+
) -> LoadJob:
|
|
207
|
+
schema = self._auto_detect_schema(table, schema)
|
|
208
|
+
if self._find_table(table, schema, if_not_found):
|
|
209
|
+
job_config = self.build_load_job_config(schema, source_format=format.upper(), write_disposition=write.upper())
|
|
210
|
+
return self.conn.load_table_from_file(file_obj, f"{self.project_id}.{table}", job_config=job_config).result()
|
|
211
|
+
|
|
212
|
+
def build_load_job_config(
|
|
213
|
+
self,
|
|
214
|
+
schema: Sequence[dict | SchemaField] | None = None,
|
|
215
|
+
source_format: Literal["AVRO", "CSV", "JSON", "ORC", "PARQUET"] | None = None,
|
|
216
|
+
write_disposition: Literal["APPEND", "EMPTY", "TRUNCATE", "TRUNCATE_DATA"] | None = None,
|
|
217
|
+
**kwargs
|
|
218
|
+
) -> LoadJobConfig:
|
|
219
|
+
from google.cloud.bigquery.job import LoadJobConfig
|
|
220
|
+
if schema is not None:
|
|
221
|
+
kwargs["schema"] = self.build_schema(schema)
|
|
222
|
+
if source_format is not None:
|
|
223
|
+
kwargs["source_format"] = "NEWLINE_DELIMITED_JSON" if source_format == "JSON" else source_format
|
|
224
|
+
if write_disposition is not None:
|
|
225
|
+
kwargs["write_disposition"] = f"WRITE_{write_disposition}"
|
|
226
|
+
return LoadJobConfig(**kwargs)
|
|
227
|
+
|
|
228
|
+
def build_schema(self, fields: Sequence[dict | SchemaField]) -> list[SchemaField]:
|
|
229
|
+
from google.cloud.bigquery import SchemaField
|
|
230
|
+
def build(type: str | None = None, **kwargs) -> SchemaField:
|
|
231
|
+
if type is not None:
|
|
232
|
+
kwargs["field_type"] = type
|
|
233
|
+
return SchemaField(**kwargs)
|
|
234
|
+
return [field if isinstance(field, SchemaField) else build(**field) for field in fields]
|
|
235
|
+
|
|
236
|
+
def _auto_detect_schema(
|
|
237
|
+
self,
|
|
238
|
+
table: str,
|
|
239
|
+
schema: Literal["auto"] | TableId | Sequence[dict | SchemaField] = "auto",
|
|
240
|
+
) -> list[dict | SchemaField]:
|
|
241
|
+
if isinstance(schema, str):
|
|
242
|
+
return self.get_schema(table if schema == "auto" else schema)
|
|
243
|
+
else:
|
|
244
|
+
return schema
|
|
245
|
+
|
|
246
|
+
def _find_table(
|
|
247
|
+
self,
|
|
248
|
+
table: str,
|
|
249
|
+
schema: list[dict | SchemaField],
|
|
250
|
+
if_not_found: Literal["create","errors","ignore"] = "errors",
|
|
251
|
+
) -> bool:
|
|
252
|
+
if if_not_found == "create":
|
|
253
|
+
if not self.table_exists(table):
|
|
254
|
+
self.create_table(table, schema)
|
|
255
|
+
elif if_not_found == "ignore":
|
|
256
|
+
return self.table_exists(table)
|
|
257
|
+
return True
|
|
258
|
+
|
|
259
|
+
############################## Merge ##############################
|
|
260
|
+
|
|
261
|
+
def merge_into_table(
|
|
262
|
+
self,
|
|
263
|
+
source_table: str,
|
|
264
|
+
target_table: str,
|
|
265
|
+
on: str | Sequence[str],
|
|
266
|
+
matched: str | dict[str,Literal["source_first","target_first","greatest","least","replace","ignore"]],
|
|
267
|
+
not_matched: str | Sequence[str],
|
|
268
|
+
where_clause: str | None = None,
|
|
269
|
+
) -> LoadJob:
|
|
270
|
+
where = [f"T.{where_clause}"] if where_clause else list()
|
|
271
|
+
on = " AND ".join([f"T.{col} = S.{col}" for col in ([on] if isinstance(on, str) else on)]+where)
|
|
272
|
+
query = f"MERGE INTO `{self.project_id}.{target_table}` AS T USING `{self.project_id}.{source_table}` AS S ON {on}"
|
|
273
|
+
query = concat_sql(query, self._merge_update(matched), self._merge_insert(not_matched))
|
|
274
|
+
self.execute_job(query)
|
|
275
|
+
|
|
276
|
+
def _merge_update(self, matched: str | dict[str,Literal["count","sum","avg","min","max","first","last","list"]]) -> str:
|
|
277
|
+
prefix = "WHEN MATCHED THEN UPDATE SET "
|
|
278
|
+
if isinstance(matched, dict):
|
|
279
|
+
def render(col: str, agg: str) -> str:
|
|
280
|
+
if agg in {"source_first","target_first"}:
|
|
281
|
+
kwargs = dict(zip(["left","right"], ('S','T') if agg == "source_first" else ('T','S')))
|
|
282
|
+
return "COALESCE({left}.{col}, {right}.{col})".format(col=col, **kwargs)
|
|
283
|
+
elif agg in {"greatest","least"}:
|
|
284
|
+
return f"{agg.upper()}(S.{col}, T.{col})"
|
|
285
|
+
elif agg in {"replace","ignore"}:
|
|
286
|
+
return f"S.{agg}" if agg == "replace" else f"T.{agg}"
|
|
287
|
+
else:
|
|
288
|
+
return f"{agg}({col})"
|
|
289
|
+
return prefix + ", ".join([f"T.{col} = {render(col, agg)}" for col, agg in matched.items()])
|
|
290
|
+
else:
|
|
291
|
+
return prefix + str(matched)
|
|
292
|
+
|
|
293
|
+
def _merge_insert(self, not_matched: Sequence[str]) -> str:
|
|
294
|
+
prefix = "WHEN NOT MATCHED THEN "
|
|
295
|
+
if (not isinstance(not_matched, str)) and isinstance(not_matched, Sequence):
|
|
296
|
+
return prefix + "INSERT ({}) VALUES ({})".format(", ".join(not_matched), "S."+", S.".join(not_matched))
|
|
297
|
+
else:
|
|
298
|
+
return prefix + not_matched
|
|
299
|
+
|
|
300
|
+
########################## Load and Merge #########################
|
|
301
|
+
|
|
302
|
+
def merge_into_table_from_file(
|
|
303
|
+
self,
|
|
304
|
+
stage_table: str,
|
|
305
|
+
target_table: str,
|
|
306
|
+
source_file: IO[bytes],
|
|
307
|
+
source_format: Literal["avgo","csv","json","orc","parquet"],
|
|
308
|
+
on: str | Sequence[str],
|
|
309
|
+
matched: str | dict[str,Literal["source_first","target_first","greatest","least","replace","ignore"]],
|
|
310
|
+
not_matched: str | Sequence[str],
|
|
311
|
+
where_clause: str | None = None,
|
|
312
|
+
schema: Literal["auto"] | TableId | Sequence[dict | SchemaField] = "auto",
|
|
313
|
+
write: Literal["append","empty","truncate","truncate_data"] = "truncate",
|
|
314
|
+
if_not_found: Literal["create","errors","ignore"] = "errors",
|
|
315
|
+
table_lock_wait_interval: float | int | None = 1.,
|
|
316
|
+
table_lock_wait_timeout: float | int | None = 60.,
|
|
317
|
+
drop_stage_after_merge: bool = True,
|
|
318
|
+
) -> LoadJob:
|
|
319
|
+
try:
|
|
320
|
+
self._wait_until_table_not_found(stage_table, table_lock_wait_interval, table_lock_wait_timeout)
|
|
321
|
+
self.load_table_from_file(source_file, stage_table, source_format, schema, write, if_not_found)
|
|
322
|
+
return self.merge_into_table(stage_table, target_table, on, matched, not_matched, where_clause)
|
|
323
|
+
finally:
|
|
324
|
+
if drop_stage_after_merge and self.table_exists(stage_table):
|
|
325
|
+
self.execute_job(f"DROP TABLE `{self.project_id}.{stage_table}`")
|
|
326
|
+
|
|
327
|
+
def _wait_until_table_not_found(self, table: str, interval: float | int | None = 1., timeout: float | int | None = 60.):
|
|
328
|
+
has_interval, has_timeout = isinstance(interval, (float,int)), isinstance(timeout, (float,int))
|
|
329
|
+
if has_interval:
|
|
330
|
+
import time
|
|
331
|
+
total = 0
|
|
332
|
+
while self.table_exists(table):
|
|
333
|
+
if has_timeout and (total > timeout):
|
|
334
|
+
raise TimeoutError("Timed out waiting until the table does not exist.")
|
|
335
|
+
total += interval
|
|
336
|
+
time.sleep(interval)
|
|
337
|
+
|
|
338
|
+
############################ Expression ###########################
|
|
339
|
+
|
|
340
|
+
def expr_cast(self, value: Any | None, type: str, alias: str = str(), safe: bool = False) -> str:
|
|
341
|
+
cast = "SAFE_CAST" if safe else "CAST"
|
|
342
|
+
alias = f" AS {alias}" if alias else str()
|
|
343
|
+
return f"{cast}({self.expr_value(value)} AS {type.upper()})" + alias
|
|
344
|
+
|
|
345
|
+
def expr_interval(expr: str, days: int | None = None, time: bool = True) -> str:
|
|
346
|
+
if isinstance(days, int):
|
|
347
|
+
func = ("DATE{}_SUB" if days < 0 else "DATE{}_ADD").format("TIME" if time else str())
|
|
348
|
+
return f"{func}({expr}, INTERVAL {abs(days)} DAY)"
|
|
349
|
+
else:
|
|
350
|
+
return expr
|
|
351
|
+
|
|
352
|
+
def expr_now(
|
|
353
|
+
self,
|
|
354
|
+
type: Literal["DATETIME","STRING"] = "DATETIME",
|
|
355
|
+
format: str | None = "%Y-%m-%d %H:%M:%S",
|
|
356
|
+
interval: str | int | None = None,
|
|
357
|
+
tzinfo: str | None = None,
|
|
358
|
+
) -> str:
|
|
359
|
+
expr = "CURRENT_DATETIME({})".format(f"'{tzinfo}'" if tzinfo else str())
|
|
360
|
+
expr = self.expr_interval(expr, interval, time=True)
|
|
361
|
+
if format:
|
|
362
|
+
expr = f"FORMAT_DATE('{format}', {expr})"
|
|
363
|
+
if type.upper() == "DATETIME":
|
|
364
|
+
return f"CAST({expr} AS DATETIME)"
|
|
365
|
+
return expr if type.upper() == "DATETIME" else "NULL"
|
|
366
|
+
|
|
367
|
+
def expr_today(
|
|
368
|
+
self,
|
|
369
|
+
type: Literal["DATE","STRING"] = "DATE",
|
|
370
|
+
format: str | None = "%Y-%m-%d",
|
|
371
|
+
interval: str | int | None = None,
|
|
372
|
+
tzinfo: str | None = None,
|
|
373
|
+
) -> str:
|
|
374
|
+
expr = "CURRENT_DATE({})".format(f"'{tzinfo}'" if tzinfo else str())
|
|
375
|
+
expr = self.expr_interval(expr, interval, time=False)
|
|
376
|
+
if (type.upper() == "STRING") and format:
|
|
377
|
+
return f"FORMAT_DATE('{format}', {expr})"
|
|
378
|
+
return expr if type.upper() == "DATE" else "NULL"
|
|
379
|
+
|
|
380
|
+
############################## DuckDB #############################
|
|
381
|
+
|
|
382
|
+
def load_table_from_duckdb(
|
|
383
|
+
self,
|
|
384
|
+
connection: DuckDBConnection,
|
|
385
|
+
source_table: DuckDBTable,
|
|
386
|
+
target_table: BigQueryTable,
|
|
387
|
+
partition_by: PartitionOptions = dict(),
|
|
388
|
+
schema: Literal["auto"] | TableId | Sequence[dict | SchemaField] = "auto",
|
|
389
|
+
write: Literal["append","empty","truncate","truncate_data"] = "append",
|
|
390
|
+
if_not_found: Literal["create","errors","ignore"] = "errors",
|
|
391
|
+
progress: bool = True,
|
|
392
|
+
) -> bool:
|
|
393
|
+
from linkmerce.common.load import DuckDBIterator
|
|
394
|
+
from linkmerce.utils.tqdm import import_tqdm
|
|
395
|
+
from io import BytesIO
|
|
396
|
+
|
|
397
|
+
if not connection.table_exists(source_table):
|
|
398
|
+
return True
|
|
399
|
+
schema = self._auto_detect_schema(target_table, schema)
|
|
400
|
+
|
|
401
|
+
iterator = DuckDBIterator(connection, format="parquet").from_table(source_table)
|
|
402
|
+
if partition_by:
|
|
403
|
+
iterator = iterator.partition_by(**partition_by)
|
|
404
|
+
|
|
405
|
+
tqdm = import_tqdm()
|
|
406
|
+
for bytes_ in tqdm(iterator, desc=f"Uploading data to '{target_table}'", disable=(not progress)):
|
|
407
|
+
self.load_table_from_file(target_table, BytesIO(bytes_), "parquet", schema, write, if_not_found)
|
|
408
|
+
return True
|
|
409
|
+
|
|
410
|
+
def overwrite_table_from_duckdb(
|
|
411
|
+
self,
|
|
412
|
+
connection: DuckDBConnection,
|
|
413
|
+
source_table: DuckDBTable,
|
|
414
|
+
target_table: BigQueryTable,
|
|
415
|
+
where_clause: str = "TRUE",
|
|
416
|
+
partition_by: PartitionOptions = dict(),
|
|
417
|
+
schema: Literal["auto"] | TableId | Sequence[dict | SchemaField] = "auto",
|
|
418
|
+
if_not_found: Literal["create","errors","ignore"] = "errors",
|
|
419
|
+
progress: bool = True,
|
|
420
|
+
backup_table: DuckDBTable | None = TEMP_TABLE,
|
|
421
|
+
) -> bool:
|
|
422
|
+
if not connection.table_exists(source_table):
|
|
423
|
+
return True
|
|
424
|
+
elif not self.table_has_rows(target_table, where_clause):
|
|
425
|
+
return self.load_table_from_duckdb(connection, source_table, target_table, partition_by, schema, "append", if_not_found, progress)
|
|
426
|
+
|
|
427
|
+
success = False
|
|
428
|
+
from_clause = concat_sql(f"FROM `{self.project_id}.{target_table}`", where(where_clause))
|
|
429
|
+
existing_values = self.fetch_all_to_json(f"SELECT * {from_clause};")
|
|
430
|
+
self.conn.query(f"DELETE {from_clause};")
|
|
431
|
+
|
|
432
|
+
try:
|
|
433
|
+
success = self.load_table_from_duckdb(connection, source_table, target_table, partition_by, schema, "append", if_not_found, progress)
|
|
434
|
+
return success
|
|
435
|
+
finally:
|
|
436
|
+
if (not success) and (backup_table is not None) and existing_values:
|
|
437
|
+
connection.copy_table(source_table, backup_table, option="replace", temp=True)
|
|
438
|
+
connection.insert_into_table_from_json(backup_table, existing_values)
|
|
439
|
+
|
|
440
|
+
def merge_into_table_from_duckdb(
|
|
441
|
+
self,
|
|
442
|
+
connection: DuckDBConnection,
|
|
443
|
+
source_table: DuckDBTable,
|
|
444
|
+
staging_table: BigQueryTable,
|
|
445
|
+
target_table: BigQueryTable,
|
|
446
|
+
on: str | Sequence[str],
|
|
447
|
+
matched: str | dict[str,Literal["source_first","target_first","greatest","least","replace","ignore"]],
|
|
448
|
+
not_matched: str | Sequence[str],
|
|
449
|
+
schema: Literal["auto"] | TableId | Sequence[dict | SchemaField] = "auto",
|
|
450
|
+
where_clause: str | None = None,
|
|
451
|
+
table_lock_wait_interval: float | int | None = 1.,
|
|
452
|
+
table_lock_wait_timeout: float | int | None = 60.,
|
|
453
|
+
drop_stage_after_merge: bool = True,
|
|
454
|
+
) -> bool:
|
|
455
|
+
if not connection.table_exists(source_table):
|
|
456
|
+
return True
|
|
457
|
+
elif not self.table_has_rows(target_table, where_clause):
|
|
458
|
+
return self.load_table_from_duckdb(connection, source_table, target_table, schema=schema, write="append")
|
|
459
|
+
|
|
460
|
+
try:
|
|
461
|
+
self._wait_until_table_not_found(staging_table, table_lock_wait_interval, table_lock_wait_timeout)
|
|
462
|
+
self.copy_table(target_table, staging_table, option="replace")
|
|
463
|
+
self.load_table_from_duckdb(connection, source_table, staging_table, schema=schema, write="append")
|
|
464
|
+
self.merge_into_table(staging_table, target_table, on, matched, not_matched, where_clause)
|
|
465
|
+
return True
|
|
466
|
+
finally:
|
|
467
|
+
if drop_stage_after_merge and self.table_exists(staging_table):
|
|
468
|
+
self.execute_job(f"DROP TABLE `{self.project_id}.{staging_table}`")
|
|
File without changes
|
|
@@ -4,13 +4,31 @@ from typing import Sequence, TYPE_CHECKING
|
|
|
4
4
|
|
|
5
5
|
if TYPE_CHECKING:
|
|
6
6
|
from typing import Any, Callable, Coroutine, Hashable, Iterable, TypeVar
|
|
7
|
+
from types import ModuleType
|
|
7
8
|
from numbers import Real
|
|
8
9
|
_KT = TypeVar("_KT", Hashable)
|
|
9
10
|
_VT = TypeVar("_VT", Any)
|
|
10
11
|
|
|
11
12
|
|
|
13
|
+
def import_tqdm() -> ModuleType:
|
|
14
|
+
try:
|
|
15
|
+
from tqdm import tqdm
|
|
16
|
+
return tqdm
|
|
17
|
+
except:
|
|
18
|
+
return lambda x, **kwargs: x
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def import_tqdm_asyncio() -> ModuleType:
|
|
22
|
+
try:
|
|
23
|
+
from tqdm.asyncio import tqdm_asyncio
|
|
24
|
+
return tqdm_asyncio
|
|
25
|
+
except:
|
|
26
|
+
import asyncio
|
|
27
|
+
return asyncio
|
|
28
|
+
|
|
29
|
+
|
|
12
30
|
###################################################################
|
|
13
|
-
##############################
|
|
31
|
+
############################## Gather #############################
|
|
14
32
|
###################################################################
|
|
15
33
|
|
|
16
34
|
def gather(
|
|
@@ -20,12 +38,13 @@ def gather(
|
|
|
20
38
|
delay: Real | Sequence[Real,Real] = 0.,
|
|
21
39
|
tqdm_options: dict = dict(),
|
|
22
40
|
) -> list:
|
|
41
|
+
import time
|
|
23
42
|
try:
|
|
24
43
|
from tqdm import tqdm
|
|
25
44
|
except:
|
|
26
45
|
tqdm = lambda x: x
|
|
27
46
|
tqdm_options = dict()
|
|
28
|
-
|
|
47
|
+
|
|
29
48
|
def run_with_delay(args: tuple[_VT,...] | dict[_KT,_VT]) -> Any:
|
|
30
49
|
try:
|
|
31
50
|
if isinstance(args, dict):
|
|
@@ -51,6 +70,7 @@ async def gather_async(
|
|
|
51
70
|
except:
|
|
52
71
|
tqdm_asyncio = asyncio
|
|
53
72
|
tqdm_options = dict()
|
|
73
|
+
|
|
54
74
|
async def run_with_delay(args: tuple[_VT,...] | dict[_KT,_VT]) -> Any:
|
|
55
75
|
try:
|
|
56
76
|
if isinstance(args, dict):
|
|
@@ -81,7 +101,7 @@ def _get_seconds(value: Real | Sequence[Real,Real]) -> Real:
|
|
|
81
101
|
|
|
82
102
|
|
|
83
103
|
###################################################################
|
|
84
|
-
##############################
|
|
104
|
+
############################## Expand #############################
|
|
85
105
|
###################################################################
|
|
86
106
|
|
|
87
107
|
def expand(
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{linkmerce-0.3.2 → linkmerce-0.3.4}/src/linkmerce/core/smartstore/brand/catalog/transform.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|