datahub-bq-connector-session-patch 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- datahub_bq_connector_session_patch/__init__.py +36 -0
- datahub_bq_connector_session_patch/patch.py +201 -0
- datahub_bq_connector_session_patch/resolver.py +298 -0
- datahub_bq_connector_session_patch/source.py +100 -0
- datahub_bq_connector_session_patch-0.1.0.dist-info/METADATA +287 -0
- datahub_bq_connector_session_patch-0.1.0.dist-info/RECORD +9 -0
- datahub_bq_connector_session_patch-0.1.0.dist-info/WHEEL +4 -0
- datahub_bq_connector_session_patch-0.1.0.dist-info/entry_points.txt +2 -0
- datahub_bq_connector_session_patch-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Recover unqualified BigQuery reads that DataHub silently discards.
|
|
2
|
+
|
|
3
|
+
Full diagnosis, design and verification live in the DataVantage docs repo:
|
|
4
|
+
``docs/datahub-unqualified-read-attribution.md``.
|
|
5
|
+
|
|
6
|
+
Recipes use the ``datahub_bq_connector_session_patch`` source type, registered by this package's
|
|
7
|
+
entry point. ``install()`` and ``install_on_aggregator()`` are for standalone scripts
|
|
8
|
+
and tests that drive a pipeline or an aggregator directly.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from datahub_bq_connector_session_patch.patch import armed, check_datahub_version, install, is_armed
|
|
12
|
+
from datahub_bq_connector_session_patch.resolver import (
|
|
13
|
+
SessionPatchStats,
|
|
14
|
+
build_index,
|
|
15
|
+
install_on_aggregator,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"SessionPatchStats",
|
|
20
|
+
"armed",
|
|
21
|
+
"build_index",
|
|
22
|
+
"check_datahub_version",
|
|
23
|
+
"install",
|
|
24
|
+
"install_on_aggregator",
|
|
25
|
+
"is_armed",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def __getattr__(name: str):
|
|
30
|
+
# Imported lazily: it pulls in datahub's bigquery extras, which a test or a script
|
|
31
|
+
# using only the resolver need not have.
|
|
32
|
+
if name == "BigQuerySessionPatchSource":
|
|
33
|
+
from datahub_bq_connector_session_patch.source import BigQuerySessionPatchSource
|
|
34
|
+
|
|
35
|
+
return BigQuerySessionPatchSource
|
|
36
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""Compose the resolver onto ``BigQueryQueriesExtractor``.
|
|
2
|
+
|
|
3
|
+
The extractor is constructed deep inside ``BigqueryV2Source.get_workunits_internal``,
|
|
4
|
+
so there is no injection point -- the composition has to happen on the class. To keep a
|
|
5
|
+
stock ``type: bigquery`` recipe stock even when this package is installed in the same
|
|
6
|
+
interpreter, the patched ``__init__`` is inert unless it is explicitly armed. The
|
|
7
|
+
``datahub_bq_connector_session_patch`` source arms it around its own run and nothing else.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import contextlib
|
|
11
|
+
import contextvars
|
|
12
|
+
import inspect
|
|
13
|
+
import logging
|
|
14
|
+
from typing import Any, Collection, Iterator, Optional, Tuple
|
|
15
|
+
|
|
16
|
+
from datahub_bq_connector_session_patch.resolver import SessionPatchStats, install_on_aggregator
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
#: Versions of acryl-datahub this patch has been tested against. It reaches into private
|
|
21
|
+
#: internals (``BigQueryQueriesExtractor.__init__``, ``SqlParsingAggregator``), so a
|
|
22
|
+
#: version outside this range is reported rather than trusted. Deliberately NOT a
|
|
23
|
+
#: packaging dependency: this wheel installs into a DataHub executor's own interpreter
|
|
24
|
+
#: beside its pinned acryl-datahub, and declaring it would let pip re-resolve that pin.
|
|
25
|
+
MIN_ACRYL_DATAHUB = (1, 6)
|
|
26
|
+
MAX_ACRYL_DATAHUB_EXCLUSIVE = (1, 8)
|
|
27
|
+
|
|
28
|
+
_ARMED: contextvars.ContextVar = contextvars.ContextVar("datahub_bq_connector_session_patch_armed", default=False)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@contextlib.contextmanager
|
|
32
|
+
def armed() -> Iterator[None]:
|
|
33
|
+
"""Arm the patch for the duration of the block."""
|
|
34
|
+
token = _ARMED.set(True)
|
|
35
|
+
try:
|
|
36
|
+
yield
|
|
37
|
+
finally:
|
|
38
|
+
_ARMED.reset(token)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def is_armed() -> bool:
|
|
42
|
+
return _ARMED.get()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _version_tuple(raw: str) -> tuple:
|
|
46
|
+
parts = []
|
|
47
|
+
for chunk in raw.split(".")[:3]:
|
|
48
|
+
digits = "".join(c for c in chunk if c.isdigit())
|
|
49
|
+
if not digits:
|
|
50
|
+
break
|
|
51
|
+
parts.append(int(digits))
|
|
52
|
+
return tuple(parts)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def check_datahub_version(report: Optional[Any] = None) -> Optional[str]:
|
|
56
|
+
"""Warn if the installed acryl-datahub is outside the tested range.
|
|
57
|
+
|
|
58
|
+
Returns the message when out of range, else None. Never raises: a version check
|
|
59
|
+
failing closed would take down an ingestion run over its own metadata lookup.
|
|
60
|
+
"""
|
|
61
|
+
try:
|
|
62
|
+
from importlib.metadata import version
|
|
63
|
+
|
|
64
|
+
raw = version("acryl-datahub")
|
|
65
|
+
except Exception:
|
|
66
|
+
return None
|
|
67
|
+
|
|
68
|
+
found = _version_tuple(raw)
|
|
69
|
+
if not found:
|
|
70
|
+
return None
|
|
71
|
+
if MIN_ACRYL_DATAHUB <= found[: len(MIN_ACRYL_DATAHUB)] and found < MAX_ACRYL_DATAHUB_EXCLUSIVE:
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
lo = ".".join(str(n) for n in MIN_ACRYL_DATAHUB)
|
|
75
|
+
hi = ".".join(str(n) for n in MAX_ACRYL_DATAHUB_EXCLUSIVE)
|
|
76
|
+
message = (
|
|
77
|
+
f"datahub-bq-connector-session-patch is tested against acryl-datahub >={lo},<{hi}; found {raw}. "
|
|
78
|
+
"It patches private internals -- re-run the test suite before trusting this run."
|
|
79
|
+
)
|
|
80
|
+
if report is not None:
|
|
81
|
+
report.warning("datahub-bq-connector-session-patch: untested acryl-datahub version", context=message)
|
|
82
|
+
else:
|
|
83
|
+
logger.warning(message)
|
|
84
|
+
return message
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def describe_outcome(stats) -> Tuple[str, str, str]:
|
|
88
|
+
"""(level, title, context) for the end-of-run report. Pure, so it can be tested.
|
|
89
|
+
|
|
90
|
+
The index is built lazily on the first classified upstream, so a run that saw no
|
|
91
|
+
queries at all leaves ``index_size`` at 0 as well. Reporting that as an empty
|
|
92
|
+
``discovered_tables`` points the operator at the wrong subsystem.
|
|
93
|
+
"""
|
|
94
|
+
queries_seen = stats.queries_passed + stats.queries_skipped + stats.queries_mixed
|
|
95
|
+
if queries_seen == 0:
|
|
96
|
+
return (
|
|
97
|
+
"warning",
|
|
98
|
+
"datahub-bq-connector-session-patch saw NO queries",
|
|
99
|
+
"the query log yielded nothing, so the resolution index was never built -- "
|
|
100
|
+
"check the audit window, the user filters and the region qualifiers, not "
|
|
101
|
+
"discovered_tables",
|
|
102
|
+
)
|
|
103
|
+
if stats.index_size == 0:
|
|
104
|
+
return (
|
|
105
|
+
"warning",
|
|
106
|
+
"datahub-bq-connector-session-patch built an EMPTY resolution index",
|
|
107
|
+
f"{queries_seen} queries seen but discovered_tables was empty -- "
|
|
108
|
+
"the patch recovered nothing",
|
|
109
|
+
)
|
|
110
|
+
if stats.rewritten == 0:
|
|
111
|
+
return (
|
|
112
|
+
"warning",
|
|
113
|
+
"datahub-bq-connector-session-patch rewrote nothing",
|
|
114
|
+
f"index={stats.index_size} names but no _SESSION upstreams matched",
|
|
115
|
+
)
|
|
116
|
+
return (
|
|
117
|
+
"info",
|
|
118
|
+
"datahub-bq-connector-session-patch recovered unqualified reads",
|
|
119
|
+
str(stats),
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _live_discovered_tables(extractor, original_init, args: tuple, kwargs: dict) -> Collection[str]:
|
|
124
|
+
"""The LIVE collection the source passed in, not ``__init__``'s snapshot.
|
|
125
|
+
|
|
126
|
+
``BigqueryV2Source`` passes ``discovered_tables=self.bq_schema_extractor.table_refs``
|
|
127
|
+
-- a set that is still empty at construction and fills during the schema pass. The
|
|
128
|
+
extractor copies it into ``self.discovered_tables`` immediately, so reading that
|
|
129
|
+
attribute yields an empty index and a silently useless run.
|
|
130
|
+
|
|
131
|
+
Bound through the real signature rather than a hand-counted argument index, so a
|
|
132
|
+
positional call site or an upstream signature change cannot silently pick the wrong
|
|
133
|
+
parameter -- the failure mode this whole patch exists to avoid.
|
|
134
|
+
"""
|
|
135
|
+
live = None
|
|
136
|
+
try:
|
|
137
|
+
bound = inspect.signature(original_init).bind(extractor, *args, **kwargs)
|
|
138
|
+
live = bound.arguments.get("discovered_tables")
|
|
139
|
+
except TypeError:
|
|
140
|
+
live = kwargs.get("discovered_tables")
|
|
141
|
+
if live is None:
|
|
142
|
+
live = extractor.discovered_tables or ()
|
|
143
|
+
return live
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def install(always_on: bool = False) -> None:
|
|
147
|
+
"""Patch ``BigQueryQueriesExtractor`` so armed runs get the fix. Idempotent.
|
|
148
|
+
|
|
149
|
+
``always_on`` arms the patch process-wide -- for standalone scripts that drive a
|
|
150
|
+
pipeline directly. Recipes should use the ``datahub_bq_connector_session_patch`` source instead,
|
|
151
|
+
which arms only its own run.
|
|
152
|
+
"""
|
|
153
|
+
from datahub.ingestion.source.bigquery_v2 import queries_extractor as qe
|
|
154
|
+
|
|
155
|
+
if always_on:
|
|
156
|
+
_ARMED.set(True)
|
|
157
|
+
|
|
158
|
+
if getattr(qe.BigQueryQueriesExtractor, "_datahub_bq_connector_session_patch_installed", False):
|
|
159
|
+
return
|
|
160
|
+
original_init = qe.BigQueryQueriesExtractor.__init__
|
|
161
|
+
|
|
162
|
+
def __init__(self, *args, **kwargs):
|
|
163
|
+
original_init(self, *args, **kwargs)
|
|
164
|
+
if not is_armed():
|
|
165
|
+
return
|
|
166
|
+
|
|
167
|
+
check_datahub_version(self.structured_report)
|
|
168
|
+
live = _live_discovered_tables(self, original_init, args, kwargs)
|
|
169
|
+
|
|
170
|
+
def discovered() -> Collection[str]:
|
|
171
|
+
standardize = self.identifiers.standardize_identifier_case
|
|
172
|
+
return {standardize(name) for name in (live or ())}
|
|
173
|
+
|
|
174
|
+
stats: SessionPatchStats = install_on_aggregator(self.aggregator, discovered)
|
|
175
|
+
self.datahub_bq_connector_session_patch_stats = stats
|
|
176
|
+
logger.info("datahub-bq-connector-session-patch: armed, index builds on first query")
|
|
177
|
+
|
|
178
|
+
original_close = self.close
|
|
179
|
+
|
|
180
|
+
def close() -> None:
|
|
181
|
+
# Both historical failures of this patch were silent: the run completed and
|
|
182
|
+
# rewrote nothing. Make that loud instead -- but name the right subsystem.
|
|
183
|
+
try:
|
|
184
|
+
level, title, context = describe_outcome(stats)
|
|
185
|
+
getattr(self.structured_report, level)(title, context=context)
|
|
186
|
+
if level == "info":
|
|
187
|
+
logger.info("datahub-bq-connector-session-patch: %s", stats)
|
|
188
|
+
logger.info(
|
|
189
|
+
"datahub-bq-connector-session-patch: per reader project\n%s",
|
|
190
|
+
stats.per_project_table(),
|
|
191
|
+
)
|
|
192
|
+
finally:
|
|
193
|
+
# The aggregator holds a file-backed sqlite connection. A failure while
|
|
194
|
+
# REPORTING must never be the reason it is left open -- least of all in
|
|
195
|
+
# a function whose whole job is making silent failures loud.
|
|
196
|
+
original_close()
|
|
197
|
+
|
|
198
|
+
self.close = close
|
|
199
|
+
|
|
200
|
+
qe.BigQueryQueriesExtractor.__init__ = __init__
|
|
201
|
+
qe.BigQueryQueriesExtractor._datahub_bq_connector_session_patch_installed = True
|
|
@@ -0,0 +1,298 @@
|
|
|
1
|
+
"""Resolve the ``_SESSION`` upstreams DataHub builds for unqualified BigQuery reads.
|
|
2
|
+
|
|
3
|
+
``BigQueryQueriesExtractor._parse_audit_log_row`` sets ``default_schema="_SESSION"``
|
|
4
|
+
(queries_extractor.py:542,558), so any table name a query does not qualify resolves to
|
|
5
|
+
``{billing_project}._SESSION.{table}``. ``is_temp_table()`` then flags it, because the
|
|
6
|
+
dataset starts with ``temp_table_dataset_prefix`` (default ``"_"``), and the aggregator
|
|
7
|
+
drops it -- for usage (sql_parsing_aggregator.py:1022) and for the query entity (:1693).
|
|
8
|
+
The net effect is that every unqualified read in the estate produces no usage, no query
|
|
9
|
+
entity and no lineage, even though the parser saw the table name.
|
|
10
|
+
|
|
11
|
+
This rewrites those upstreams back to the real object by matching the bare table name
|
|
12
|
+
against the tables DataHub discovered during its schema pass. A name matching more than
|
|
13
|
+
one discovered table is left untouched -- dropped exactly as before, never guessed.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import dataclasses
|
|
17
|
+
import logging
|
|
18
|
+
import re
|
|
19
|
+
from collections import defaultdict
|
|
20
|
+
from typing import Callable, Collection, Dict, Optional, Set, Tuple, Union
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
_SESSION = "_session"
|
|
25
|
+
|
|
26
|
+
#: Outcome of classifying one upstream URN. Returned alongside the resolved URN so the
|
|
27
|
+
#: caller can count per-reference volume without re-running (and re-counting) the lookup.
|
|
28
|
+
PASSTHROUGH = "passthrough"
|
|
29
|
+
REWRITTEN = "rewritten"
|
|
30
|
+
AMBIGUOUS = "ambiguous"
|
|
31
|
+
UNKNOWN = "unknown"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclasses.dataclass
|
|
35
|
+
class SessionPatchStats:
|
|
36
|
+
"""Distinct-table counters, plus the per-reference volume behind each.
|
|
37
|
+
|
|
38
|
+
The distinct counters answer "how much of the catalog did we recover"; the
|
|
39
|
+
``*_refs`` counters answer "how many reads did that attribute". Both are needed to
|
|
40
|
+
judge a run: a single hot table read 2,000 times is one rewrite and 2,000 references.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
# Distinct. NOT distinct _SESSION URNs: those embed the reader (billing) project,
|
|
44
|
+
# so one table read from five projects would count five times. `rewritten` counts
|
|
45
|
+
# distinct resolved TARGET tables; `ambiguous`/`unknown` count distinct bare NAMES.
|
|
46
|
+
rewritten: int = 0
|
|
47
|
+
ambiguous: int = 0
|
|
48
|
+
unknown: int = 0
|
|
49
|
+
# Volume: upstream references seen across all queries. Counted only over
|
|
50
|
+
# ``parsed.upstreams``; the column_usage remap reuses the memoised result, so it
|
|
51
|
+
# can never inflate these.
|
|
52
|
+
rewritten_refs: int = 0
|
|
53
|
+
ambiguous_refs: int = 0
|
|
54
|
+
unknown_refs: int = 0
|
|
55
|
+
queries_passed: int = 0
|
|
56
|
+
queries_skipped: int = 0
|
|
57
|
+
queries_mixed: int = 0
|
|
58
|
+
index_size: int = 0
|
|
59
|
+
ambiguous_names: int = 0
|
|
60
|
+
#: reader (billing) project -> {"rewritten"|"ambiguous"|"unknown": reference count}.
|
|
61
|
+
#: Keyed by the project that RAN the query, not the one the table lives in: when the
|
|
62
|
+
#: sidecar's scope is widened, the question being asked is whose traffic benefits.
|
|
63
|
+
refs_by_reader_project: Dict[str, Dict[str, int]] = dataclasses.field(default_factory=dict)
|
|
64
|
+
|
|
65
|
+
def record_ref(self, project: Optional[str], outcome: str) -> None:
|
|
66
|
+
if project is None:
|
|
67
|
+
return
|
|
68
|
+
self.refs_by_reader_project.setdefault(project, {}).setdefault(outcome, 0)
|
|
69
|
+
self.refs_by_reader_project[project][outcome] += 1
|
|
70
|
+
|
|
71
|
+
def per_project_table(self) -> str:
|
|
72
|
+
"""One line per reader project, worst recovery first. For run reports."""
|
|
73
|
+
if not self.refs_by_reader_project:
|
|
74
|
+
return " (no _SESSION references seen)"
|
|
75
|
+
rows = []
|
|
76
|
+
for project, counts in self.refs_by_reader_project.items():
|
|
77
|
+
got = counts.get(REWRITTEN, 0)
|
|
78
|
+
amb = counts.get(AMBIGUOUS, 0)
|
|
79
|
+
unk = counts.get(UNKNOWN, 0)
|
|
80
|
+
total = got + amb + unk
|
|
81
|
+
pct = (100.0 * got / total) if total else 0.0
|
|
82
|
+
rows.append((pct, project, got, amb, unk, total))
|
|
83
|
+
rows.sort()
|
|
84
|
+
return "\n".join(
|
|
85
|
+
f" {project:<34} recovered={got:<7} ambiguous={amb:<7} unknown={unk:<7} "
|
|
86
|
+
f"of {total:<7} ({pct:.1f}%)"
|
|
87
|
+
for pct, project, got, amb, unk, total in rows
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
def __str__(self) -> str:
|
|
91
|
+
return (
|
|
92
|
+
f"rewritten={self.rewritten} tables ({self.rewritten_refs} refs) "
|
|
93
|
+
f"ambiguous={self.ambiguous} names ({self.ambiguous_refs} refs) "
|
|
94
|
+
f"unknown={self.unknown} names ({self.unknown_refs} refs) "
|
|
95
|
+
f"queries_passed={self.queries_passed} queries_skipped={self.queries_skipped} "
|
|
96
|
+
f"queries_mixed={self.queries_mixed} "
|
|
97
|
+
f"(index={self.index_size} unique names, {self.ambiguous_names} ambiguous names)"
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
_TABLE_REF_RE = re.compile(r"^projects/([^/]+)/datasets/([^/]+)/tables/(.+)$")
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _normalise(ref: str) -> Optional[str]:
|
|
105
|
+
"""Discovered tables arrive as ``projects/P/datasets/D/tables/T``; URNs use P.D.T."""
|
|
106
|
+
match = _TABLE_REF_RE.match(ref)
|
|
107
|
+
if match:
|
|
108
|
+
return ".".join(match.groups())
|
|
109
|
+
parts = ref.split(".")
|
|
110
|
+
return ".".join(parts[-3:]) if len(parts) >= 3 else None
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def build_index(discovered: Collection[str]) -> Tuple[Dict[str, str], Set[str]]:
|
|
114
|
+
"""bare table name -> full name, only where the bare name is unambiguous.
|
|
115
|
+
|
|
116
|
+
Names matching several discovered tables go to the ambiguous set and are never
|
|
117
|
+
resolved. Narrowing the recipe's scope *removes* alternatives rather than resolving
|
|
118
|
+
between them, so the index must be built over a scope broad enough that real
|
|
119
|
+
collisions stay visible -- see the design doc, section 9.
|
|
120
|
+
"""
|
|
121
|
+
candidates: Dict[str, Set[str]] = defaultdict(set)
|
|
122
|
+
for ref in discovered:
|
|
123
|
+
full = _normalise(ref)
|
|
124
|
+
if full:
|
|
125
|
+
candidates[full.split(".")[-1].lower()].add(full)
|
|
126
|
+
index = {name: next(iter(v)) for name, v in candidates.items() if len(v) == 1}
|
|
127
|
+
ambiguous = {name for name, v in candidates.items() if len(v) > 1}
|
|
128
|
+
return index, ambiguous
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _split_urn(urn: str) -> Optional[Tuple[str, str, str]]:
|
|
132
|
+
try:
|
|
133
|
+
inner = urn[urn.index("(") + 1 : urn.rindex(")")]
|
|
134
|
+
platform, name, env = inner.split(",")
|
|
135
|
+
return platform, name, env
|
|
136
|
+
except Exception:
|
|
137
|
+
return None
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _session_table(name: str) -> Optional[str]:
|
|
141
|
+
"""Bare table name if this is a ``_SESSION``-qualified name, else None."""
|
|
142
|
+
parts = name.split(".")
|
|
143
|
+
if len(parts) < 3 or parts[-2].lower() != _SESSION:
|
|
144
|
+
return None
|
|
145
|
+
return parts[-1].lower()
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
DiscoveredTables = Union[Collection[str], Callable[[], Collection[str]]]
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def install_on_aggregator(
|
|
152
|
+
aggregator,
|
|
153
|
+
discovered_tables: DiscoveredTables,
|
|
154
|
+
disjoint_only: bool = True,
|
|
155
|
+
) -> SessionPatchStats:
|
|
156
|
+
"""Wrap an aggregator so ``_SESSION`` upstreams resolve to real datasets.
|
|
157
|
+
|
|
158
|
+
With ``disjoint_only`` (the default) the aggregator sees ONLY the upstreams this
|
|
159
|
+
resolver actually recovered, and queries where nothing was recovered are dropped
|
|
160
|
+
entirely. That keeps a sidecar recipe strictly complementary to the main one: it
|
|
161
|
+
emits exactly what the main recipe structurally cannot, and nothing else, so the two
|
|
162
|
+
never write usage for the same (dataset, day) and widening the sidecar's scope is
|
|
163
|
+
harmless.
|
|
164
|
+
|
|
165
|
+
``discovered_tables`` may be a collection or a zero-arg callable returning one.
|
|
166
|
+
The index is built lazily on first use: BigqueryV2Source hands the extractor a live
|
|
167
|
+
reference to ``bq_schema_extractor.table_refs``, which is still empty when the
|
|
168
|
+
extractor is constructed and only fills during the schema pass.
|
|
169
|
+
"""
|
|
170
|
+
stats = SessionPatchStats()
|
|
171
|
+
state: Dict[str, object] = {}
|
|
172
|
+
# Distinct-ness is measured on what the counter CLAIMS to count, not on the cache
|
|
173
|
+
# key: the cache is keyed by the full _SESSION URN, which embeds the reader project.
|
|
174
|
+
seen_targets: Set[str] = set()
|
|
175
|
+
seen_ambiguous: Set[str] = set()
|
|
176
|
+
seen_unknown: Set[str] = set()
|
|
177
|
+
# Memoised so each distinct URN is classified -- and counted -- exactly once. The
|
|
178
|
+
# column_usage remap below resolves the same URNs a second time; without this the
|
|
179
|
+
# counters would report resolution calls rather than distinct tables.
|
|
180
|
+
cache: Dict[str, Tuple[str, str]] = {}
|
|
181
|
+
|
|
182
|
+
def get_index() -> Tuple[Dict[str, str], Set[str]]:
|
|
183
|
+
if "index" not in state:
|
|
184
|
+
tables = discovered_tables() if callable(discovered_tables) else discovered_tables
|
|
185
|
+
index, ambiguous = build_index(tables or ())
|
|
186
|
+
stats.index_size, stats.ambiguous_names = len(index), len(ambiguous)
|
|
187
|
+
state["index"] = (index, ambiguous)
|
|
188
|
+
logger.info("datahub-bq-connector-session-patch: resolution index built (%s)", stats)
|
|
189
|
+
return state["index"] # type: ignore[return-value]
|
|
190
|
+
|
|
191
|
+
original = aggregator.add_preparsed_query
|
|
192
|
+
|
|
193
|
+
def classify(urn: str) -> Tuple[str, str, Optional[str]]:
|
|
194
|
+
"""(resolved_urn, outcome, reader_project).
|
|
195
|
+
|
|
196
|
+
Counts DISTINCT outcomes on first sight only; per-reference volume is counted by
|
|
197
|
+
the caller, so the column_usage remap below cannot inflate either.
|
|
198
|
+
"""
|
|
199
|
+
hit = cache.get(urn)
|
|
200
|
+
if hit is not None:
|
|
201
|
+
return hit
|
|
202
|
+
|
|
203
|
+
index, ambiguous = get_index()
|
|
204
|
+
result: Tuple[str, str, Optional[str]] = (urn, PASSTHROUGH, None)
|
|
205
|
+
split = _split_urn(urn)
|
|
206
|
+
if split is not None:
|
|
207
|
+
platform, name, env = split
|
|
208
|
+
table = _session_table(name)
|
|
209
|
+
if table is not None:
|
|
210
|
+
parts = name.split(".")
|
|
211
|
+
# `{billing_project}._SESSION.{table}` -- the project that ran the query.
|
|
212
|
+
reader = parts[-3] if len(parts) >= 3 else None
|
|
213
|
+
target = index.get(table)
|
|
214
|
+
if target is not None:
|
|
215
|
+
prefix = parts[:-3]
|
|
216
|
+
resolved = ".".join(prefix + [target])
|
|
217
|
+
target_urn = f"urn:li:dataset:({platform},{resolved},{env})"
|
|
218
|
+
seen_targets.add(target_urn)
|
|
219
|
+
stats.rewritten = len(seen_targets)
|
|
220
|
+
result = (target_urn, REWRITTEN, reader)
|
|
221
|
+
elif table in ambiguous:
|
|
222
|
+
seen_ambiguous.add(table)
|
|
223
|
+
stats.ambiguous = len(seen_ambiguous)
|
|
224
|
+
logger.debug("datahub-bq-connector-session-patch: %r is ambiguous, leaving as temp", table)
|
|
225
|
+
result = (urn, AMBIGUOUS, reader)
|
|
226
|
+
else:
|
|
227
|
+
seen_unknown.add(table)
|
|
228
|
+
stats.unknown = len(seen_unknown)
|
|
229
|
+
result = (urn, UNKNOWN, reader)
|
|
230
|
+
|
|
231
|
+
cache[urn] = result
|
|
232
|
+
return result
|
|
233
|
+
|
|
234
|
+
def resolve(urn: str) -> str:
|
|
235
|
+
return classify(urn)[0]
|
|
236
|
+
|
|
237
|
+
def add_preparsed_query(parsed, *args, **kwargs):
|
|
238
|
+
recovered = set()
|
|
239
|
+
already_resolvable = set()
|
|
240
|
+
resolved = []
|
|
241
|
+
for upstream in parsed.upstreams:
|
|
242
|
+
new, outcome, reader = classify(upstream)
|
|
243
|
+
resolved.append(new)
|
|
244
|
+
if outcome == PASSTHROUGH:
|
|
245
|
+
# A real, already-qualified reference -- the main recipe resolves this
|
|
246
|
+
# one itself and will emit a query entity carrying it.
|
|
247
|
+
already_resolvable.add(new)
|
|
248
|
+
if outcome == REWRITTEN:
|
|
249
|
+
stats.rewritten_refs += 1
|
|
250
|
+
recovered.add(new)
|
|
251
|
+
elif outcome == AMBIGUOUS:
|
|
252
|
+
stats.ambiguous_refs += 1
|
|
253
|
+
elif outcome == UNKNOWN:
|
|
254
|
+
stats.unknown_refs += 1
|
|
255
|
+
if outcome != PASSTHROUGH:
|
|
256
|
+
stats.record_ref(reader, outcome)
|
|
257
|
+
|
|
258
|
+
if disjoint_only:
|
|
259
|
+
if not recovered:
|
|
260
|
+
# Every reference was already resolvable -- the main recipe has it.
|
|
261
|
+
stats.queries_skipped += 1
|
|
262
|
+
return
|
|
263
|
+
if already_resolvable:
|
|
264
|
+
# MIXED query: some references were already resolvable, some we
|
|
265
|
+
# recovered. Disjointness does NOT hold here. The query entity is keyed
|
|
266
|
+
# by get_query_fingerprint(sql), a pure function of the SQL text, so both
|
|
267
|
+
# pipelines emit the SAME urn:li:query:<fp> -- and querySubjects is a
|
|
268
|
+
# versioned aspect, not timeseries. Emitting our narrowed subject set
|
|
269
|
+
# would overwrite the main recipe's, and the entity would flip between
|
|
270
|
+
# the two on every scheduled run. Passing the union instead would
|
|
271
|
+
# double-book usage for the already-resolvable half. Neither is
|
|
272
|
+
# acceptable, so the query is left entirely to the main recipe.
|
|
273
|
+
stats.queries_mixed += 1
|
|
274
|
+
return
|
|
275
|
+
parsed.upstreams = sorted(recovered)
|
|
276
|
+
else:
|
|
277
|
+
parsed.upstreams = resolved
|
|
278
|
+
if parsed.downstream:
|
|
279
|
+
parsed.downstream = resolve(parsed.downstream)
|
|
280
|
+
|
|
281
|
+
if parsed.column_usage:
|
|
282
|
+
# Union rather than assign: two keys can collapse onto one URN (a self-join
|
|
283
|
+
# referencing the same table both qualified and unqualified), and dict
|
|
284
|
+
# comprehension would silently keep only the last column set.
|
|
285
|
+
remapped: Dict[str, Set[str]] = {}
|
|
286
|
+
for key, columns in parsed.column_usage.items():
|
|
287
|
+
remapped.setdefault(resolve(key), set()).update(columns or ())
|
|
288
|
+
parsed.column_usage = (
|
|
289
|
+
{k: v for k, v in remapped.items() if k in recovered}
|
|
290
|
+
if disjoint_only
|
|
291
|
+
else remapped
|
|
292
|
+
)
|
|
293
|
+
|
|
294
|
+
stats.queries_passed += 1
|
|
295
|
+
return original(parsed, *args, **kwargs)
|
|
296
|
+
|
|
297
|
+
aggregator.add_preparsed_query = add_preparsed_query
|
|
298
|
+
return stats
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""``datahub_bq_connector_session_patch`` -- the sidecar source type.
|
|
2
|
+
|
|
3
|
+
A ``BigqueryV2Source`` that arms the resolver for its own run and nothing else. Point a
|
|
4
|
+
narrow sidecar recipe at ``type: datahub_bq_connector_session_patch``; the production recipe keeps
|
|
5
|
+
``type: bigquery`` and is unaffected even though both interpreters have this installed.
|
|
6
|
+
|
|
7
|
+
It also refuses to start on the two misconfigurations that are actively destructive
|
|
8
|
+
rather than merely wrong -- see ``_check_sidecar_safety``.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import logging
|
|
12
|
+
from typing import Iterable, List
|
|
13
|
+
|
|
14
|
+
from datahub.ingestion.api.common import PipelineContext
|
|
15
|
+
from datahub.ingestion.api.decorators import (
|
|
16
|
+
SupportStatus,
|
|
17
|
+
config_class,
|
|
18
|
+
platform_name,
|
|
19
|
+
support_status,
|
|
20
|
+
)
|
|
21
|
+
from datahub.ingestion.api.workunit import MetadataWorkUnit
|
|
22
|
+
from datahub.ingestion.source.bigquery_v2.bigquery import BigqueryV2Source
|
|
23
|
+
from datahub.ingestion.source.bigquery_v2.bigquery_config import BigQueryV2Config
|
|
24
|
+
|
|
25
|
+
from datahub_bq_connector_session_patch.patch import armed, install
|
|
26
|
+
|
|
27
|
+
logger = logging.getLogger(__name__)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class SidecarMisconfigured(ValueError):
|
|
31
|
+
"""The recipe would damage the catalog. Raised before any metadata is read."""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _check_sidecar_safety(config: BigQueryV2Config) -> None:
|
|
35
|
+
"""Refuse the two settings that make this sidecar destructive.
|
|
36
|
+
|
|
37
|
+
Both were reproduced against a live catalog; neither is a style preference:
|
|
38
|
+
|
|
39
|
+
* ``remove_stale_metadata`` -- the sidecar writes a ``status`` aspect on every
|
|
40
|
+
dataset it produces usage for, which enrols them in *its* stale-removal
|
|
41
|
+
checkpoint. Its URN set is usage-driven, so any dataset not read in the window
|
|
42
|
+
falls out of the checkpoint and gets soft-deleted -- datasets the main recipe
|
|
43
|
+
owns. This is normal operation, not an edge case.
|
|
44
|
+
* ``include_table_lineage`` -- a mis-resolved write would create a wrong edge into a
|
|
45
|
+
real table, which is worse than the missing usage this package exists to fix.
|
|
46
|
+
"""
|
|
47
|
+
problems: List[str] = []
|
|
48
|
+
|
|
49
|
+
stateful = config.stateful_ingestion
|
|
50
|
+
# remove_stale_metadata defaults to True, so only the combination bites: stale
|
|
51
|
+
# removal cannot run at all when stateful ingestion is off, and refusing that
|
|
52
|
+
# combination would block every file-sink measurement run.
|
|
53
|
+
if (
|
|
54
|
+
stateful is not None
|
|
55
|
+
and getattr(stateful, "enabled", False)
|
|
56
|
+
and getattr(stateful, "remove_stale_metadata", False)
|
|
57
|
+
):
|
|
58
|
+
problems.append(
|
|
59
|
+
"stateful_ingestion.remove_stale_metadata must be false. This sidecar's URN "
|
|
60
|
+
"set is usage-driven, so stale removal soft-deletes datasets the main "
|
|
61
|
+
"recipe owns, on every run."
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
if config.include_table_lineage:
|
|
65
|
+
problems.append(
|
|
66
|
+
"include_table_lineage must be false. Resolved names are used for usage "
|
|
67
|
+
"attribution only; emitting lineage from a resolved name risks a wrong edge "
|
|
68
|
+
"into a real table."
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
if problems:
|
|
72
|
+
raise SidecarMisconfigured(
|
|
73
|
+
"datahub_bq_connector_session_patch refuses to run with this recipe:\n - "
|
|
74
|
+
+ "\n - ".join(problems)
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
@platform_name("BigQuery")
|
|
79
|
+
@config_class(BigQueryV2Config)
|
|
80
|
+
@support_status(SupportStatus.BETA)
|
|
81
|
+
class BigQuerySessionPatchSource(BigqueryV2Source):
|
|
82
|
+
"""BigQuery source that recovers reads whose SQL did not qualify its table names."""
|
|
83
|
+
|
|
84
|
+
def __init__(self, ctx: PipelineContext, config: BigQueryV2Config):
|
|
85
|
+
_check_sidecar_safety(config)
|
|
86
|
+
super().__init__(ctx, config)
|
|
87
|
+
|
|
88
|
+
@classmethod
|
|
89
|
+
def create(cls, config_dict: dict, ctx: PipelineContext) -> "BigQuerySessionPatchSource":
|
|
90
|
+
config = BigQueryV2Config.model_validate(config_dict)
|
|
91
|
+
return cls(ctx, config)
|
|
92
|
+
|
|
93
|
+
def get_workunits_internal(self) -> Iterable[MetadataWorkUnit]:
|
|
94
|
+
# install() patches the class; armed() is what makes the patch do anything, and
|
|
95
|
+
# it covers only this generator's execution. The extractor is constructed inside
|
|
96
|
+
# super().get_workunits_internal(), so both have to be in place before we
|
|
97
|
+
# delegate.
|
|
98
|
+
install()
|
|
99
|
+
with armed():
|
|
100
|
+
yield from super().get_workunits_internal()
|
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: datahub-bq-connector-session-patch
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: DataHub BigQuery sidecar source: attribute reads whose SQL does not qualify its table names
|
|
5
|
+
Project-URL: Homepage, https://github.com/gutro/dv-datahub/tree/master/datahub-bq-connector-session-patch
|
|
6
|
+
Project-URL: Repository, https://github.com/gutro/dv-datahub
|
|
7
|
+
Project-URL: Issues, https://github.com/gutro/dv-datahub/issues
|
|
8
|
+
Author: DataVantage Platform
|
|
9
|
+
License-Expression: Apache-2.0
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: bigquery,datahub,ingestion,lineage,usage
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Topic :: Database
|
|
16
|
+
Requires-Python: >=3.9
|
|
17
|
+
Provides-Extra: test
|
|
18
|
+
Requires-Dist: acryl-datahub[bigquery]<1.8,>=1.6; extra == 'test'
|
|
19
|
+
Requires-Dist: pytest>=7; extra == 'test'
|
|
20
|
+
Requires-Dist: pyyaml; extra == 'test'
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
|
|
23
|
+
# datahub-bq-connector-session-patch
|
|
24
|
+
|
|
25
|
+
A DataHub BigQuery **sidecar source** that attributes reads whose SQL did not qualify
|
|
26
|
+
its table names — the ones the stock connector sees, misfiles and silently discards.
|
|
27
|
+
|
|
28
|
+
Full diagnosis, design, verification and limits:
|
|
29
|
+
`datavantage/docs/datahub-unqualified-read-attribution.md`.
|
|
30
|
+
|
|
31
|
+
## The problem
|
|
32
|
+
|
|
33
|
+
A client with a BigQuery *default dataset* set can write `SELECT ... FROM Bets_Details`.
|
|
34
|
+
BigQuery fills in the blank at run time, but never records it: `INFORMATION_SCHEMA.JOBS`
|
|
35
|
+
carries the SQL text and the billing project, not the job's default dataset.
|
|
36
|
+
|
|
37
|
+
DataHub's connector substitutes `_SESSION` for the missing dataset
|
|
38
|
+
(`queries_extractor.py:542,558`), on the assumption that a bare name is usually a
|
|
39
|
+
temporary table. Its own temp-table rule then flags anything whose dataset starts with
|
|
40
|
+
`_` (`:309`), and the aggregator drops it — for usage
|
|
41
|
+
(`sql_parsing_aggregator.py:1022`) and for the query entity (`:1693`).
|
|
42
|
+
|
|
43
|
+
So the connector *sees* the table name, guesses the wrong container, and discards the
|
|
44
|
+
result. Nothing is reported: the sibling branch of `is_temp_table` records what it drops,
|
|
45
|
+
this one returns `True` in silence. Measured here: one client, 2,866 jobs in 7 days,
|
|
46
|
+
100% unqualified, 0 usage rows.
|
|
47
|
+
|
|
48
|
+
**A `0` in DataHub usage is therefore ambiguous** — "nobody reads this", or "read
|
|
49
|
+
constantly, with unqualified SQL". Decommissioning decisions depend on telling those apart.
|
|
50
|
+
|
|
51
|
+
## What this does
|
|
52
|
+
|
|
53
|
+
For upstreams that landed in `_SESSION`, it resolves the bare table name against the
|
|
54
|
+
tables DataHub discovered during its own schema pass. A name matching more than one
|
|
55
|
+
discovered table is **left untouched** — dropped exactly as before, never guessed.
|
|
56
|
+
|
|
57
|
+
It ships as a distinct source type, so the production recipe stays stock:
|
|
58
|
+
|
|
59
|
+
```yaml
|
|
60
|
+
source:
|
|
61
|
+
type: datahub_bq_connector_session_patch
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
## Installing
|
|
65
|
+
|
|
66
|
+
Per-recipe, via **Extra Pip Libraries** (DataHub UI) or `extra_pip_requirements`. That
|
|
67
|
+
field is scoped to one recipe, which is what keeps the production recipe unaffected.
|
|
68
|
+
|
|
69
|
+
```
|
|
70
|
+
datahub-bq-connector-session-patch
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
The wheel declares **no dependencies** — it imports only `datahub.*`, which any
|
|
74
|
+
environment running a recipe already has. Declaring `acryl-datahub` would let pip
|
|
75
|
+
re-resolve the executor's own pinned copy while installing this. The tested range
|
|
76
|
+
(`>=1.6,<1.8`) is enforced at runtime instead: an out-of-range executor raises a
|
|
77
|
+
structured warning into the ingestion report rather than failing the run.
|
|
78
|
+
|
|
79
|
+
## The sidecar recipe
|
|
80
|
+
|
|
81
|
+
```yaml
|
|
82
|
+
pipeline_name: bq-session-attribution # its OWN name — never the main recipe's
|
|
83
|
+
source:
|
|
84
|
+
type: datahub_bq_connector_session_patch
|
|
85
|
+
config:
|
|
86
|
+
project_id_pattern:
|
|
87
|
+
allow:
|
|
88
|
+
- '^dv-prod-eu-w1(-.+)?-data$'
|
|
89
|
+
- '^dv-ext-prod-eu-w1(-.+)?-data$'
|
|
90
|
+
- '^dv-prod-eu-w1(-.+)?-comp\d+$'
|
|
91
|
+
include_tables: false # lightweight discovery still fills table_refs
|
|
92
|
+
include_views: false
|
|
93
|
+
include_table_lineage: false # refused if true — see below
|
|
94
|
+
include_usage_statistics: true
|
|
95
|
+
use_queries_v2: true
|
|
96
|
+
include_queries: true
|
|
97
|
+
include_query_usage_statistics: true
|
|
98
|
+
region_qualifiers: ['region-europe-west1']
|
|
99
|
+
start_time: '-7 days'
|
|
100
|
+
enable_stateful_time_window: true
|
|
101
|
+
stateful_ingestion:
|
|
102
|
+
enabled: true # the time-window watermark
|
|
103
|
+
remove_stale_metadata: false # refused if true — see below
|
|
104
|
+
env: PROD
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
### Two settings this source refuses to start with
|
|
108
|
+
|
|
109
|
+
Both were reproduced against a live catalog. They are not style preferences, so they are
|
|
110
|
+
enforced in code rather than left to the YAML:
|
|
111
|
+
|
|
112
|
+
| setting | why it is refused |
|
|
113
|
+
|---|---|
|
|
114
|
+
| `stateful_ingestion.remove_stale_metadata: true` | The sidecar writes a `status` aspect on every dataset it produces usage for, enrolling them in *its* stale-removal checkpoint. Its URN set is **usage-driven**, so any dataset not read that day falls out of the checkpoint and is soft-deleted — datasets the main recipe owns. Normal operation, not an edge case. |
|
|
115
|
+
| `include_table_lineage: true` | Resolved names drive usage attribution only. A mis-resolved write would create a **wrong edge into a real table**, which is worse than the missing usage this package exists to fix. |
|
|
116
|
+
|
|
117
|
+
`remove_stale_metadata` defaults to `true`, so it must be set explicitly. Only the
|
|
118
|
+
combination bites — with `stateful_ingestion.enabled: false` there is no checkpoint and
|
|
119
|
+
nothing is refused.
|
|
120
|
+
|
|
121
|
+
### Scope it to the readers, not to minimise ambiguity
|
|
122
|
+
|
|
123
|
+
`dataset_pattern` should be **wide**. "Ambiguous" means the name matches more than one
|
|
124
|
+
entry in `discovered_tables` *within the recipe's scope*, so narrowing scope removes the
|
|
125
|
+
alternatives rather than resolving between them. If a reader's default dataset is
|
|
126
|
+
`prod_spain_*` but the index holds only `prod_malta_*`, the name resolves **confidently
|
|
127
|
+
and wrongly**.
|
|
128
|
+
|
|
129
|
+
Ambiguity fails safe. A too-narrow index fails silently, in the wrong direction.
|
|
130
|
+
|
|
131
|
+
Widening is otherwise harmless: `disjoint_only` means the sidecar emits only URNs it
|
|
132
|
+
actually recovered, so it stays quiet wherever SQL is already qualified and never
|
|
133
|
+
double-books usage with the main recipe.
|
|
134
|
+
|
|
135
|
+
## Reading the run report
|
|
136
|
+
|
|
137
|
+
```
|
|
138
|
+
rewritten=41 tables (2792 refs) ambiguous=18 names (63 refs) unknown=7 names (12 refs)
|
|
139
|
+
queries_passed=954 queries_skipped=3118 (index=1631 unique names, 412 ambiguous names)
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
- **distinct counters** — how much of the catalog was recovered. `rewritten` counts
|
|
143
|
+
distinct resolved **target tables**; `ambiguous` and `unknown` count distinct bare
|
|
144
|
+
**names**. Deliberately not distinct `_SESSION` URNs: those embed the reader project,
|
|
145
|
+
so one table read from five projects would report as five.
|
|
146
|
+
- **`*_refs`** — how many reads that attributed. One hot table read 2,000 times is one
|
|
147
|
+
rewrite and 2,000 references.
|
|
148
|
+
- **`queries_skipped`** — queries whose SQL was already fully qualified. The main recipe
|
|
149
|
+
has those; the sidecar correctly stays out of the way.
|
|
150
|
+
- **`queries_mixed`** — queries carrying *both* an already-resolvable reference and a
|
|
151
|
+
recovered one. Also left to the main recipe, for a subtler reason, below.
|
|
152
|
+
|
|
153
|
+
### Why mixed queries are skipped
|
|
154
|
+
|
|
155
|
+
`disjoint_only` keeps the two pipelines from double-booking **usage**, which is a
|
|
156
|
+
timeseries aspect. It does not, by itself, protect the **query entity**.
|
|
157
|
+
|
|
158
|
+
Query entities are keyed by `get_query_fingerprint(sql, platform, fast=True)` — a pure
|
|
159
|
+
function of the SQL text, so both pipelines derive the *same* `urn:li:query:<fp>`. But
|
|
160
|
+
`querySubjects` is a versioned aspect, not timeseries. If the sidecar emitted its
|
|
161
|
+
narrowed subject set for a query the main recipe also sees, whichever ran last would
|
|
162
|
+
win, and the entity would flip between the two subject sets on every scheduled run.
|
|
163
|
+
|
|
164
|
+
Passing the union instead would fix the subjects and double-book usage for the
|
|
165
|
+
already-resolvable half. Neither is acceptable, so a mixed query is left entirely to the
|
|
166
|
+
main recipe and counted. Measured on `dv-ext-prod-eu-w1-data` over 7 days:
|
|
167
|
+
`queries_mixed=0` — the guard costs nothing there, because that client's SQL is
|
|
168
|
+
uniformly unqualified.
|
|
169
|
+
|
|
170
|
+
A per-reader-project breakdown is logged alongside it, so a project that recovers nothing
|
|
171
|
+
shows up at 0.0% rather than vanishing.
|
|
172
|
+
|
|
173
|
+
### Two failures that look like success
|
|
174
|
+
|
|
175
|
+
Both were hit during development. Each leaves the run green and the output plausible:
|
|
176
|
+
|
|
177
|
+
1. **`discovered_tables` is empty at construction.** The source passes a *live reference*
|
|
178
|
+
to `bq_schema_extractor.table_refs`, which only fills during the schema pass. The
|
|
179
|
+
index is built lazily on first use, never in `__init__`.
|
|
180
|
+
2. **Refs are `projects/P/datasets/D/tables/T`, not `P.D.T`.** Naive dotted parsing
|
|
181
|
+
matches nothing.
|
|
182
|
+
|
|
183
|
+
Because of these the source raises a structured **warning** if the index came out empty
|
|
184
|
+
or nothing was rewritten. **Never trust a negative result without checking the index
|
|
185
|
+
size.**
|
|
186
|
+
|
|
187
|
+
## Limits
|
|
188
|
+
|
|
189
|
+
Unique-name resolution only. Where a bare name matches several catalogued objects it is
|
|
190
|
+
dropped. How much that costs depends entirely on the estate:
|
|
191
|
+
|
|
192
|
+
| population | ambiguous names |
|
|
193
|
+
|---|---|
|
|
194
|
+
| views in `dv-ext-prod-eu-w1-data` | **0.0%** (162/162 unique) |
|
|
195
|
+
| views in `dv-prod-eu-w1-data` | **54.4%** |
|
|
196
|
+
| tables referenced estate-wide | 53.0% |
|
|
197
|
+
|
|
198
|
+
The external-facing views carry licence/brand suffixes and are unique by construction.
|
|
199
|
+
The internal estate's per-market layout duplicates names across `prod_malta_*`,
|
|
200
|
+
`prod_spain_*`, `prod_italy_*`…, and the cross-market `common_` layer duplicates a
|
|
201
|
+
further 39 names against them — so roughly half of any bare reads against it stay
|
|
202
|
+
unrecovered.
|
|
203
|
+
|
|
204
|
+
Resolving those needs **leaf-set matching**, not name matching: the colliding
|
|
205
|
+
common-vs-market views have identical names but cleanly distinct leaf sets. That is
|
|
206
|
+
explicitly out of scope — see §9.1 of the design doc for what it would take.
|
|
207
|
+
|
|
208
|
+
## Measuring a scope before shipping it
|
|
209
|
+
|
|
210
|
+
```bash
|
|
211
|
+
export BQ_SESSION_PATCH_BILLING_PROJECT=dv-ext-prod-eu-w1-data
|
|
212
|
+
python tools/measure_scope.py eu baseline --days 7
|
|
213
|
+
python tools/measure_scope.py eu patched --days 7
|
|
214
|
+
python tools/diff_runs.py runs/mcps_eu_baseline_7d.json runs/mcps_eu_patched_7d.json
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
Scope flags: `--projects` for explicit ids, `--project-pattern` to override the region's
|
|
218
|
+
`project_id_pattern.allow` regexes, `--days` for the window.
|
|
219
|
+
|
|
220
|
+
**The billing project must be explicit** — `--billing`, or
|
|
221
|
+
`$BQ_SESSION_PATCH_BILLING_PROJECT`, or inferred when `--projects` names exactly one.
|
|
222
|
+
DataHub passes it straight to `bigquery.Client(...)`, and when it is `None` the client
|
|
223
|
+
inherits whatever `google.auth.default()` resolves: `GOOGLE_CLOUD_PROJECT` first, then the
|
|
224
|
+
ADC file's quota project, then the active gcloud config. A stray env var will therefore
|
|
225
|
+
bill every `INFORMATION_SCHEMA` job to an unrelated project and fail as
|
|
226
|
+
`bigquery.jobs.create` denied against a project you never named. The script refuses to
|
|
227
|
+
start rather than inherit one, and prints both the billing and the ambient project so a
|
|
228
|
+
mismatch is visible.
|
|
229
|
+
|
|
230
|
+
`measure_scope.py` writes MCPs to a file sink (nothing reaches GMS) and prints the
|
|
231
|
+
per-project recovery table; `diff_runs.py` reports datasets gaining usage, query entities
|
|
232
|
+
gained and readers newly attributed, and fails if any `schemaMetadata`,
|
|
233
|
+
`upstreamLineage` or `datasetProperties` aspect leaked into the output.
|
|
234
|
+
|
|
235
|
+
## Trying it against a local DataHub
|
|
236
|
+
|
|
237
|
+
`recipes/local/` holds a two-step pair for a quickstart instance, scoped to one project:
|
|
238
|
+
|
|
239
|
+
```bash
|
|
240
|
+
make venv && source .venv/bin/activate
|
|
241
|
+
datahub ingest -c recipes/local/01-catalog.yml # stock `bigquery` — the catalog
|
|
242
|
+
datahub ingest -c recipes/local/02-usage-sidecar.yml # this package — the recovered usage
|
|
243
|
+
```
|
|
244
|
+
|
|
245
|
+
Step 1 runs with usage **on**, exactly as production does, and still misses the
|
|
246
|
+
unqualified reads — that is the premise. Step 2 shows what it missed. Measured against
|
|
247
|
+
`dv-ext-prod-eu-w1-data` over 7 days:
|
|
248
|
+
|
|
249
|
+
| | step 1 (catalog) | step 2 (sidecar) |
|
|
250
|
+
|---|---|---|
|
|
251
|
+
| datasets with a usage aspect | 5 | **+59** |
|
|
252
|
+
| query entities | 170 | **+61** |
|
|
253
|
+
| aspects emitted | schema, lineage, properties, usage, queries | usage and queries only |
|
|
254
|
+
|
|
255
|
+
Both use ADC, so the VPN must be up — VPC-SC blocks ADC while the `bq` CLI keeps
|
|
256
|
+
working, which makes the failure look like a permissions problem.
|
|
257
|
+
|
|
258
|
+
They run **stateless** (`stateful_ingestion.enabled: false`), so re-running re-emits the
|
|
259
|
+
same days. Timeseries aspects append rather than upsert, so repeated local runs
|
|
260
|
+
accumulate duplicate usage documents for a day. Harmless locally; in a real deployment
|
|
261
|
+
keep the stateful time window, which is what makes each day emit exactly once.
|
|
262
|
+
|
|
263
|
+
## Developing
|
|
264
|
+
|
|
265
|
+
```bash
|
|
266
|
+
make venv # .venv on python 3.11, package installed editable + test extras
|
|
267
|
+
make test # 49 tests against a real acryl-datahub
|
|
268
|
+
make gate # the acceptance gate: must FAIL on stock, PASS patched
|
|
269
|
+
make release # test, gate, bump, rebuild, verify -> dist/
|
|
270
|
+
make publish
|
|
271
|
+
```
|
|
272
|
+
|
|
273
|
+
`make venv` exists for interactive work and for the `tools/` scripts, which need real
|
|
274
|
+
BigQuery credentials and so cannot run in a throwaway environment. `make test` and
|
|
275
|
+
`make gate` deliberately **do not** use it — they build their own environment per run, so
|
|
276
|
+
a stale or hand-modified `.venv` can never make them pass.
|
|
277
|
+
|
|
278
|
+
`make gate` is the one that matters on an acryl-datahub bump. This package patches
|
|
279
|
+
private internals, so `make test` alone would stay green even if upstream fixed the
|
|
280
|
+
defect out from under it — at which point this package is dead weight and should be
|
|
281
|
+
retired, not shipped. `gate` fails loudly in that case.
|
|
282
|
+
|
|
283
|
+
## Upstream
|
|
284
|
+
|
|
285
|
+
The defect is unfixed on `datahub-project/datahub` master as of 2026-09-14. The ask
|
|
286
|
+
there is modest: report the `_SESSION` discard, or offer opt-in resolution against
|
|
287
|
+
discovered tables. See `UPSTREAM-ISSUE.md`.
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
datahub_bq_connector_session_patch/__init__.py,sha256=U_kaflaxg_0rAwthDzHJDyrNL2W867klJyEUHtPhEzY,1247
|
|
2
|
+
datahub_bq_connector_session_patch/patch.py,sha256=GwkNiMIbACZM8HJrJHVWRCqKcu7uTRhyeDkDhCNXtqQ,8128
|
|
3
|
+
datahub_bq_connector_session_patch/resolver.py,sha256=szUyW622C4NGZhi-nIEcRwhxsfpGfxw9o5aG21mk8Og,13419
|
|
4
|
+
datahub_bq_connector_session_patch/source.py,sha256=16lsU8UDdOoC9qzbv69xP1RG4ZsxF0RDlOOTVXI822Q,4168
|
|
5
|
+
datahub_bq_connector_session_patch-0.1.0.dist-info/METADATA,sha256=6dNNhK5hKxpN9trR0RX5y9-jfa8PPDlk3isoTqtUMhQ,13628
|
|
6
|
+
datahub_bq_connector_session_patch-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
7
|
+
datahub_bq_connector_session_patch-0.1.0.dist-info/entry_points.txt,sha256=HGFQhMcd2ToOmoOBYO4Q3TEu_KsFvDxG98J8M4sDNQc,141
|
|
8
|
+
datahub_bq_connector_session_patch-0.1.0.dist-info/licenses/LICENSE,sha256=z8d0m5b2O9McPEK1xHG_dWgUBT6EfBDz6wA0F7xSPTA,11358
|
|
9
|
+
datahub_bq_connector_session_patch-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
|
|
2
|
+
Apache License
|
|
3
|
+
Version 2.0, January 2004
|
|
4
|
+
http://www.apache.org/licenses/
|
|
5
|
+
|
|
6
|
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
|
7
|
+
|
|
8
|
+
1. Definitions.
|
|
9
|
+
|
|
10
|
+
"License" shall mean the terms and conditions for use, reproduction,
|
|
11
|
+
and distribution as defined by Sections 1 through 9 of this document.
|
|
12
|
+
|
|
13
|
+
"Licensor" shall mean the copyright owner or entity authorized by
|
|
14
|
+
the copyright owner that is granting the License.
|
|
15
|
+
|
|
16
|
+
"Legal Entity" shall mean the union of the acting entity and all
|
|
17
|
+
other entities that control, are controlled by, or are under common
|
|
18
|
+
control with that entity. For the purposes of this definition,
|
|
19
|
+
"control" means (i) the power, direct or indirect, to cause the
|
|
20
|
+
direction or management of such entity, whether by contract or
|
|
21
|
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
|
22
|
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
|
23
|
+
|
|
24
|
+
"You" (or "Your") shall mean an individual or Legal Entity
|
|
25
|
+
exercising permissions granted by this License.
|
|
26
|
+
|
|
27
|
+
"Source" form shall mean the preferred form for making modifications,
|
|
28
|
+
including but not limited to software source code, documentation
|
|
29
|
+
source, and configuration files.
|
|
30
|
+
|
|
31
|
+
"Object" form shall mean any form resulting from mechanical
|
|
32
|
+
transformation or translation of a Source form, including but
|
|
33
|
+
not limited to compiled object code, generated documentation,
|
|
34
|
+
and conversions to other media types.
|
|
35
|
+
|
|
36
|
+
"Work" shall mean the work of authorship, whether in Source or
|
|
37
|
+
Object form, made available under the License, as indicated by a
|
|
38
|
+
copyright notice that is included in or attached to the work
|
|
39
|
+
(an example is provided in the Appendix below).
|
|
40
|
+
|
|
41
|
+
"Derivative Works" shall mean any work, whether in Source or Object
|
|
42
|
+
form, that is based on (or derived from) the Work and for which the
|
|
43
|
+
editorial revisions, annotations, elaborations, or other modifications
|
|
44
|
+
represent, as a whole, an original work of authorship. For the purposes
|
|
45
|
+
of this License, Derivative Works shall not include works that remain
|
|
46
|
+
separable from, or merely link (or bind by name) to the interfaces of,
|
|
47
|
+
the Work and Derivative Works thereof.
|
|
48
|
+
|
|
49
|
+
"Contribution" shall mean any work of authorship, including
|
|
50
|
+
the original version of the Work and any modifications or additions
|
|
51
|
+
to that Work or Derivative Works thereof, that is intentionally
|
|
52
|
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
|
53
|
+
or by an individual or Legal Entity authorized to submit on behalf of
|
|
54
|
+
the copyright owner. For the purposes of this definition, "submitted"
|
|
55
|
+
means any form of electronic, verbal, or written communication sent
|
|
56
|
+
to the Licensor or its representatives, including but not limited to
|
|
57
|
+
communication on electronic mailing lists, source code control systems,
|
|
58
|
+
and issue tracking systems that are managed by, or on behalf of, the
|
|
59
|
+
Licensor for the purpose of discussing and improving the Work, but
|
|
60
|
+
excluding communication that is conspicuously marked or otherwise
|
|
61
|
+
designated in writing by the copyright owner as "Not a Contribution."
|
|
62
|
+
|
|
63
|
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
|
64
|
+
on behalf of whom a Contribution has been received by Licensor and
|
|
65
|
+
subsequently incorporated within the Work.
|
|
66
|
+
|
|
67
|
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
|
68
|
+
this License, each Contributor hereby grants to You a perpetual,
|
|
69
|
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
|
70
|
+
copyright license to reproduce, prepare Derivative Works of,
|
|
71
|
+
publicly display, publicly perform, sublicense, and distribute the
|
|
72
|
+
Work and such Derivative Works in Source or Object form.
|
|
73
|
+
|
|
74
|
+
3. Grant of Patent License. Subject to the terms and conditions of
|
|
75
|
+
this License, each Contributor hereby grants to You a perpetual,
|
|
76
|
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
|
77
|
+
(except as stated in this section) patent license to make, have made,
|
|
78
|
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
|
79
|
+
where such license applies only to those patent claims licensable
|
|
80
|
+
by such Contributor that are necessarily infringed by their
|
|
81
|
+
Contribution(s) alone or by combination of their Contribution(s)
|
|
82
|
+
with the Work to which such Contribution(s) was submitted. If You
|
|
83
|
+
institute patent litigation against any entity (including a
|
|
84
|
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
|
85
|
+
or a Contribution incorporated within the Work constitutes direct
|
|
86
|
+
or contributory patent infringement, then any patent licenses
|
|
87
|
+
granted to You under this License for that Work shall terminate
|
|
88
|
+
as of the date such litigation is filed.
|
|
89
|
+
|
|
90
|
+
4. Redistribution. You may reproduce and distribute copies of the
|
|
91
|
+
Work or Derivative Works thereof in any medium, with or without
|
|
92
|
+
modifications, and in Source or Object form, provided that You
|
|
93
|
+
meet the following conditions:
|
|
94
|
+
|
|
95
|
+
(a) You must give any other recipients of the Work or
|
|
96
|
+
Derivative Works a copy of this License; and
|
|
97
|
+
|
|
98
|
+
(b) You must cause any modified files to carry prominent notices
|
|
99
|
+
stating that You changed the files; and
|
|
100
|
+
|
|
101
|
+
(c) You must retain, in the Source form of any Derivative Works
|
|
102
|
+
that You distribute, all copyright, patent, trademark, and
|
|
103
|
+
attribution notices from the Source form of the Work,
|
|
104
|
+
excluding those notices that do not pertain to any part of
|
|
105
|
+
the Derivative Works; and
|
|
106
|
+
|
|
107
|
+
(d) If the Work includes a "NOTICE" text file as part of its
|
|
108
|
+
distribution, then any Derivative Works that You distribute must
|
|
109
|
+
include a readable copy of the attribution notices contained
|
|
110
|
+
within such NOTICE file, excluding those notices that do not
|
|
111
|
+
pertain to any part of the Derivative Works, in at least one
|
|
112
|
+
of the following places: within a NOTICE text file distributed
|
|
113
|
+
as part of the Derivative Works; within the Source form or
|
|
114
|
+
documentation, if provided along with the Derivative Works; or,
|
|
115
|
+
within a display generated by the Derivative Works, if and
|
|
116
|
+
wherever such third-party notices normally appear. The contents
|
|
117
|
+
of the NOTICE file are for informational purposes only and
|
|
118
|
+
do not modify the License. You may add Your own attribution
|
|
119
|
+
notices within Derivative Works that You distribute, alongside
|
|
120
|
+
or as an addendum to the NOTICE text from the Work, provided
|
|
121
|
+
that such additional attribution notices cannot be construed
|
|
122
|
+
as modifying the License.
|
|
123
|
+
|
|
124
|
+
You may add Your own copyright statement to Your modifications and
|
|
125
|
+
may provide additional or different license terms and conditions
|
|
126
|
+
for use, reproduction, or distribution of Your modifications, or
|
|
127
|
+
for any such Derivative Works as a whole, provided Your use,
|
|
128
|
+
reproduction, and distribution of the Work otherwise complies with
|
|
129
|
+
the conditions stated in this License.
|
|
130
|
+
|
|
131
|
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
|
132
|
+
any Contribution intentionally submitted for inclusion in the Work
|
|
133
|
+
by You to the Licensor shall be under the terms and conditions of
|
|
134
|
+
this License, without any additional terms or conditions.
|
|
135
|
+
Notwithstanding the above, nothing herein shall supersede or modify
|
|
136
|
+
the terms of any separate license agreement you may have executed
|
|
137
|
+
with Licensor regarding such Contributions.
|
|
138
|
+
|
|
139
|
+
6. Trademarks. This License does not grant permission to use the trade
|
|
140
|
+
names, trademarks, service marks, or product names of the Licensor,
|
|
141
|
+
except as required for reasonable and customary use in describing the
|
|
142
|
+
origin of the Work and reproducing the content of the NOTICE file.
|
|
143
|
+
|
|
144
|
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
|
145
|
+
agreed to in writing, Licensor provides the Work (and each
|
|
146
|
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
|
147
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
|
148
|
+
implied, including, without limitation, any warranties or conditions
|
|
149
|
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
|
150
|
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
|
151
|
+
appropriateness of using or redistributing the Work and assume any
|
|
152
|
+
risks associated with Your exercise of permissions under this License.
|
|
153
|
+
|
|
154
|
+
8. Limitation of Liability. In no event and under no legal theory,
|
|
155
|
+
whether in tort (including negligence), contract, or otherwise,
|
|
156
|
+
unless required by applicable law (such as deliberate and grossly
|
|
157
|
+
negligent acts) or agreed to in writing, shall any Contributor be
|
|
158
|
+
liable to You for damages, including any direct, indirect, special,
|
|
159
|
+
incidental, or consequential damages of any character arising as a
|
|
160
|
+
result of this License or out of the use or inability to use the
|
|
161
|
+
Work (including but not limited to damages for loss of goodwill,
|
|
162
|
+
work stoppage, computer failure or malfunction, or any and all
|
|
163
|
+
other commercial damages or losses), even if such Contributor
|
|
164
|
+
has been advised of the possibility of such damages.
|
|
165
|
+
|
|
166
|
+
9. Accepting Warranty or Additional Liability. While redistributing
|
|
167
|
+
the Work or Derivative Works thereof, You may choose to offer,
|
|
168
|
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
|
169
|
+
or other liability obligations and/or rights consistent with this
|
|
170
|
+
License. However, in accepting such obligations, You may act only
|
|
171
|
+
on Your own behalf and on Your sole responsibility, not on behalf
|
|
172
|
+
of any other Contributor, and only if You agree to indemnify,
|
|
173
|
+
defend, and hold each Contributor harmless for any liability
|
|
174
|
+
incurred by, or claims asserted against, such Contributor by reason
|
|
175
|
+
of your accepting any such warranty or additional liability.
|
|
176
|
+
|
|
177
|
+
END OF TERMS AND CONDITIONS
|
|
178
|
+
|
|
179
|
+
APPENDIX: How to apply the Apache License to your work.
|
|
180
|
+
|
|
181
|
+
To apply the Apache License to your work, attach the following
|
|
182
|
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
|
183
|
+
replaced with your own identifying information. (Don't include
|
|
184
|
+
the brackets!) The text should be enclosed in the appropriate
|
|
185
|
+
comment syntax for the file format. We also recommend that a
|
|
186
|
+
file or class name and description of purpose be included on the
|
|
187
|
+
same "printed page" as the copyright notice for easier
|
|
188
|
+
identification within third-party archives.
|
|
189
|
+
|
|
190
|
+
Copyright [yyyy] [name of copyright owner]
|
|
191
|
+
|
|
192
|
+
Licensed under the Apache License, Version 2.0 (the "License");
|
|
193
|
+
you may not use this file except in compliance with the License.
|
|
194
|
+
You may obtain a copy of the License at
|
|
195
|
+
|
|
196
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
197
|
+
|
|
198
|
+
Unless required by applicable law or agreed to in writing, software
|
|
199
|
+
distributed under the License is distributed on an "AS IS" BASIS,
|
|
200
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
201
|
+
See the License for the specific language governing permissions and
|
|
202
|
+
limitations under the License.
|