datahub-bq-connector-session-patch 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,36 @@
1
+ """Recover unqualified BigQuery reads that DataHub silently discards.
2
+
3
+ Full diagnosis, design and verification live in the DataVantage docs repo:
4
+ ``docs/datahub-unqualified-read-attribution.md``.
5
+
6
+ Recipes use the ``datahub_bq_connector_session_patch`` source type, registered by this package's
7
+ entry point. ``install()`` and ``install_on_aggregator()`` are for standalone scripts
8
+ and tests that drive a pipeline or an aggregator directly.
9
+ """
10
+
11
+ from datahub_bq_connector_session_patch.patch import armed, check_datahub_version, install, is_armed
12
+ from datahub_bq_connector_session_patch.resolver import (
13
+ SessionPatchStats,
14
+ build_index,
15
+ install_on_aggregator,
16
+ )
17
+
18
+ __all__ = [
19
+ "SessionPatchStats",
20
+ "armed",
21
+ "build_index",
22
+ "check_datahub_version",
23
+ "install",
24
+ "install_on_aggregator",
25
+ "is_armed",
26
+ ]
27
+
28
+
29
+ def __getattr__(name: str):
30
+ # Imported lazily: it pulls in datahub's bigquery extras, which a test or a script
31
+ # using only the resolver need not have.
32
+ if name == "BigQuerySessionPatchSource":
33
+ from datahub_bq_connector_session_patch.source import BigQuerySessionPatchSource
34
+
35
+ return BigQuerySessionPatchSource
36
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
@@ -0,0 +1,201 @@
1
+ """Compose the resolver onto ``BigQueryQueriesExtractor``.
2
+
3
+ The extractor is constructed deep inside ``BigqueryV2Source.get_workunits_internal``,
4
+ so there is no injection point -- the composition has to happen on the class. To keep a
5
+ stock ``type: bigquery`` recipe stock even when this package is installed in the same
6
+ interpreter, the patched ``__init__`` is inert unless it is explicitly armed. The
7
+ ``datahub_bq_connector_session_patch`` source arms it around its own run and nothing else.
8
+ """
9
+
10
+ import contextlib
11
+ import contextvars
12
+ import inspect
13
+ import logging
14
+ from typing import Any, Collection, Iterator, Optional, Tuple
15
+
16
+ from datahub_bq_connector_session_patch.resolver import SessionPatchStats, install_on_aggregator
17
+
18
+ logger = logging.getLogger(__name__)
19
+
20
+ #: Versions of acryl-datahub this patch has been tested against. It reaches into private
21
+ #: internals (``BigQueryQueriesExtractor.__init__``, ``SqlParsingAggregator``), so a
22
+ #: version outside this range is reported rather than trusted. Deliberately NOT a
23
+ #: packaging dependency: this wheel installs into a DataHub executor's own interpreter
24
+ #: beside its pinned acryl-datahub, and declaring it would let pip re-resolve that pin.
25
+ MIN_ACRYL_DATAHUB = (1, 6)
26
+ MAX_ACRYL_DATAHUB_EXCLUSIVE = (1, 8)
27
+
28
+ _ARMED: contextvars.ContextVar = contextvars.ContextVar("datahub_bq_connector_session_patch_armed", default=False)
29
+
30
+
31
+ @contextlib.contextmanager
32
+ def armed() -> Iterator[None]:
33
+ """Arm the patch for the duration of the block."""
34
+ token = _ARMED.set(True)
35
+ try:
36
+ yield
37
+ finally:
38
+ _ARMED.reset(token)
39
+
40
+
41
+ def is_armed() -> bool:
42
+ return _ARMED.get()
43
+
44
+
45
+ def _version_tuple(raw: str) -> tuple:
46
+ parts = []
47
+ for chunk in raw.split(".")[:3]:
48
+ digits = "".join(c for c in chunk if c.isdigit())
49
+ if not digits:
50
+ break
51
+ parts.append(int(digits))
52
+ return tuple(parts)
53
+
54
+
55
+ def check_datahub_version(report: Optional[Any] = None) -> Optional[str]:
56
+ """Warn if the installed acryl-datahub is outside the tested range.
57
+
58
+ Returns the message when out of range, else None. Never raises: a version check
59
+ failing closed would take down an ingestion run over its own metadata lookup.
60
+ """
61
+ try:
62
+ from importlib.metadata import version
63
+
64
+ raw = version("acryl-datahub")
65
+ except Exception:
66
+ return None
67
+
68
+ found = _version_tuple(raw)
69
+ if not found:
70
+ return None
71
+ if MIN_ACRYL_DATAHUB <= found[: len(MIN_ACRYL_DATAHUB)] and found < MAX_ACRYL_DATAHUB_EXCLUSIVE:
72
+ return None
73
+
74
+ lo = ".".join(str(n) for n in MIN_ACRYL_DATAHUB)
75
+ hi = ".".join(str(n) for n in MAX_ACRYL_DATAHUB_EXCLUSIVE)
76
+ message = (
77
+ f"datahub-bq-connector-session-patch is tested against acryl-datahub >={lo},<{hi}; found {raw}. "
78
+ "It patches private internals -- re-run the test suite before trusting this run."
79
+ )
80
+ if report is not None:
81
+ report.warning("datahub-bq-connector-session-patch: untested acryl-datahub version", context=message)
82
+ else:
83
+ logger.warning(message)
84
+ return message
85
+
86
+
87
+ def describe_outcome(stats) -> Tuple[str, str, str]:
88
+ """(level, title, context) for the end-of-run report. Pure, so it can be tested.
89
+
90
+ The index is built lazily on the first classified upstream, so a run that saw no
91
+ queries at all leaves ``index_size`` at 0 as well. Reporting that as an empty
92
+ ``discovered_tables`` points the operator at the wrong subsystem.
93
+ """
94
+ queries_seen = stats.queries_passed + stats.queries_skipped + stats.queries_mixed
95
+ if queries_seen == 0:
96
+ return (
97
+ "warning",
98
+ "datahub-bq-connector-session-patch saw NO queries",
99
+ "the query log yielded nothing, so the resolution index was never built -- "
100
+ "check the audit window, the user filters and the region qualifiers, not "
101
+ "discovered_tables",
102
+ )
103
+ if stats.index_size == 0:
104
+ return (
105
+ "warning",
106
+ "datahub-bq-connector-session-patch built an EMPTY resolution index",
107
+ f"{queries_seen} queries seen but discovered_tables was empty -- "
108
+ "the patch recovered nothing",
109
+ )
110
+ if stats.rewritten == 0:
111
+ return (
112
+ "warning",
113
+ "datahub-bq-connector-session-patch rewrote nothing",
114
+ f"index={stats.index_size} names but no _SESSION upstreams matched",
115
+ )
116
+ return (
117
+ "info",
118
+ "datahub-bq-connector-session-patch recovered unqualified reads",
119
+ str(stats),
120
+ )
121
+
122
+
123
+ def _live_discovered_tables(extractor, original_init, args: tuple, kwargs: dict) -> Collection[str]:
124
+ """The LIVE collection the source passed in, not ``__init__``'s snapshot.
125
+
126
+ ``BigqueryV2Source`` passes ``discovered_tables=self.bq_schema_extractor.table_refs``
127
+ -- a set that is still empty at construction and fills during the schema pass. The
128
+ extractor copies it into ``self.discovered_tables`` immediately, so reading that
129
+ attribute yields an empty index and a silently useless run.
130
+
131
+ Bound through the real signature rather than a hand-counted argument index, so a
132
+ positional call site or an upstream signature change cannot silently pick the wrong
133
+ parameter -- the failure mode this whole patch exists to avoid.
134
+ """
135
+ live = None
136
+ try:
137
+ bound = inspect.signature(original_init).bind(extractor, *args, **kwargs)
138
+ live = bound.arguments.get("discovered_tables")
139
+ except TypeError:
140
+ live = kwargs.get("discovered_tables")
141
+ if live is None:
142
+ live = extractor.discovered_tables or ()
143
+ return live
144
+
145
+
146
+ def install(always_on: bool = False) -> None:
147
+ """Patch ``BigQueryQueriesExtractor`` so armed runs get the fix. Idempotent.
148
+
149
+ ``always_on`` arms the patch process-wide -- for standalone scripts that drive a
150
+ pipeline directly. Recipes should use the ``datahub_bq_connector_session_patch`` source instead,
151
+ which arms only its own run.
152
+ """
153
+ from datahub.ingestion.source.bigquery_v2 import queries_extractor as qe
154
+
155
+ if always_on:
156
+ _ARMED.set(True)
157
+
158
+ if getattr(qe.BigQueryQueriesExtractor, "_datahub_bq_connector_session_patch_installed", False):
159
+ return
160
+ original_init = qe.BigQueryQueriesExtractor.__init__
161
+
162
+ def __init__(self, *args, **kwargs):
163
+ original_init(self, *args, **kwargs)
164
+ if not is_armed():
165
+ return
166
+
167
+ check_datahub_version(self.structured_report)
168
+ live = _live_discovered_tables(self, original_init, args, kwargs)
169
+
170
+ def discovered() -> Collection[str]:
171
+ standardize = self.identifiers.standardize_identifier_case
172
+ return {standardize(name) for name in (live or ())}
173
+
174
+ stats: SessionPatchStats = install_on_aggregator(self.aggregator, discovered)
175
+ self.datahub_bq_connector_session_patch_stats = stats
176
+ logger.info("datahub-bq-connector-session-patch: armed, index builds on first query")
177
+
178
+ original_close = self.close
179
+
180
+ def close() -> None:
181
+ # Both historical failures of this patch were silent: the run completed and
182
+ # rewrote nothing. Make that loud instead -- but name the right subsystem.
183
+ try:
184
+ level, title, context = describe_outcome(stats)
185
+ getattr(self.structured_report, level)(title, context=context)
186
+ if level == "info":
187
+ logger.info("datahub-bq-connector-session-patch: %s", stats)
188
+ logger.info(
189
+ "datahub-bq-connector-session-patch: per reader project\n%s",
190
+ stats.per_project_table(),
191
+ )
192
+ finally:
193
+ # The aggregator holds a file-backed sqlite connection. A failure while
194
+ # REPORTING must never be the reason it is left open -- least of all in
195
+ # a function whose whole job is making silent failures loud.
196
+ original_close()
197
+
198
+ self.close = close
199
+
200
+ qe.BigQueryQueriesExtractor.__init__ = __init__
201
+ qe.BigQueryQueriesExtractor._datahub_bq_connector_session_patch_installed = True
@@ -0,0 +1,298 @@
1
+ """Resolve the ``_SESSION`` upstreams DataHub builds for unqualified BigQuery reads.
2
+
3
+ ``BigQueryQueriesExtractor._parse_audit_log_row`` sets ``default_schema="_SESSION"``
4
+ (queries_extractor.py:542,558), so any table name a query does not qualify resolves to
5
+ ``{billing_project}._SESSION.{table}``. ``is_temp_table()`` then flags it, because the
6
+ dataset starts with ``temp_table_dataset_prefix`` (default ``"_"``), and the aggregator
7
+ drops it -- for usage (sql_parsing_aggregator.py:1022) and for the query entity (:1693).
8
+ The net effect is that every unqualified read in the estate produces no usage, no query
9
+ entity and no lineage, even though the parser saw the table name.
10
+
11
+ This rewrites those upstreams back to the real object by matching the bare table name
12
+ against the tables DataHub discovered during its schema pass. A name matching more than
13
+ one discovered table is left untouched -- dropped exactly as before, never guessed.
14
+ """
15
+
16
+ import dataclasses
17
+ import logging
18
+ import re
19
+ from collections import defaultdict
20
+ from typing import Callable, Collection, Dict, Optional, Set, Tuple, Union
21
+
22
+ logger = logging.getLogger(__name__)
23
+
24
+ _SESSION = "_session"
25
+
26
+ #: Outcome of classifying one upstream URN. Returned alongside the resolved URN so the
27
+ #: caller can count per-reference volume without re-running (and re-counting) the lookup.
28
+ PASSTHROUGH = "passthrough"
29
+ REWRITTEN = "rewritten"
30
+ AMBIGUOUS = "ambiguous"
31
+ UNKNOWN = "unknown"
32
+
33
+
34
+ @dataclasses.dataclass
35
+ class SessionPatchStats:
36
+ """Distinct-table counters, plus the per-reference volume behind each.
37
+
38
+ The distinct counters answer "how much of the catalog did we recover"; the
39
+ ``*_refs`` counters answer "how many reads did that attribute". Both are needed to
40
+ judge a run: a single hot table read 2,000 times is one rewrite and 2,000 references.
41
+ """
42
+
43
+ # Distinct. NOT distinct _SESSION URNs: those embed the reader (billing) project,
44
+ # so one table read from five projects would count five times. `rewritten` counts
45
+ # distinct resolved TARGET tables; `ambiguous`/`unknown` count distinct bare NAMES.
46
+ rewritten: int = 0
47
+ ambiguous: int = 0
48
+ unknown: int = 0
49
+ # Volume: upstream references seen across all queries. Counted only over
50
+ # ``parsed.upstreams``; the column_usage remap reuses the memoised result, so it
51
+ # can never inflate these.
52
+ rewritten_refs: int = 0
53
+ ambiguous_refs: int = 0
54
+ unknown_refs: int = 0
55
+ queries_passed: int = 0
56
+ queries_skipped: int = 0
57
+ queries_mixed: int = 0
58
+ index_size: int = 0
59
+ ambiguous_names: int = 0
60
+ #: reader (billing) project -> {"rewritten"|"ambiguous"|"unknown": reference count}.
61
+ #: Keyed by the project that RAN the query, not the one the table lives in: when the
62
+ #: sidecar's scope is widened, the question being asked is whose traffic benefits.
63
+ refs_by_reader_project: Dict[str, Dict[str, int]] = dataclasses.field(default_factory=dict)
64
+
65
+ def record_ref(self, project: Optional[str], outcome: str) -> None:
66
+ if project is None:
67
+ return
68
+ self.refs_by_reader_project.setdefault(project, {}).setdefault(outcome, 0)
69
+ self.refs_by_reader_project[project][outcome] += 1
70
+
71
+ def per_project_table(self) -> str:
72
+ """One line per reader project, worst recovery first. For run reports."""
73
+ if not self.refs_by_reader_project:
74
+ return " (no _SESSION references seen)"
75
+ rows = []
76
+ for project, counts in self.refs_by_reader_project.items():
77
+ got = counts.get(REWRITTEN, 0)
78
+ amb = counts.get(AMBIGUOUS, 0)
79
+ unk = counts.get(UNKNOWN, 0)
80
+ total = got + amb + unk
81
+ pct = (100.0 * got / total) if total else 0.0
82
+ rows.append((pct, project, got, amb, unk, total))
83
+ rows.sort()
84
+ return "\n".join(
85
+ f" {project:<34} recovered={got:<7} ambiguous={amb:<7} unknown={unk:<7} "
86
+ f"of {total:<7} ({pct:.1f}%)"
87
+ for pct, project, got, amb, unk, total in rows
88
+ )
89
+
90
+ def __str__(self) -> str:
91
+ return (
92
+ f"rewritten={self.rewritten} tables ({self.rewritten_refs} refs) "
93
+ f"ambiguous={self.ambiguous} names ({self.ambiguous_refs} refs) "
94
+ f"unknown={self.unknown} names ({self.unknown_refs} refs) "
95
+ f"queries_passed={self.queries_passed} queries_skipped={self.queries_skipped} "
96
+ f"queries_mixed={self.queries_mixed} "
97
+ f"(index={self.index_size} unique names, {self.ambiguous_names} ambiguous names)"
98
+ )
99
+
100
+
101
+ _TABLE_REF_RE = re.compile(r"^projects/([^/]+)/datasets/([^/]+)/tables/(.+)$")
102
+
103
+
104
+ def _normalise(ref: str) -> Optional[str]:
105
+ """Discovered tables arrive as ``projects/P/datasets/D/tables/T``; URNs use P.D.T."""
106
+ match = _TABLE_REF_RE.match(ref)
107
+ if match:
108
+ return ".".join(match.groups())
109
+ parts = ref.split(".")
110
+ return ".".join(parts[-3:]) if len(parts) >= 3 else None
111
+
112
+
113
+ def build_index(discovered: Collection[str]) -> Tuple[Dict[str, str], Set[str]]:
114
+ """bare table name -> full name, only where the bare name is unambiguous.
115
+
116
+ Names matching several discovered tables go to the ambiguous set and are never
117
+ resolved. Narrowing the recipe's scope *removes* alternatives rather than resolving
118
+ between them, so the index must be built over a scope broad enough that real
119
+ collisions stay visible -- see the design doc, section 9.
120
+ """
121
+ candidates: Dict[str, Set[str]] = defaultdict(set)
122
+ for ref in discovered:
123
+ full = _normalise(ref)
124
+ if full:
125
+ candidates[full.split(".")[-1].lower()].add(full)
126
+ index = {name: next(iter(v)) for name, v in candidates.items() if len(v) == 1}
127
+ ambiguous = {name for name, v in candidates.items() if len(v) > 1}
128
+ return index, ambiguous
129
+
130
+
131
+ def _split_urn(urn: str) -> Optional[Tuple[str, str, str]]:
132
+ try:
133
+ inner = urn[urn.index("(") + 1 : urn.rindex(")")]
134
+ platform, name, env = inner.split(",")
135
+ return platform, name, env
136
+ except Exception:
137
+ return None
138
+
139
+
140
+ def _session_table(name: str) -> Optional[str]:
141
+ """Bare table name if this is a ``_SESSION``-qualified name, else None."""
142
+ parts = name.split(".")
143
+ if len(parts) < 3 or parts[-2].lower() != _SESSION:
144
+ return None
145
+ return parts[-1].lower()
146
+
147
+
148
+ DiscoveredTables = Union[Collection[str], Callable[[], Collection[str]]]
149
+
150
+
151
+ def install_on_aggregator(
152
+ aggregator,
153
+ discovered_tables: DiscoveredTables,
154
+ disjoint_only: bool = True,
155
+ ) -> SessionPatchStats:
156
+ """Wrap an aggregator so ``_SESSION`` upstreams resolve to real datasets.
157
+
158
+ With ``disjoint_only`` (the default) the aggregator sees ONLY the upstreams this
159
+ resolver actually recovered, and queries where nothing was recovered are dropped
160
+ entirely. That keeps a sidecar recipe strictly complementary to the main one: it
161
+ emits exactly what the main recipe structurally cannot, and nothing else, so the two
162
+ never write usage for the same (dataset, day) and widening the sidecar's scope is
163
+ harmless.
164
+
165
+ ``discovered_tables`` may be a collection or a zero-arg callable returning one.
166
+ The index is built lazily on first use: BigqueryV2Source hands the extractor a live
167
+ reference to ``bq_schema_extractor.table_refs``, which is still empty when the
168
+ extractor is constructed and only fills during the schema pass.
169
+ """
170
+ stats = SessionPatchStats()
171
+ state: Dict[str, object] = {}
172
+ # Distinct-ness is measured on what the counter CLAIMS to count, not on the cache
173
+ # key: the cache is keyed by the full _SESSION URN, which embeds the reader project.
174
+ seen_targets: Set[str] = set()
175
+ seen_ambiguous: Set[str] = set()
176
+ seen_unknown: Set[str] = set()
177
+ # Memoised so each distinct URN is classified -- and counted -- exactly once. The
178
+ # column_usage remap below resolves the same URNs a second time; without this the
179
+ # counters would report resolution calls rather than distinct tables.
180
+ cache: Dict[str, Tuple[str, str]] = {}
181
+
182
+ def get_index() -> Tuple[Dict[str, str], Set[str]]:
183
+ if "index" not in state:
184
+ tables = discovered_tables() if callable(discovered_tables) else discovered_tables
185
+ index, ambiguous = build_index(tables or ())
186
+ stats.index_size, stats.ambiguous_names = len(index), len(ambiguous)
187
+ state["index"] = (index, ambiguous)
188
+ logger.info("datahub-bq-connector-session-patch: resolution index built (%s)", stats)
189
+ return state["index"] # type: ignore[return-value]
190
+
191
+ original = aggregator.add_preparsed_query
192
+
193
+ def classify(urn: str) -> Tuple[str, str, Optional[str]]:
194
+ """(resolved_urn, outcome, reader_project).
195
+
196
+ Counts DISTINCT outcomes on first sight only; per-reference volume is counted by
197
+ the caller, so the column_usage remap below cannot inflate either.
198
+ """
199
+ hit = cache.get(urn)
200
+ if hit is not None:
201
+ return hit
202
+
203
+ index, ambiguous = get_index()
204
+ result: Tuple[str, str, Optional[str]] = (urn, PASSTHROUGH, None)
205
+ split = _split_urn(urn)
206
+ if split is not None:
207
+ platform, name, env = split
208
+ table = _session_table(name)
209
+ if table is not None:
210
+ parts = name.split(".")
211
+ # `{billing_project}._SESSION.{table}` -- the project that ran the query.
212
+ reader = parts[-3] if len(parts) >= 3 else None
213
+ target = index.get(table)
214
+ if target is not None:
215
+ prefix = parts[:-3]
216
+ resolved = ".".join(prefix + [target])
217
+ target_urn = f"urn:li:dataset:({platform},{resolved},{env})"
218
+ seen_targets.add(target_urn)
219
+ stats.rewritten = len(seen_targets)
220
+ result = (target_urn, REWRITTEN, reader)
221
+ elif table in ambiguous:
222
+ seen_ambiguous.add(table)
223
+ stats.ambiguous = len(seen_ambiguous)
224
+ logger.debug("datahub-bq-connector-session-patch: %r is ambiguous, leaving as temp", table)
225
+ result = (urn, AMBIGUOUS, reader)
226
+ else:
227
+ seen_unknown.add(table)
228
+ stats.unknown = len(seen_unknown)
229
+ result = (urn, UNKNOWN, reader)
230
+
231
+ cache[urn] = result
232
+ return result
233
+
234
+ def resolve(urn: str) -> str:
235
+ return classify(urn)[0]
236
+
237
+ def add_preparsed_query(parsed, *args, **kwargs):
238
+ recovered = set()
239
+ already_resolvable = set()
240
+ resolved = []
241
+ for upstream in parsed.upstreams:
242
+ new, outcome, reader = classify(upstream)
243
+ resolved.append(new)
244
+ if outcome == PASSTHROUGH:
245
+ # A real, already-qualified reference -- the main recipe resolves this
246
+ # one itself and will emit a query entity carrying it.
247
+ already_resolvable.add(new)
248
+ if outcome == REWRITTEN:
249
+ stats.rewritten_refs += 1
250
+ recovered.add(new)
251
+ elif outcome == AMBIGUOUS:
252
+ stats.ambiguous_refs += 1
253
+ elif outcome == UNKNOWN:
254
+ stats.unknown_refs += 1
255
+ if outcome != PASSTHROUGH:
256
+ stats.record_ref(reader, outcome)
257
+
258
+ if disjoint_only:
259
+ if not recovered:
260
+ # Every reference was already resolvable -- the main recipe has it.
261
+ stats.queries_skipped += 1
262
+ return
263
+ if already_resolvable:
264
+ # MIXED query: some references were already resolvable, some we
265
+ # recovered. Disjointness does NOT hold here. The query entity is keyed
266
+ # by get_query_fingerprint(sql), a pure function of the SQL text, so both
267
+ # pipelines emit the SAME urn:li:query:<fp> -- and querySubjects is a
268
+ # versioned aspect, not timeseries. Emitting our narrowed subject set
269
+ # would overwrite the main recipe's, and the entity would flip between
270
+ # the two on every scheduled run. Passing the union instead would
271
+ # double-book usage for the already-resolvable half. Neither is
272
+ # acceptable, so the query is left entirely to the main recipe.
273
+ stats.queries_mixed += 1
274
+ return
275
+ parsed.upstreams = sorted(recovered)
276
+ else:
277
+ parsed.upstreams = resolved
278
+ if parsed.downstream:
279
+ parsed.downstream = resolve(parsed.downstream)
280
+
281
+ if parsed.column_usage:
282
+ # Union rather than assign: two keys can collapse onto one URN (a self-join
283
+ # referencing the same table both qualified and unqualified), and dict
284
+ # comprehension would silently keep only the last column set.
285
+ remapped: Dict[str, Set[str]] = {}
286
+ for key, columns in parsed.column_usage.items():
287
+ remapped.setdefault(resolve(key), set()).update(columns or ())
288
+ parsed.column_usage = (
289
+ {k: v for k, v in remapped.items() if k in recovered}
290
+ if disjoint_only
291
+ else remapped
292
+ )
293
+
294
+ stats.queries_passed += 1
295
+ return original(parsed, *args, **kwargs)
296
+
297
+ aggregator.add_preparsed_query = add_preparsed_query
298
+ return stats
@@ -0,0 +1,100 @@
1
+ """``datahub_bq_connector_session_patch`` -- the sidecar source type.
2
+
3
+ A ``BigqueryV2Source`` that arms the resolver for its own run and nothing else. Point a
4
+ narrow sidecar recipe at ``type: datahub_bq_connector_session_patch``; the production recipe keeps
5
+ ``type: bigquery`` and is unaffected even though both interpreters have this installed.
6
+
7
+ It also refuses to start on the two misconfigurations that are actively destructive
8
+ rather than merely wrong -- see ``_check_sidecar_safety``.
9
+ """
10
+
11
+ import logging
12
+ from typing import Iterable, List
13
+
14
+ from datahub.ingestion.api.common import PipelineContext
15
+ from datahub.ingestion.api.decorators import (
16
+ SupportStatus,
17
+ config_class,
18
+ platform_name,
19
+ support_status,
20
+ )
21
+ from datahub.ingestion.api.workunit import MetadataWorkUnit
22
+ from datahub.ingestion.source.bigquery_v2.bigquery import BigqueryV2Source
23
+ from datahub.ingestion.source.bigquery_v2.bigquery_config import BigQueryV2Config
24
+
25
+ from datahub_bq_connector_session_patch.patch import armed, install
26
+
27
+ logger = logging.getLogger(__name__)
28
+
29
+
30
+ class SidecarMisconfigured(ValueError):
31
+ """The recipe would damage the catalog. Raised before any metadata is read."""
32
+
33
+
34
+ def _check_sidecar_safety(config: BigQueryV2Config) -> None:
35
+ """Refuse the two settings that make this sidecar destructive.
36
+
37
+ Both were reproduced against a live catalog; neither is a style preference:
38
+
39
+ * ``remove_stale_metadata`` -- the sidecar writes a ``status`` aspect on every
40
+ dataset it produces usage for, which enrols them in *its* stale-removal
41
+ checkpoint. Its URN set is usage-driven, so any dataset not read in the window
42
+ falls out of the checkpoint and gets soft-deleted -- datasets the main recipe
43
+ owns. This is normal operation, not an edge case.
44
+ * ``include_table_lineage`` -- a mis-resolved write would create a wrong edge into a
45
+ real table, which is worse than the missing usage this package exists to fix.
46
+ """
47
+ problems: List[str] = []
48
+
49
+ stateful = config.stateful_ingestion
50
+ # remove_stale_metadata defaults to True, so only the combination bites: stale
51
+ # removal cannot run at all when stateful ingestion is off, and refusing that
52
+ # combination would block every file-sink measurement run.
53
+ if (
54
+ stateful is not None
55
+ and getattr(stateful, "enabled", False)
56
+ and getattr(stateful, "remove_stale_metadata", False)
57
+ ):
58
+ problems.append(
59
+ "stateful_ingestion.remove_stale_metadata must be false. This sidecar's URN "
60
+ "set is usage-driven, so stale removal soft-deletes datasets the main "
61
+ "recipe owns, on every run."
62
+ )
63
+
64
+ if config.include_table_lineage:
65
+ problems.append(
66
+ "include_table_lineage must be false. Resolved names are used for usage "
67
+ "attribution only; emitting lineage from a resolved name risks a wrong edge "
68
+ "into a real table."
69
+ )
70
+
71
+ if problems:
72
+ raise SidecarMisconfigured(
73
+ "datahub_bq_connector_session_patch refuses to run with this recipe:\n - "
74
+ + "\n - ".join(problems)
75
+ )
76
+
77
+
78
+ @platform_name("BigQuery")
79
+ @config_class(BigQueryV2Config)
80
+ @support_status(SupportStatus.BETA)
81
+ class BigQuerySessionPatchSource(BigqueryV2Source):
82
+ """BigQuery source that recovers reads whose SQL did not qualify its table names."""
83
+
84
+ def __init__(self, ctx: PipelineContext, config: BigQueryV2Config):
85
+ _check_sidecar_safety(config)
86
+ super().__init__(ctx, config)
87
+
88
+ @classmethod
89
+ def create(cls, config_dict: dict, ctx: PipelineContext) -> "BigQuerySessionPatchSource":
90
+ config = BigQueryV2Config.model_validate(config_dict)
91
+ return cls(ctx, config)
92
+
93
+ def get_workunits_internal(self) -> Iterable[MetadataWorkUnit]:
94
+ # install() patches the class; armed() is what makes the patch do anything, and
95
+ # it covers only this generator's execution. The extractor is constructed inside
96
+ # super().get_workunits_internal(), so both have to be in place before we
97
+ # delegate.
98
+ install()
99
+ with armed():
100
+ yield from super().get_workunits_internal()
@@ -0,0 +1,287 @@
1
+ Metadata-Version: 2.5
2
+ Name: datahub-bq-connector-session-patch
3
+ Version: 0.1.0
4
+ Summary: DataHub BigQuery sidecar source: attribute reads whose SQL does not qualify its table names
5
+ Project-URL: Homepage, https://github.com/gutro/dv-datahub/tree/master/datahub-bq-connector-session-patch
6
+ Project-URL: Repository, https://github.com/gutro/dv-datahub
7
+ Project-URL: Issues, https://github.com/gutro/dv-datahub/issues
8
+ Author: DataVantage Platform
9
+ License-Expression: Apache-2.0
10
+ License-File: LICENSE
11
+ Keywords: bigquery,datahub,ingestion,lineage,usage
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Topic :: Database
16
+ Requires-Python: >=3.9
17
+ Provides-Extra: test
18
+ Requires-Dist: acryl-datahub[bigquery]<1.8,>=1.6; extra == 'test'
19
+ Requires-Dist: pytest>=7; extra == 'test'
20
+ Requires-Dist: pyyaml; extra == 'test'
21
+ Description-Content-Type: text/markdown
22
+
23
+ # datahub-bq-connector-session-patch
24
+
25
+ A DataHub BigQuery **sidecar source** that attributes reads whose SQL did not qualify
26
+ its table names — the ones the stock connector sees, misfiles and silently discards.
27
+
28
+ Full diagnosis, design, verification and limits:
29
+ `datavantage/docs/datahub-unqualified-read-attribution.md`.
30
+
31
+ ## The problem
32
+
33
+ A client with a BigQuery *default dataset* set can write `SELECT ... FROM Bets_Details`.
34
+ BigQuery fills in the blank at run time, but never records it: `INFORMATION_SCHEMA.JOBS`
35
+ carries the SQL text and the billing project, not the job's default dataset.
36
+
37
+ DataHub's connector substitutes `_SESSION` for the missing dataset
38
+ (`queries_extractor.py:542,558`), on the assumption that a bare name is usually a
39
+ temporary table. Its own temp-table rule then flags anything whose dataset starts with
40
+ `_` (`:309`), and the aggregator drops it — for usage
41
+ (`sql_parsing_aggregator.py:1022`) and for the query entity (`:1693`).
42
+
43
+ So the connector *sees* the table name, guesses the wrong container, and discards the
44
+ result. Nothing is reported: the sibling branch of `is_temp_table` records what it drops,
45
+ this one returns `True` in silence. Measured here: one client, 2,866 jobs in 7 days,
46
+ 100% unqualified, 0 usage rows.
47
+
48
+ **A `0` in DataHub usage is therefore ambiguous** — "nobody reads this", or "read
49
+ constantly, with unqualified SQL". Decommissioning decisions depend on telling those apart.
50
+
51
+ ## What this does
52
+
53
+ For upstreams that landed in `_SESSION`, it resolves the bare table name against the
54
+ tables DataHub discovered during its own schema pass. A name matching more than one
55
+ discovered table is **left untouched** — dropped exactly as before, never guessed.
56
+
57
+ It ships as a distinct source type, so the production recipe stays stock:
58
+
59
+ ```yaml
60
+ source:
61
+ type: datahub_bq_connector_session_patch
62
+ ```
63
+
64
+ ## Installing
65
+
66
+ Per-recipe, via **Extra Pip Libraries** (DataHub UI) or `extra_pip_requirements`. That
67
+ field is scoped to one recipe, which is what keeps the production recipe unaffected.
68
+
69
+ ```
70
+ datahub-bq-connector-session-patch
71
+ ```
72
+
73
+ The wheel declares **no dependencies** — it imports only `datahub.*`, which any
74
+ environment running a recipe already has. Declaring `acryl-datahub` would let pip
75
+ re-resolve the executor's own pinned copy while installing this. The tested range
76
+ (`>=1.6,<1.8`) is enforced at runtime instead: an out-of-range executor raises a
77
+ structured warning into the ingestion report rather than failing the run.
78
+
79
+ ## The sidecar recipe
80
+
81
+ ```yaml
82
+ pipeline_name: bq-session-attribution # its OWN name — never the main recipe's
83
+ source:
84
+ type: datahub_bq_connector_session_patch
85
+ config:
86
+ project_id_pattern:
87
+ allow:
88
+ - '^dv-prod-eu-w1(-.+)?-data$'
89
+ - '^dv-ext-prod-eu-w1(-.+)?-data$'
90
+ - '^dv-prod-eu-w1(-.+)?-comp\d+$'
91
+ include_tables: false # lightweight discovery still fills table_refs
92
+ include_views: false
93
+ include_table_lineage: false # refused if true — see below
94
+ include_usage_statistics: true
95
+ use_queries_v2: true
96
+ include_queries: true
97
+ include_query_usage_statistics: true
98
+ region_qualifiers: ['region-europe-west1']
99
+ start_time: '-7 days'
100
+ enable_stateful_time_window: true
101
+ stateful_ingestion:
102
+ enabled: true # the time-window watermark
103
+ remove_stale_metadata: false # refused if true — see below
104
+ env: PROD
105
+ ```
106
+
107
+ ### Two settings this source refuses to start with
108
+
109
+ Both were reproduced against a live catalog. They are not style preferences, so they are
110
+ enforced in code rather than left to the YAML:
111
+
112
+ | setting | why it is refused |
113
+ |---|---|
114
+ | `stateful_ingestion.remove_stale_metadata: true` | The sidecar writes a `status` aspect on every dataset it produces usage for, enrolling them in *its* stale-removal checkpoint. Its URN set is **usage-driven**, so any dataset not read that day falls out of the checkpoint and is soft-deleted — datasets the main recipe owns. Normal operation, not an edge case. |
115
+ | `include_table_lineage: true` | Resolved names drive usage attribution only. A mis-resolved write would create a **wrong edge into a real table**, which is worse than the missing usage this package exists to fix. |
116
+
117
+ `remove_stale_metadata` defaults to `true`, so it must be set explicitly. Only the
118
+ combination bites — with `stateful_ingestion.enabled: false` there is no checkpoint and
119
+ nothing is refused.
120
+
121
+ ### Scope it to the readers, not to minimise ambiguity
122
+
123
+ `dataset_pattern` should be **wide**. "Ambiguous" means the name matches more than one
124
+ entry in `discovered_tables` *within the recipe's scope*, so narrowing scope removes the
125
+ alternatives rather than resolving between them. If a reader's default dataset is
126
+ `prod_spain_*` but the index holds only `prod_malta_*`, the name resolves **confidently
127
+ and wrongly**.
128
+
129
+ Ambiguity fails safe. A too-narrow index fails silently, in the wrong direction.
130
+
131
+ Widening is otherwise harmless: `disjoint_only` means the sidecar emits only URNs it
132
+ actually recovered, so it stays quiet wherever SQL is already qualified and never
133
+ double-books usage with the main recipe.
134
+
135
+ ## Reading the run report
136
+
137
+ ```
138
+ rewritten=41 tables (2792 refs) ambiguous=18 names (63 refs) unknown=7 names (12 refs)
139
+ queries_passed=954 queries_skipped=3118 (index=1631 unique names, 412 ambiguous names)
140
+ ```
141
+
142
+ - **distinct counters** — how much of the catalog was recovered. `rewritten` counts
143
+ distinct resolved **target tables**; `ambiguous` and `unknown` count distinct bare
144
+ **names**. Deliberately not distinct `_SESSION` URNs: those embed the reader project,
145
+ so one table read from five projects would report as five.
146
+ - **`*_refs`** — how many reads that attributed. One hot table read 2,000 times is one
147
+ rewrite and 2,000 references.
148
+ - **`queries_skipped`** — queries whose SQL was already fully qualified. The main recipe
149
+ has those; the sidecar correctly stays out of the way.
150
+ - **`queries_mixed`** — queries carrying *both* an already-resolvable reference and a
151
+ recovered one. Also left to the main recipe, for a subtler reason, below.
152
+
153
+ ### Why mixed queries are skipped
154
+
155
+ `disjoint_only` keeps the two pipelines from double-booking **usage**, which is a
156
+ timeseries aspect. It does not, by itself, protect the **query entity**.
157
+
158
+ Query entities are keyed by `get_query_fingerprint(sql, platform, fast=True)` — a pure
159
+ function of the SQL text, so both pipelines derive the *same* `urn:li:query:<fp>`. But
160
+ `querySubjects` is a versioned aspect, not timeseries. If the sidecar emitted its
161
+ narrowed subject set for a query the main recipe also sees, whichever ran last would
162
+ win, and the entity would flip between the two subject sets on every scheduled run.
163
+
164
+ Passing the union instead would fix the subjects and double-book usage for the
165
+ already-resolvable half. Neither is acceptable, so a mixed query is left entirely to the
166
+ main recipe and counted. Measured on `dv-ext-prod-eu-w1-data` over 7 days:
167
+ `queries_mixed=0` — the guard costs nothing there, because that client's SQL is
168
+ uniformly unqualified.
169
+
170
+ A per-reader-project breakdown is logged alongside it, so a project that recovers nothing
171
+ shows up at 0.0% rather than vanishing.
172
+
173
+ ### Two failures that look like success
174
+
175
+ Both were hit during development. Each leaves the run green and the output plausible:
176
+
177
+ 1. **`discovered_tables` is empty at construction.** The source passes a *live reference*
178
+ to `bq_schema_extractor.table_refs`, which only fills during the schema pass. The
179
+ index is built lazily on first use, never in `__init__`.
180
+ 2. **Refs are `projects/P/datasets/D/tables/T`, not `P.D.T`.** Naive dotted parsing
181
+ matches nothing.
182
+
183
+ Because of these the source raises a structured **warning** if the index came out empty
184
+ or nothing was rewritten. **Never trust a negative result without checking the index
185
+ size.**
186
+
187
+ ## Limits
188
+
189
+ Unique-name resolution only. Where a bare name matches several catalogued objects it is
190
+ dropped. How much that costs depends entirely on the estate:
191
+
192
+ | population | ambiguous names |
193
+ |---|---|
194
+ | views in `dv-ext-prod-eu-w1-data` | **0.0%** (162/162 unique) |
195
+ | views in `dv-prod-eu-w1-data` | **54.4%** |
196
+ | tables referenced estate-wide | 53.0% |
197
+
198
+ The external-facing views carry licence/brand suffixes and are unique by construction.
199
+ The internal estate's per-market layout duplicates names across `prod_malta_*`,
200
+ `prod_spain_*`, `prod_italy_*`…, and the cross-market `common_` layer duplicates a
201
+ further 39 names against them — so roughly half of any bare reads against it stay
202
+ unrecovered.
203
+
204
+ Resolving those needs **leaf-set matching**, not name matching: the colliding
205
+ common-vs-market views have identical names but cleanly distinct leaf sets. That is
206
+ explicitly out of scope — see §9.1 of the design doc for what it would take.
207
+
208
+ ## Measuring a scope before shipping it
209
+
210
+ ```bash
211
+ export BQ_SESSION_PATCH_BILLING_PROJECT=dv-ext-prod-eu-w1-data
212
+ python tools/measure_scope.py eu baseline --days 7
213
+ python tools/measure_scope.py eu patched --days 7
214
+ python tools/diff_runs.py runs/mcps_eu_baseline_7d.json runs/mcps_eu_patched_7d.json
215
+ ```
216
+
217
+ Scope flags: `--projects` for explicit ids, `--project-pattern` to override the region's
218
+ `project_id_pattern.allow` regexes, `--days` for the window.
219
+
220
+ **The billing project must be explicit** — `--billing`, or
221
+ `$BQ_SESSION_PATCH_BILLING_PROJECT`, or inferred when `--projects` names exactly one.
222
+ DataHub passes it straight to `bigquery.Client(...)`, and when it is `None` the client
223
+ inherits whatever `google.auth.default()` resolves: `GOOGLE_CLOUD_PROJECT` first, then the
224
+ ADC file's quota project, then the active gcloud config. A stray env var will therefore
225
+ bill every `INFORMATION_SCHEMA` job to an unrelated project and fail as
226
+ `bigquery.jobs.create` denied against a project you never named. The script refuses to
227
+ start rather than inherit one, and prints both the billing and the ambient project so a
228
+ mismatch is visible.
229
+
230
+ `measure_scope.py` writes MCPs to a file sink (nothing reaches GMS) and prints the
231
+ per-project recovery table; `diff_runs.py` reports datasets gaining usage, query entities
232
+ gained and readers newly attributed, and fails if any `schemaMetadata`,
233
+ `upstreamLineage` or `datasetProperties` aspect leaked into the output.
234
+
235
+ ## Trying it against a local DataHub
236
+
237
+ `recipes/local/` holds a two-step pair for a quickstart instance, scoped to one project:
238
+
239
+ ```bash
240
+ make venv && source .venv/bin/activate
241
+ datahub ingest -c recipes/local/01-catalog.yml # stock `bigquery` — the catalog
242
+ datahub ingest -c recipes/local/02-usage-sidecar.yml # this package — the recovered usage
243
+ ```
244
+
245
+ Step 1 runs with usage **on**, exactly as production does, and still misses the
246
+ unqualified reads — that is the premise. Step 2 shows what it missed. Measured against
247
+ `dv-ext-prod-eu-w1-data` over 7 days:
248
+
249
+ | | step 1 (catalog) | step 2 (sidecar) |
250
+ |---|---|---|
251
+ | datasets with a usage aspect | 5 | **+59** |
252
+ | query entities | 170 | **+61** |
253
+ | aspects emitted | schema, lineage, properties, usage, queries | usage and queries only |
254
+
255
+ Both use ADC, so the VPN must be up — VPC-SC blocks ADC while the `bq` CLI keeps
256
+ working, which makes the failure look like a permissions problem.
257
+
258
+ They run **stateless** (`stateful_ingestion.enabled: false`), so re-running re-emits the
259
+ same days. Timeseries aspects append rather than upsert, so repeated local runs
260
+ accumulate duplicate usage documents for a day. Harmless locally; in a real deployment
261
+ keep the stateful time window, which is what makes each day emit exactly once.
262
+
263
+ ## Developing
264
+
265
+ ```bash
266
+ make venv # .venv on python 3.11, package installed editable + test extras
267
+ make test # 49 tests against a real acryl-datahub
268
+ make gate # the acceptance gate: must FAIL on stock, PASS patched
269
+ make release # test, gate, bump, rebuild, verify -> dist/
270
+ make publish
271
+ ```
272
+
273
+ `make venv` exists for interactive work and for the `tools/` scripts, which need real
274
+ BigQuery credentials and so cannot run in a throwaway environment. `make test` and
275
+ `make gate` deliberately **do not** use it — they build their own environment per run, so
276
+ a stale or hand-modified `.venv` can never make them pass.
277
+
278
+ `make gate` is the one that matters on an acryl-datahub bump. This package patches
279
+ private internals, so `make test` alone would stay green even if upstream fixed the
280
+ defect out from under it — at which point this package is dead weight and should be
281
+ retired, not shipped. `gate` fails loudly in that case.
282
+
283
+ ## Upstream
284
+
285
+ The defect is unfixed on `datahub-project/datahub` master as of 2026-09-14. The ask
286
+ there is modest: report the `_SESSION` discard, or offer opt-in resolution against
287
+ discovered tables. See `UPSTREAM-ISSUE.md`.
@@ -0,0 +1,9 @@
1
+ datahub_bq_connector_session_patch/__init__.py,sha256=U_kaflaxg_0rAwthDzHJDyrNL2W867klJyEUHtPhEzY,1247
2
+ datahub_bq_connector_session_patch/patch.py,sha256=GwkNiMIbACZM8HJrJHVWRCqKcu7uTRhyeDkDhCNXtqQ,8128
3
+ datahub_bq_connector_session_patch/resolver.py,sha256=szUyW622C4NGZhi-nIEcRwhxsfpGfxw9o5aG21mk8Og,13419
4
+ datahub_bq_connector_session_patch/source.py,sha256=16lsU8UDdOoC9qzbv69xP1RG4ZsxF0RDlOOTVXI822Q,4168
5
+ datahub_bq_connector_session_patch-0.1.0.dist-info/METADATA,sha256=6dNNhK5hKxpN9trR0RX5y9-jfa8PPDlk3isoTqtUMhQ,13628
6
+ datahub_bq_connector_session_patch-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
7
+ datahub_bq_connector_session_patch-0.1.0.dist-info/entry_points.txt,sha256=HGFQhMcd2ToOmoOBYO4Q3TEu_KsFvDxG98J8M4sDNQc,141
8
+ datahub_bq_connector_session_patch-0.1.0.dist-info/licenses/LICENSE,sha256=z8d0m5b2O9McPEK1xHG_dWgUBT6EfBDz6wA0F7xSPTA,11358
9
+ datahub_bq_connector_session_patch-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [datahub.ingestion.source.plugins]
2
+ datahub_bq_connector_session_patch = datahub_bq_connector_session_patch.source:BigQuerySessionPatchSource
@@ -0,0 +1,202 @@
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright [yyyy] [name of copyright owner]
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.