lakeview-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
lakeview/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """Local data inspection without a warehouse."""
2
+
3
+ __version__ = "0.1.0"
lakeview/__main__.py ADDED
@@ -0,0 +1,4 @@
1
+ from lakeview.cli import app
2
+
3
+ if __name__ == "__main__":
4
+ app()
lakeview/cli.py ADDED
@@ -0,0 +1,174 @@
1
+ """Terminal entry points and stable automation exit codes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import sys
7
+ from collections.abc import Callable, Iterator
8
+ from contextlib import contextmanager
9
+ from enum import StrEnum
10
+ from io import TextIOWrapper
11
+ from typing import TYPE_CHECKING, Annotated, Any
12
+
13
+ import typer
14
+ from rich.console import Console
15
+ from rich.table import Table
16
+ from rich.text import Text
17
+
18
+ from lakeview import __version__
19
+ from lakeview.formatters.markdown import clean, markdown
20
+ from lakeview.formatters.terminal import render
21
+
22
+ if TYPE_CHECKING:
23
+ from lakeview.core.engine import Engine
24
+
25
+ # Windows redirected streams may default to cp1252; reports are UTF-8.
26
+ if sys.platform == "win32":
27
+ for stream in (sys.stdout, sys.stderr):
28
+ if isinstance(stream, TextIOWrapper):
29
+ stream.reconfigure(encoding="utf-8", errors="replace")
30
+
31
+ app = typer.Typer(
32
+ no_args_is_help=True,
33
+ pretty_exceptions_enable=False,
34
+ help="Local data profiling, exact diffs, and streaming SQL.",
35
+ )
36
+ console = Console(highlight=False)
37
+ errors = Console(stderr=True, highlight=False)
38
+
39
+
40
+ class Format(StrEnum):
41
+ terminal = "terminal"
42
+ json = "json"
43
+ markdown = "markdown"
44
+
45
+
46
+ class QueryFormat(StrEnum):
47
+ terminal = "terminal"
48
+ jsonl = "jsonl"
49
+
50
+
51
+ @contextmanager
52
+ def guarded() -> Iterator[None]:
53
+ import duckdb
54
+ import pyarrow as pa
55
+
56
+ from lakeview.core.engine import LakeviewError
57
+
58
+ try:
59
+ yield
60
+ except (LakeviewError, duckdb.Error, pa.ArrowException, OSError, ValueError) as exc:
61
+ errors.print(Text(f"Error: {exc}", style="red"))
62
+ raise typer.Exit(2) from None
63
+
64
+
65
+ def output(
66
+ run: Callable[[Engine], dict[str, Any]], fmt: Format, memory_limit: str, threads: int
67
+ ) -> dict[str, Any]:
68
+ from lakeview.core.engine import Engine
69
+
70
+ with guarded(), Engine(memory_limit, threads) as engine:
71
+ with errors.status("Scanning local data…") if errors.is_terminal else _quiet():
72
+ report = clean(run(engine))
73
+ if fmt == Format.json:
74
+ typer.echo(json.dumps(report, ensure_ascii=True, allow_nan=False, default=str))
75
+ elif fmt == Format.markdown:
76
+ typer.echo(markdown(report), nl=False)
77
+ else:
78
+ render(report, console)
79
+ return dict(report)
80
+
81
+
82
+ @contextmanager
83
+ def _quiet() -> Iterator[None]:
84
+ yield
85
+
86
+
87
+ def version(value: bool) -> None:
88
+ if value:
89
+ typer.echo(__version__)
90
+ raise typer.Exit()
91
+
92
+
93
+ @app.callback()
94
+ def main(
95
+ version_flag: Annotated[
96
+ bool, typer.Option("--version", callback=version, is_eager=True)
97
+ ] = False,
98
+ ) -> None:
99
+ """Inspect local data with DuckDB. No service, account, or warehouse."""
100
+
101
+
102
+ @app.command()
103
+ def profile(
104
+ file: str, format: Format = Format.terminal, memory_limit: str = "512MB", threads: int = 4
105
+ ) -> None:
106
+ """Profile a file (or database.duckdb::schema.table)."""
107
+ from lakeview.core.profiler import profile as summarize
108
+
109
+ output(lambda e: summarize(e, file), format, memory_limit, threads)
110
+
111
+
112
+ @app.command()
113
+ def diff(
114
+ source: str,
115
+ target: str,
116
+ on: Annotated[
117
+ str | None, typer.Option(help="Comma-separated unique keys; NULL matches NULL.")
118
+ ] = None,
119
+ format: Format = Format.terminal,
120
+ fail_on_change: bool = False,
121
+ memory_limit: str = "512MB",
122
+ threads: int = 4,
123
+ ) -> None:
124
+ """Compare source → target; --fail-on-change returns 1 for differences, 2 for errors."""
125
+ from lakeview.core.differ import diff as compare
126
+
127
+ keys = [k.strip() for k in on.split(",")] if on is not None else None
128
+ report = output(lambda e: compare(e, source, target, keys), format, memory_limit, threads)
129
+ if fail_on_change and report["different"]:
130
+ raise typer.Exit(1)
131
+
132
+
133
+ @app.command()
134
+ def query(
135
+ sql: str,
136
+ files: Annotated[list[str] | None, typer.Argument()] = None,
137
+ format: QueryFormat = QueryFormat.terminal,
138
+ limit: Annotated[
139
+ int, typer.Option(min=1, help="Terminal display limit; JSONL streams all rows.")
140
+ ] = 100,
141
+ memory_limit: str = "512MB",
142
+ threads: int = 4,
143
+ ) -> None:
144
+ """Run one SELECT. File aliases: f1, f2, …; data aliases f1. JSONL streams all rows."""
145
+ from lakeview.core.engine import Engine
146
+
147
+ with guarded(), Engine(memory_limit, threads) as engine:
148
+ for i, spec in enumerate(files or [], 1):
149
+ engine.load(spec, f"f{i}")
150
+ if files:
151
+ engine.con.execute("CREATE VIEW data AS SELECT * FROM f1")
152
+ if format == QueryFormat.jsonl:
153
+ for batch in engine.query(sql):
154
+ for row in batch:
155
+ typer.echo(
156
+ json.dumps(clean(row), allow_nan=False, ensure_ascii=True, default=str)
157
+ )
158
+ else:
159
+ rows: list[dict[str, Any]] = []
160
+ for batch in engine.query(sql, min(limit + 1, 8192)):
161
+ rows.extend(batch[: limit + 1 - len(rows)])
162
+ if len(rows) > limit:
163
+ break
164
+ table = Table(title="lakeview · query")
165
+ if rows:
166
+ for name in rows[0]:
167
+ table.add_column(Text(name))
168
+ for row in rows[:limit]:
169
+ table.add_row(*(Text(str(value)) for value in row.values()))
170
+ console.print(table)
171
+ console.print(
172
+ f"{min(len(rows), limit)} rows shown"
173
+ + (" (truncated; use --format jsonl for all rows)" if len(rows) > limit else "")
174
+ )
@@ -0,0 +1 @@
1
+ """Streaming execution, profiling, and exact comparison."""
@@ -0,0 +1,136 @@
1
+ """Exact bag comparisons and null-safe, unique-key row comparisons."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import time
6
+ from typing import Any
7
+
8
+ from lakeview.core.engine import Engine, LakeviewError, ident, numeric
9
+
10
+
11
+ def metrics(engine: Engine, table: str, columns: list[str]) -> dict[str, dict[str, Any]]:
12
+ if not columns:
13
+ return {}
14
+ expressions: list[str] = []
15
+ scales = engine.scales(table, columns)
16
+ for name in columns:
17
+ value = f"CASE WHEN isfinite({ident(name)}::DOUBLE) THEN {ident(name)}::DOUBLE END"
18
+ scale = scales[name]
19
+ expressions.extend(
20
+ [
21
+ f"avg(({value})/{scale!r})*{scale!r}",
22
+ f"var_pop(({value})/{scale!r})*{scale!r}*{scale!r}",
23
+ f"min({value})",
24
+ f"max({value})",
25
+ ]
26
+ )
27
+ row = engine.con.execute(f"SELECT {', '.join(expressions)} FROM {ident(table)}").fetchone()
28
+ assert row is not None
29
+ return {
30
+ name: dict(zip(("mean", "variance", "min", "max"), row[i * 4 : i * 4 + 4], strict=True))
31
+ for i, name in enumerate(columns)
32
+ }
33
+
34
+
35
+ def diff(engine: Engine, source: str, target: str, keys: list[str] | None = None) -> dict[str, Any]:
36
+ started = time.perf_counter()
37
+ engine.load(source, "src")
38
+ engine.load(target, "dst")
39
+ left, right = engine.schema("src"), engine.schema("dst")
40
+ common = [name for name in left if name in right]
41
+ added_columns = {name: right[name] for name in right if name not in left}
42
+ removed_columns = {name: left[name] for name in left if name not in right}
43
+ changed_types = {
44
+ name: {"source": left[name], "target": right[name]}
45
+ for name in common
46
+ if left[name] != right[name]
47
+ }
48
+ counts = {
49
+ table: int(engine.scalar(f"SELECT count(*) FROM {table}")) for table in ("src", "dst")
50
+ }
51
+ comparable = [name for name in common if left[name] == right[name]]
52
+ unchanged = 0
53
+ modified: int | None = None
54
+ if keys:
55
+ if len(keys) != len(set(keys)):
56
+ raise LakeviewError("Join keys must not be repeated.")
57
+ for key in keys:
58
+ if key not in comparable:
59
+ raise LakeviewError(f"Key {key!r} must exist on both sides with identical types.")
60
+ key_sql = ", ".join(ident(k) for k in keys)
61
+ for table in ("src", "dst"):
62
+ if engine.scalar(
63
+ f"SELECT EXISTS (SELECT 1 FROM {table} GROUP BY {key_sql} HAVING count(*) > 1)"
64
+ ):
65
+ raise LakeviewError(
66
+ f"Duplicate keys in {table}; use unique keys or omit --on for bag comparison."
67
+ )
68
+ join = " AND ".join(f"s.{ident(k)} IS NOT DISTINCT FROM t.{ident(k)}" for k in keys)
69
+ shared_values = [name for name in comparable if name not in keys]
70
+ changes = (
71
+ " OR ".join(f"s.{ident(c)} IS DISTINCT FROM t.{ident(c)}" for c in shared_values)
72
+ or "false"
73
+ )
74
+ matched, modified_raw = engine.con.execute(
75
+ f"SELECT count(*), count(*) FILTER (WHERE {changes}) FROM src s JOIN dst t ON {join}"
76
+ ).fetchone() or (0, 0)
77
+ modified = int(modified_raw)
78
+ unchanged = int(matched) - modified
79
+ added, removed = counts["dst"] - int(matched), counts["src"] - int(matched)
80
+ else:
81
+ if set(left) != set(right) or changed_types:
82
+ added, removed = counts["dst"], counts["src"]
83
+ else:
84
+ cols = ", ".join(ident(c) for c in left)
85
+ removed = int(
86
+ engine.scalar(
87
+ f"SELECT count(*) FROM (SELECT {cols} FROM src "
88
+ f"EXCEPT ALL SELECT {cols} FROM dst)"
89
+ )
90
+ )
91
+ added = int(
92
+ engine.scalar(
93
+ f"SELECT count(*) FROM (SELECT {cols} FROM dst "
94
+ f"EXCEPT ALL SELECT {cols} FROM src)"
95
+ )
96
+ )
97
+ unchanged = counts["src"] - removed
98
+ metric_names = [name for name in common if numeric(left[name]) and numeric(right[name])]
99
+ source_metrics, target_metrics = (
100
+ metrics(engine, "src", metric_names),
101
+ metrics(engine, "dst", metric_names),
102
+ )
103
+ drift = {
104
+ name: {
105
+ "source": source_metrics[name],
106
+ "target": target_metrics[name],
107
+ "delta": {
108
+ stat: (
109
+ target_metrics[name][stat] - source_metrics[name][stat]
110
+ if source_metrics[name][stat] is not None
111
+ and target_metrics[name][stat] is not None
112
+ else None
113
+ )
114
+ for stat in source_metrics[name]
115
+ },
116
+ }
117
+ for name in metric_names
118
+ }
119
+ schema_changed = bool(added_columns or removed_columns or changed_types)
120
+ return {
121
+ "source": source,
122
+ "target": target,
123
+ "mode": "keyed" if keys else "multiset",
124
+ "keys": keys or [],
125
+ "source_rows": counts["src"],
126
+ "target_rows": counts["dst"],
127
+ "added": added,
128
+ "removed": removed,
129
+ "modified": modified,
130
+ "unchanged": unchanged,
131
+ "schema": {"added": added_columns, "removed": removed_columns, "changed": changed_types},
132
+ "compared_columns": comparable if keys else list(left),
133
+ "metrics": drift,
134
+ "different": schema_changed or bool(added or removed or modified),
135
+ "elapsed_ms": round((time.perf_counter() - started) * 1000, 3),
136
+ }
@@ -0,0 +1,165 @@
1
+ """Bounded DuckDB execution over local, read-only input datasets."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Iterator
7
+ from contextlib import AbstractContextManager
8
+ from pathlib import Path
9
+ from tempfile import TemporaryDirectory
10
+ from types import TracebackType
11
+ from typing import Any
12
+
13
+ import duckdb
14
+ import pyarrow as pa
15
+ import pyarrow.dataset as ds
16
+ import pyarrow.ipc as ipc
17
+
18
+
19
+ class LakeviewError(Exception):
20
+ """An actionable user-facing error."""
21
+
22
+
23
+ def ident(value: str) -> str:
24
+ return '"' + value.replace('"', '""') + '"'
25
+
26
+
27
+ def literal(value: str) -> str:
28
+ return "'" + value.replace("'", "''") + "'"
29
+
30
+
31
+ def numeric(dtype: str) -> bool:
32
+ return bool(
33
+ re.match(r"^(U?(TINYINT|SMALLINT|INTEGER|BIGINT|HUGEINT)|FLOAT|DOUBLE|DECIMAL)", dtype)
34
+ )
35
+
36
+
37
+ class Engine(AbstractContextManager["Engine"]):
38
+ """Ephemeral database with disk spill; memory_limit is not a total RSS cap."""
39
+
40
+ def __init__(self, memory_limit: str = "512MB", threads: int = 4) -> None:
41
+ if not re.fullmatch(r"[1-9][0-9]*(?:\.[0-9]+)?(?:KB|MB|GB|TB)", memory_limit.upper()):
42
+ raise LakeviewError("Memory limit must look like 512MB or 2GB.")
43
+ if threads < 1:
44
+ raise LakeviewError("Threads must be positive.")
45
+ self.temp = TemporaryDirectory(prefix="lakeview-")
46
+ self.resources: list[Any] = []
47
+ self.databases: dict[Path, str] = {}
48
+ try:
49
+ self.con = duckdb.connect(
50
+ ":memory:",
51
+ config={
52
+ "memory_limit": memory_limit,
53
+ "threads": threads,
54
+ "temp_directory": self.temp.name,
55
+ "preserve_insertion_order": False,
56
+ "autoinstall_known_extensions": False,
57
+ "autoload_known_extensions": False,
58
+ },
59
+ )
60
+ except Exception:
61
+ self.temp.cleanup()
62
+ raise
63
+
64
+ def __exit__(
65
+ self,
66
+ exc_type: type[BaseException] | None,
67
+ exc: BaseException | None,
68
+ tb: TracebackType | None,
69
+ ) -> None:
70
+ self.con.close()
71
+ for resource in reversed(self.resources):
72
+ resource.close()
73
+ self.temp.cleanup()
74
+
75
+ def scalar(self, sql: str) -> Any:
76
+ row = self.con.execute(sql).fetchone()
77
+ if row is None:
78
+ raise LakeviewError("Query did not return a result.")
79
+ return row[0]
80
+
81
+ def schema(self, alias: str) -> dict[str, str]:
82
+ return {
83
+ str(row[0]): str(row[1])
84
+ for row in self.con.execute(f"DESCRIBE {ident(alias)}").fetchall()
85
+ }
86
+
87
+ def scales(self, alias: str, columns: list[str]) -> dict[str, float]:
88
+ """Finite magnitudes for stable aggregation without intermediate overflow."""
89
+ if not columns:
90
+ return {}
91
+ expressions = [
92
+ f"max(abs({ident(c)}::DOUBLE)) FILTER (WHERE isfinite({ident(c)}::DOUBLE))"
93
+ for c in columns
94
+ ]
95
+ row = self.con.execute(f"SELECT {', '.join(expressions)} FROM {ident(alias)}").fetchone()
96
+ assert row is not None
97
+ return {c: float(v) if v else 1.0 for c, v in zip(columns, row, strict=True)}
98
+
99
+ def load(self, spec: str, alias: str) -> Path:
100
+ raw, sep, table = spec.partition("::")
101
+ path = Path(raw).expanduser().resolve()
102
+ if not path.is_file():
103
+ raise LakeviewError(f"File not found: {path}")
104
+ suffix = path.suffix.lower()
105
+ quoted = literal(str(path))
106
+ if suffix in {".duckdb", ".db"}:
107
+ database = self.databases.get(path, f"db_{alias}")
108
+ if path not in self.databases:
109
+ self.con.execute(f"ATTACH {quoted} AS {ident(database)} (READ_ONLY)")
110
+ self.databases[path] = database
111
+ tables = self.con.execute(
112
+ "SELECT table_schema, table_name FROM information_schema.tables "
113
+ "WHERE table_catalog = ? AND table_type = 'BASE TABLE' ORDER BY 1, 2",
114
+ [database],
115
+ ).fetchall()
116
+ if sep:
117
+ matches = [(s, t) for s, t in tables if table in (t, f"{s}.{t}")]
118
+ else:
119
+ matches = tables
120
+ if len(matches) != 1:
121
+ available = ", ".join(f"{s}.{t}" for s, t in tables)
122
+ raise LakeviewError(
123
+ f"Choose one table with file.duckdb::schema.table. Available: {available}"
124
+ )
125
+ schema, name = matches[0]
126
+ source = f"{ident(database)}.{ident(schema)}.{ident(name)}"
127
+ elif sep:
128
+ raise LakeviewError("The ::table selector is only supported for DuckDB files.")
129
+ elif suffix in {".parquet", ".pq"}:
130
+ source = f"read_parquet({quoted})"
131
+ elif suffix in {".csv", ".tsv"}:
132
+ delimiter = "\t" if suffix == ".tsv" else ","
133
+ source = f"read_csv({quoted}, header=true, delim={literal(delimiter)}, sample_size=-1)"
134
+ elif suffix in {".arrow", ".ipc", ".feather"}:
135
+ handle = pa.memory_map(str(path), "r")
136
+ self.resources.append(handle)
137
+ try:
138
+ ipc.open_file(handle)
139
+ except pa.ArrowInvalid:
140
+ import pyarrow.parquet as pq
141
+
142
+ handle.seek(0)
143
+ reader = ipc.open_stream(handle)
144
+ converted = Path(self.temp.name) / f"{alias}.parquet"
145
+ with pq.ParquetWriter(str(converted), reader.schema) as writer:
146
+ for batch in reader:
147
+ writer.write_batch(batch)
148
+ source = f"read_parquet({literal(str(converted))})"
149
+ else:
150
+ self.con.register(f"arrow_{alias}", ds.dataset(str(path), format="ipc"))
151
+ source = ident(f"arrow_{alias}")
152
+ else:
153
+ raise LakeviewError(
154
+ f"Unsupported file extension: {suffix}. Use Parquet, CSV, TSV, DuckDB, or IPC."
155
+ )
156
+ self.con.execute(f"CREATE VIEW {ident(alias)} AS SELECT * FROM {source}")
157
+ return path
158
+
159
+ def query(self, sql: str, batch_size: int = 8192) -> Iterator[list[dict[str, Any]]]:
160
+ statements = self.con.extract_statements(sql)
161
+ if len(statements) != 1 or statements[0].type != duckdb.StatementType.SELECT:
162
+ raise LakeviewError("query accepts exactly one SELECT (including WITH).")
163
+ reader = self.con.execute(sql).to_arrow_reader(batch_size)
164
+ for batch in reader:
165
+ yield batch.to_pylist()
@@ -0,0 +1,86 @@
1
+ """Exact counts and finite-value statistics, with approximate quantiles."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import time
6
+ from typing import Any
7
+
8
+ import psutil
9
+
10
+ from lakeview.core.engine import Engine, ident, numeric
11
+
12
+
13
+ def profile(engine: Engine, spec: str) -> dict[str, Any]:
14
+ started = time.perf_counter()
15
+ path = engine.load(spec, "data")
16
+ schema = engine.schema("data")
17
+ scales = engine.scales("data", [c for c, dtype in schema.items() if numeric(dtype)])
18
+ expressions = ["count(*)"]
19
+ for name, dtype in schema.items():
20
+ col = ident(name)
21
+ expressions.append(f"count(*) - count({col})")
22
+ if numeric(dtype):
23
+ value = f"CASE WHEN isfinite({col}::DOUBLE) THEN {col}::DOUBLE END"
24
+ scale = scales[name]
25
+ expressions.extend(
26
+ [
27
+ f"count({col}) - count({value})",
28
+ f"avg(({value})/{scale!r})*{scale!r}",
29
+ f"var_pop(({value})/{scale!r})*{scale!r}*{scale!r}",
30
+ f"min({value})",
31
+ f"max({value})",
32
+ f"approx_quantile({value}, [0.25, 0.5, 0.75])",
33
+ ]
34
+ )
35
+ row = engine.con.execute("SELECT " + ", ".join(expressions) + " FROM data").fetchone()
36
+ assert row is not None
37
+ count, cursor = int(row[0]), 1
38
+ columns: list[dict[str, Any]] = []
39
+ histograms: list[str] = []
40
+ histogram_columns: list[dict[str, Any]] = []
41
+ for name, dtype in schema.items():
42
+ nulls = int(row[cursor])
43
+ cursor += 1
44
+ column: dict[str, Any] = {
45
+ "name": name,
46
+ "type": dtype,
47
+ "nulls": nulls,
48
+ "null_percent": 100 * nulls / count if count else 0.0,
49
+ }
50
+ if numeric(dtype):
51
+ for field in ("nonfinite", "mean", "variance", "min", "max", "quantiles"):
52
+ column[field] = row[cursor]
53
+ cursor += 1
54
+ low, high = column["min"], column["max"]
55
+ column["histogram"] = [0] * 8
56
+ if low is not None:
57
+ col = f"{ident(name)}::DOUBLE"
58
+ if high == low:
59
+ bucket = "0"
60
+ else:
61
+ scale = max(abs(low), abs(high), 1.0)
62
+ bucket = (
63
+ f"least(7, greatest(0, floor((({col}/{scale!r}) - "
64
+ f"({low!r}/{scale!r})) / "
65
+ f"(({high!r}/{scale!r}) - ({low!r}/{scale!r})) * 8)))"
66
+ )
67
+ for i in range(8):
68
+ histograms.append(
69
+ f"count(*) FILTER (WHERE isfinite({col}) AND ({bucket}) = {i})"
70
+ )
71
+ histogram_columns.append(column)
72
+ columns.append(column)
73
+ if histograms:
74
+ bins = engine.con.execute("SELECT " + ", ".join(histograms) + " FROM data").fetchone()
75
+ assert bins is not None
76
+ for index, column in enumerate(histogram_columns):
77
+ column["histogram"] = list(bins[index * 8 : index * 8 + 8])
78
+ return {
79
+ "file": str(path),
80
+ "rows": count,
81
+ "file_bytes": path.stat().st_size,
82
+ "process_rss_bytes": psutil.Process().memory_info().rss,
83
+ "columns": columns,
84
+ "quantiles": "approximate",
85
+ "elapsed_ms": round((time.perf_counter() - started) * 1000, 3),
86
+ }
@@ -0,0 +1 @@
1
+ """Human and machine readable reports."""
@@ -0,0 +1,103 @@
1
+ """Portable reports with escaped data-controlled Markdown cells."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import html
6
+ import math
7
+ from typing import Any
8
+
9
+
10
+ def clean(value: Any) -> Any:
11
+ """Convert non-finite results to JSON null, recursively."""
12
+ if isinstance(value, float) and not math.isfinite(value):
13
+ return None
14
+ if isinstance(value, dict):
15
+ return {str(k): clean(v) for k, v in value.items()}
16
+ if isinstance(value, (list, tuple)):
17
+ return [clean(v) for v in value]
18
+ return value
19
+
20
+
21
+ def cell(value: Any) -> str:
22
+ text = "—" if value is None else str(value)
23
+ return (
24
+ html.escape(text)
25
+ .replace("|", "&#124;")
26
+ .replace("\r", " ")
27
+ .replace("\n", " ")
28
+ .replace("`", "&#96;")
29
+ )
30
+
31
+
32
+ def markdown(report: dict[str, Any]) -> str:
33
+ if "schema" in report:
34
+ lines = [
35
+ "## lakeview data diff",
36
+ "",
37
+ "| Added | Removed | Modified | Unchanged |",
38
+ "| ---: | ---: | ---: | ---: |",
39
+ "| "
40
+ + " | ".join(cell(report[k]) for k in ("added", "removed", "modified", "unchanged"))
41
+ + " |",
42
+ "",
43
+ f"Mode: {report['mode']}. Modified is unavailable in multiset mode.",
44
+ "",
45
+ "### Schema drift",
46
+ "",
47
+ "| Change | Column | Type |",
48
+ "| --- | --- | --- |",
49
+ ]
50
+ for kind, columns in report["schema"].items():
51
+ for name, dtype in columns.items():
52
+ lines.append(f"| {cell(kind)} | {cell(name)} | {cell(dtype)} |")
53
+ lines.extend(
54
+ [
55
+ "",
56
+ "### Numerical drift (finite values)",
57
+ "",
58
+ "| Column | Statistic | Source | Target | Delta |",
59
+ "| --- | --- | ---: | ---: | ---: |",
60
+ ]
61
+ )
62
+ for name, values in report["metrics"].items():
63
+ for stat in ("mean", "variance", "min", "max"):
64
+ lines.append(
65
+ "| "
66
+ + " | ".join(
67
+ cell(v)
68
+ for v in (
69
+ name,
70
+ stat,
71
+ values["source"][stat],
72
+ values["target"][stat],
73
+ values["delta"][stat],
74
+ )
75
+ )
76
+ + " |"
77
+ )
78
+ lines.extend(
79
+ [
80
+ "",
81
+ "Row comparisons use shared columns with identical types; "
82
+ "schema changes are separate.",
83
+ ]
84
+ )
85
+ else:
86
+ lines = [
87
+ "## lakeview profile",
88
+ "",
89
+ f"Rows: {report['rows']:,}. File bytes: {report['file_bytes']:,}.",
90
+ "",
91
+ "| Column | Type | Null % | Mean | Min | Max | Approx. Q25/Q50/Q75 |",
92
+ "| --- | --- | ---: | ---: | ---: | ---: | --- |",
93
+ ]
94
+ for col in report["columns"]:
95
+ lines.append(
96
+ "| "
97
+ + " | ".join(
98
+ cell(col.get(k))
99
+ for k in ("name", "type", "null_percent", "mean", "min", "max", "quantiles")
100
+ )
101
+ + " |"
102
+ )
103
+ return "\n".join(lines) + "\n"
@@ -0,0 +1,85 @@
1
+ """Compact Rich reports; data values are never interpreted as markup."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from rich.console import Console
8
+ from rich.table import Table
9
+ from rich.text import Text
10
+
11
+
12
+ def number(value: Any) -> str:
13
+ return "—" if value is None else f"{value:.5g}" if isinstance(value, float) else str(value)
14
+
15
+
16
+ def sparkline(bins: list[int]) -> str:
17
+ maximum = max(bins, default=0)
18
+ return "".join("▁▂▃▄▅▆▇█"[round(v / maximum * 7)] for v in bins) if maximum else "────────"
19
+
20
+
21
+ def render(report: dict[str, Any], console: Console) -> None:
22
+ if "schema" in report:
23
+ table = Table(title="lakeview · semantic diff")
24
+ for name, color in (
25
+ ("Added", "green"),
26
+ ("Removed", "red"),
27
+ ("Modified", "yellow"),
28
+ ("Unchanged", "cyan"),
29
+ ):
30
+ table.add_column(name, style=color, justify="right")
31
+ table.add_row(*(number(report[k]) for k in ("added", "removed", "modified", "unchanged")))
32
+ console.print(table)
33
+ schema = Table("Change", "Column", "Type", title="Schema drift")
34
+ for kind, columns in report["schema"].items():
35
+ for name, dtype in columns.items():
36
+ schema.add_row(Text(kind), Text(name), Text(str(dtype)))
37
+ if schema.row_count:
38
+ console.print(schema)
39
+ drift = Table(
40
+ "Column",
41
+ "Statistic",
42
+ "Source",
43
+ "Target",
44
+ "Delta",
45
+ title="Numerical drift · finite values",
46
+ )
47
+ for name, values in report["metrics"].items():
48
+ for stat in ("mean", "variance", "min", "max"):
49
+ drift.add_row(
50
+ Text(name),
51
+ stat,
52
+ *(number(values[side][stat]) for side in ("source", "target", "delta")),
53
+ )
54
+ if drift.row_count:
55
+ console.print(drift)
56
+ console.print(
57
+ "Mode: " + report["mode"] + ". Schema changes reported separately; "
58
+ "values compared on shared identical-type columns.",
59
+ markup=False,
60
+ )
61
+ else:
62
+ console.print(
63
+ f"lakeview · {report['rows']:,} rows · {report['file_bytes']:,} file bytes",
64
+ markup=False,
65
+ )
66
+ console.print(f"Process RSS after scan: {report['process_rss_bytes'] / 1024**2:.1f} MiB")
67
+ table = Table(
68
+ "Column", "Type", "Null %", "Mean", "Min / Max", "Q25 / Q50 / Q75 ≈", "Distribution"
69
+ )
70
+ for col in report["columns"]:
71
+ table.add_row(
72
+ Text(col["name"]),
73
+ col["type"],
74
+ f"{col['null_percent']:.2f}",
75
+ number(col.get("mean")),
76
+ f"{number(col.get('min'))} / {number(col.get('max'))}",
77
+ " / ".join(number(v) for v in (col.get("quantiles") or [])),
78
+ sparkline(col.get("histogram", [])),
79
+ )
80
+ console.print(table)
81
+ console.print(
82
+ "Counts exact · quantiles approximate · statistics exclude NaN/Infinity · "
83
+ "file bytes are not RAM usage"
84
+ )
85
+ console.print(f"Completed in {report['elapsed_ms']:,.1f} ms", style="dim")
lakeview/py.typed ADDED
File without changes
@@ -0,0 +1,230 @@
1
+ Metadata-Version: 2.5
2
+ Name: lakeview-cli
3
+ Version: 0.1.0
4
+ Summary: Local data profiling, exact semantic diffs, and streaming SQL with DuckDB
5
+ Project-URL: Homepage, https://github.com/Karthikvk1899/lakeview
6
+ Project-URL: Issues, https://github.com/Karthikvk1899/lakeview/issues
7
+ Author: Karthik Beesa
8
+ License-Expression: Apache-2.0
9
+ License-File: LICENSE
10
+ Keywords: cli,data-engineering,diff,duckdb,parquet
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Typing :: Typed
14
+ Requires-Python: >=3.11
15
+ Requires-Dist: duckdb<2,>=1.4
16
+ Requires-Dist: psutil<8,>=6
17
+ Requires-Dist: pyarrow<24,>=19
18
+ Requires-Dist: rich<15,>=13.9
19
+ Requires-Dist: typer<1,>=0.16
20
+ Provides-Extra: benchmark
21
+ Requires-Dist: pandas<4,>=2.2; extra == 'benchmark'
22
+ Description-Content-Type: text/markdown
23
+
24
+ # lakeview — The missing CLI for local data profiling, semantic diffing, and fast SQL.
25
+
26
+ [![CI](https://github.com/Karthikvk1899/lakeview/actions/workflows/ci.yml/badge.svg)](https://github.com/Karthikvk1899/lakeview/actions/workflows/ci.yml)
27
+ [![Python 3.11+](https://img.shields.io/badge/python-3.11%2B-blue)](pyproject.toml)
28
+ [![Apache 2.0](https://img.shields.io/badge/license-Apache--2.0-green)](LICENSE)
29
+
30
+ Your dbt model ran. The row count looks fine. Something still changed.
31
+
32
+ You can open a notebook, load two DataFrames, remember how to compare NULLs,
33
+ then discover that duplicate rows broke your join. Or start a warehouse just
34
+ to inspect two files already on your laptop.
35
+
36
+ For the small-file, two-second check you wanted in the first place:
37
+
38
+ ```sh
39
+ lakeview diff prod.parquet staging.parquet --on id
40
+ ```
41
+
42
+ **Exact row comparisons. Duplicate-aware counts. Read-only inputs. No account.**
43
+
44
+ ![Actual Rich terminal output from the synthetic demo](docs/demo.svg)
45
+
46
+ ```diff
47
+ $ lakeview diff prod.parquet staging.parquet --on id
48
+ + added rows: 1
49
+ - removed rows: 1
50
+ ! modified rows: 1
51
+ unchanged rows: 1
52
+ + schema: region VARCHAR
53
+ ! revenue mean: 200.00 -> 316.67
54
+ ```
55
+
56
+ The condensed example above comes from the three-row [demo generator](examples/make_demo.py).
57
+ The CLI renders insertions in green, deletions in red, and modifications in yellow.
58
+
59
+ ## Install and try it
60
+
61
+ Python 3.11 or newer. Install the tagged source now:
62
+
63
+ ```sh
64
+ pip install "lakeview-cli @ git+https://github.com/Karthikvk1899/lakeview.git@v0.1.0"
65
+ lakeview --help
66
+ ```
67
+
68
+ Or install a wheel / download a platform bundle from [Releases](https://github.com/Karthikvk1899/lakeview/releases).
69
+ The repository includes automated PyPI publishing, but registry publication requires
70
+ maintainer setup. `pip install lakeview-cli` is only appropriate after this project's
71
+ package has been published there. There is no Homebrew formula yet.
72
+
73
+ ```sh
74
+ git clone https://github.com/Karthikvk1899/lakeview.git
75
+ cd lakeview
76
+ uv sync --locked
77
+ uv run python examples/make_demo.py
78
+ uv run lakeview profile demo/prod.parquet
79
+ uv run lakeview diff demo/prod.parquet demo/staging.parquet --on id
80
+ uv run lakeview query 'SELECT avg(revenue) FROM data' demo/staging.parquet
81
+ ```
82
+
83
+ ## Three commands, useful defaults
84
+
85
+ | Command | What you get |
86
+ | --- | --- |
87
+ | `profile` | Schema, exact row/null counts, null rates, mean, population variance, min/max, approximate quartiles, eight-bin histograms, file size and process RSS |
88
+ | `diff` | Schema additions/removals/type changes, exact row deltas, mean/variance/min/max drift |
89
+ | `query` | DuckDB SELECT queries, bounded terminal previews, streaming JSONL output |
90
+
91
+ ```sh
92
+ lakeview profile events.parquet --format json
93
+ lakeview profile 'warehouse.duckdb::main.orders' --memory-limit 256MB
94
+ lakeview diff yesterday.csv today.csv --on account_id,event_id --format markdown
95
+ lakeview diff baseline.arrow candidate.arrow --fail-on-change
96
+ lakeview query 'SELECT f1.id FROM f1 JOIN f2 USING (id)' left.parquet right.csv
97
+ lakeview query 'SELECT * FROM data ORDER BY id' events.parquet --format jsonl > rows.jsonl
98
+ ```
99
+
100
+ Formats: Parquet (`.parquet`, `.pq`), CSV/TSV with headers, DuckDB (`.duckdb`, `.db`),
101
+ and Arrow IPC file/stream (`.arrow`, `.ipc`) / Feather V2 (`.feather`). A database
102
+ with one table can omit the selector. With several tables, use `file.duckdb::schema.table`.
103
+ The first query input is `f1` (also `data`), followed by `f2`, `f3`, etc.
104
+
105
+ The terminal query preview defaults to 100 rows; change it with `--limit`.
106
+ JSONL streams every row in batches of 8,192. Decimal/HUGEINT and temporal values
107
+ serialize as strings to preserve precision. SQL NULL and non-finite floats become
108
+ JSON `null`. Use ORDER BY when output order matters.
109
+
110
+ ## What does “different” mean?
111
+
112
+ | Case | Behavior |
113
+ | --- | --- |
114
+ | With `--on` | Keys must exist with identical types and be unique on each side. Composite keys supported; NULL matches NULL. Duplicate keys produce an error instead of a many-to-many join. |
115
+ | Without `--on` | Exact multiset comparison via `EXCEPT ALL`: duplicate multiplicities count. A changed row is a removal plus an addition; modified is unavailable. |
116
+ | Column order | Ignored for equality. |
117
+ | Schema drift with keys | Reported separately. Modified counts compare shared, identical-type non-key columns only. A changed-type column is not silently cast. |
118
+ | Schema drift without keys | Incompatible schemas make all source rows removed and all target rows added. |
119
+ | Floating values | Exact DuckDB equality for row comparisons, including its NaN semantics. No rounding or tolerance is applied. |
120
+ | Numeric summaries | Finite values only, converted to DOUBLE; high-precision decimals/integers may round in statistics, but row comparisons retain their original types. |
121
+
122
+ Exit codes: **0** success, **1** differences when `--fail-on-change` is enabled,
123
+ **2** input, SQL, or execution error. Without `--fail-on-change`, a successful diff
124
+ returns 0 even if data changed. Schema drift counts as a difference.
125
+
126
+ ## Performance, measured
127
+
128
+ One million rows, three columns, 11.6 MB Parquet. Windows build 26200, eight logical
129
+ CPUs, Python 3.11.15, DuckDB 1.5.6, Arrow 23.0.1, Pandas 3.0.6.
130
+ Medians of three fresh processes; OS file cache may be warm. Peak RSS is sampled
131
+ every 10 ms. [Raw samples](docs/benchmarks/windows-1m.json) and [benchmark script](benchmarks/run.py).
132
+
133
+ | Workflow | Startup probe | Full operation | Peak RSS during operation |
134
+ | --- | ---: | ---: | ---: |
135
+ | lakeview | 382 ms (`--help`) | 1,314 ms (`profile --format json`) | 131.5 MiB |
136
+ | Pandas | 737 ms (`import pandas`) | 995 ms (`read_parquet`, `describe`, null counts) | 173.4 MiB |
137
+ | Spark | Not measured | Not measured | Not measured |
138
+
139
+ These operations are not feature-equivalent: Pandas `describe` also computes distinct
140
+ categorical summaries; lakeview computes histograms. Startup probes are different
141
+ operations too. This fixture favors Pandas on elapsed time. lakeview's value is a
142
+ ready-to-use CLI, explicit diff semantics, and a spill-capable engine.
143
+
144
+ **There is no universal “under 200 ms” promise.** Full statistics require scans;
145
+ storage, data shape, string width, CPU, and startup matter. Quantiles are approximate.
146
+ CSV performs full-file type inference for consistent types, adding work. IPC streams
147
+ are converted batchwise into temporary Parquet for repeatable scans.
148
+
149
+ ```sh
150
+ uv run --extra benchmark python benchmarks/run.py --rows 1000000 --repeats 3
151
+ ```
152
+
153
+ `--memory-limit` controls DuckDB's buffer manager, not total process RSS. Arrow,
154
+ Python, and some engine allocations sit outside it. Large joins and sorts can spill
155
+ to the temporary directory; adequate disk space is required. Resource exhaustion is
156
+ reported as a CLI error, not a guarantee that every workload fits any memory limit.
157
+ The profile's RSS is a process snapshot after scanning, not a peak measurement.
158
+
159
+ A separate large-file smoke test profiled **2.49 GB of uncompressed Parquet**
160
+ (1.2 million rows) in **3.10 seconds**, using `--memory-limit 128MB --threads 2`.
161
+ Process RSS after scanning was 144.5 MiB. This fixture has a constant 2,048-byte
162
+ string payload and benefits from column projection; it is not representative of
163
+ all 2.49 GB files. [Raw result](docs/benchmarks/large-parquet.json).
164
+
165
+ ## Put the diff in a pull request
166
+
167
+ The job below assumes your existing build produced `baseline.parquet` and
168
+ `candidate.parquet` in the workspace. Use `pull_request`, never run untrusted PR
169
+ code with a privileged `pull_request_target` token. Fork PRs generally cannot write
170
+ comments with their read-only token; they can still retain a job summary.
171
+
172
+ ```yaml
173
+ permissions:
174
+ contents: read
175
+ pull-requests: write
176
+ steps:
177
+ - uses: actions/checkout@v4
178
+ - uses: actions/setup-python@v5
179
+ with:
180
+ python-version: '3.11'
181
+ - run: pip install "lakeview-cli @ git+https://github.com/Karthikvk1899/lakeview.git@v0.1.0"
182
+ - name: Compute report
183
+ shell: bash
184
+ run: |
185
+ lakeview diff baseline.parquet candidate.parquet --on id --format markdown > diff.md
186
+ cat diff.md >> "$GITHUB_STEP_SUMMARY"
187
+ - uses: actions/github-script@v7
188
+ if: github.event_name == 'pull_request' && github.event.pull_request.head.repo.fork == false
189
+ with:
190
+ script: |
191
+ const fs = require('fs');
192
+ const body = fs.readFileSync('diff.md', 'utf8');
193
+ await github.rest.issues.createComment({
194
+ ...context.repo,
195
+ issue_number: context.issue.number,
196
+ body: body.slice(0, 60000)
197
+ });
198
+ - name: Fail on data changes
199
+ run: lakeview diff baseline.parquet candidate.parquet --on id --fail-on-change
200
+ ```
201
+
202
+ Reports can expose business aggregates. Review what is appropriate for your repository.
203
+ Markdown report values are escaped and passed as file contents, not interpolated
204
+ into executable JavaScript or shell commands.
205
+
206
+ ## Under the hood
207
+
208
+ Each command owns an in-memory DuckDB connection and a temporary spill directory.
209
+ Parquet/CSV scans stay in DuckDB; IPC files use Arrow Dataset scans. IPC streams are
210
+ written one record batch at a time to temporary Parquet. DuckDB databases attach
211
+ read-only. Normal exit closes resources and removes temporary files.
212
+
213
+ Profiling computes finite magnitudes, aggregate statistics, then equal-width histogram
214
+ bins in separate vectorized passes. Scaling prevents intermediate variance overflow;
215
+ statistics outside representable floating-point range render as unavailable. Diffing
216
+ uses exact comparisons, not hash equality. Python receives summaries or query batches,
217
+ never a whole input DataFrame.
218
+
219
+ See [architecture and limits](docs/architecture.md), [security](SECURITY.md), and
220
+ [contributing](CONTRIBUTING.md). This is an initial beta release. It does not include
221
+ remote storage, fuzzy matching, row samples, or automatic key discovery.
222
+
223
+ ## Help make local data work less tedious
224
+
225
+ Try it on a real model output. Open an issue with a synthetic reproduction when the
226
+ semantics surprise you. Contributions around datasets, profiling accuracy, and
227
+ reproducible performance are especially useful. If it earns a place in your workflow,
228
+ a star helps other data engineers find it.
229
+
230
+ Apache-2.0. Built with [DuckDB](https://duckdb.org/) and [Apache Arrow](https://arrow.apache.org/).
@@ -0,0 +1,16 @@
1
+ lakeview/__init__.py,sha256=V0T9QzL4VfSol_l-AMSBNMb56TAe4oWQnM0sOLB7QUg,72
2
+ lakeview/__main__.py,sha256=9O9_ORluXXp5r8BvGfm7XZ242nU8srFHf-EA92I1QDQ,67
3
+ lakeview/cli.py,sha256=TffGfy-4ED_OAiQhJfxgPb5TYjYFSWleCrPX3bR8wM8,5475
4
+ lakeview/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
5
+ lakeview/core/__init__.py,sha256=H6xilnOtqV2h0buDyZ0lz6DDOYtp7M8SwglGtvO_ed8,60
6
+ lakeview/core/differ.py,sha256=krKMJR61Fjs-CSOB2ZHPt3GIxgbPCx6g62eeL2JuaHY,5465
7
+ lakeview/core/engine.py,sha256=M0K3x-0_nkIOlWa-M2Fmo0v3RBMVQt7Hbx6uYNb-2Ng,6512
8
+ lakeview/core/profiler.py,sha256=uD6iwZZ7rQfG5NHe6lGvbTd2Okfa_uiEquQ7cqklI1A,3351
9
+ lakeview/formatters/__init__.py,sha256=Wd_lMGsCE7l0TEVLx1iLel3Z-TTkpuI3RtpP-8Y1Up8,42
10
+ lakeview/formatters/markdown.py,sha256=QH9fDTH6eqFUoq1H5VGp6fEPTL-3-Jfn-3wUbTQhMhU,3319
11
+ lakeview/formatters/terminal.py,sha256=VjDAHEoLM33Lit0lrSf38XRYJrf4skFzZCFLB9skxQk,3263
12
+ lakeview_cli-0.1.0.dist-info/METADATA,sha256=mhDJ2BzJ9wm-pvXhXSwpAaCDDih7Zh7YG2vmyAxd_q8,11001
13
+ lakeview_cli-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
14
+ lakeview_cli-0.1.0.dist-info/entry_points.txt,sha256=UWXg_MqvPOb5DZqJAeAg7Q-G6jndmTZz3Zwv8IL7Y20,46
15
+ lakeview_cli-0.1.0.dist-info/licenses/LICENSE,sha256=z8d0m5b2O9McPEK1xHG_dWgUBT6EfBDz6wA0F7xSPTA,11358
16
+ lakeview_cli-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ lakeview = lakeview.cli:app
@@ -0,0 +1,202 @@
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright [yyyy] [name of copyright owner]
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.