table-validator 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,84 @@
1
+ """Config manager: loads/saves non-secret config under ~/.table_validator/config.yaml."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Optional
7
+
8
+ import yaml
9
+
10
+ from table_validator.config.schema import ValidatorConfig
11
+
12
+ CONFIG_DIR = Path.home() / ".table_validator"
13
+ CONFIG_PATH = CONFIG_DIR / "config.yaml"
14
+
15
+
16
+ class ConfigNotFoundError(Exception):
17
+ """Raised when a config file is required but doesn't exist yet.
18
+
19
+ Distinct from a generic FileNotFoundError so callers (e.g. the
20
+ `validate` command) can catch it specifically and print a targeted
21
+ "run configure first" message instead of a raw traceback.
22
+ """
23
+
24
+
25
+ def default_config() -> ValidatorConfig:
26
+ """Return an empty/default config (used when no config file exists yet)."""
27
+ return ValidatorConfig()
28
+
29
+
30
+ def load_config(path: Optional[Path] = None) -> ValidatorConfig:
31
+ """Load config from `path` (default: CONFIG_PATH), returning a default
32
+ config if it doesn't exist.
33
+
34
+ `path` defaults to None (resolved to CONFIG_PATH at call time, not
35
+ definition time) so this always reflects the current value of
36
+ CONFIG_PATH - including in tests that patch it.
37
+
38
+ Use this when a missing config is a legitimate, silent starting point
39
+ (e.g. the wizard pre-populating its prompts with existing values).
40
+ For a context where a config file is required - like `validate` - use
41
+ require_config() instead, which raises ConfigNotFoundError.
42
+ """
43
+ path = path or CONFIG_PATH
44
+
45
+ if not path.exists():
46
+ return default_config()
47
+
48
+ raw = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
49
+ return ValidatorConfig.model_validate(raw)
50
+
51
+
52
+ def require_config(path: Optional[Path] = None) -> ValidatorConfig:
53
+ """Load config from `path` (default: CONFIG_PATH), raising
54
+ ConfigNotFoundError if it doesn't exist.
55
+
56
+ Intended for `tablevalidator validate`: running validation before
57
+ `tablevalidator configure` has ever been run is a user error that
58
+ deserves a clear, actionable message rather than silently validating
59
+ against an empty config.
60
+ """
61
+ path = path or CONFIG_PATH
62
+
63
+ if not path.exists():
64
+ raise ConfigNotFoundError(
65
+ f"No configuration found at {path}. "
66
+ "Run 'tablevalidator configure' first to set up your source/target "
67
+ "tables and Databricks connection."
68
+ )
69
+
70
+ return load_config(path)
71
+
72
+
73
+ def save_config(config: ValidatorConfig, path: Optional[Path] = None) -> None:
74
+ """Save `config` to `path` (default: CONFIG_PATH), creating the parent
75
+ directory if missing."""
76
+ path = path or CONFIG_PATH
77
+
78
+ path.parent.mkdir(parents=True, exist_ok=True)
79
+
80
+ data = config.model_dump(mode="json", by_alias=True)
81
+ path.write_text(
82
+ yaml.safe_dump(data, sort_keys=False, default_flow_style=False),
83
+ encoding="utf-8",
84
+ )
@@ -0,0 +1,179 @@
1
+ """Config schema: pydantic models describing stored (non-secret) configuration.
2
+
3
+ Secrets (Databricks PAT, Azure Storage key, Azure SQL password, etc.) are
4
+ handled separately in Phase 4 and never appear on these models.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from enum import Enum
10
+ from typing import List, Optional
11
+
12
+ from pydantic import BaseModel, Field
13
+
14
+
15
+ class ValidationType(str, Enum):
16
+ CATALOG = "catalog"
17
+ SCHEMA = "schema"
18
+ COLUMN = "column"
19
+ ROW = "row"
20
+
21
+
22
+ class SourceType(str, Enum):
23
+ """
24
+ What's being compared against the (always-Databricks) target catalog.
25
+ Defaults to DATABRICKS so existing saved configs without this field
26
+ (from before source_type existed) load unchanged - they were always
27
+ Databricks-to-Databricks comparisons.
28
+ """
29
+
30
+ DATABRICKS = "databricks"
31
+ AZURE_BLOB = "azure_blob"
32
+ AZURE_SQL = "azure_sql"
33
+
34
+
35
+ class AzureConfig(BaseModel):
36
+ """Non-secret Azure connection details.
37
+
38
+ tenant_id / subscription_id are captured now as groundwork for the
39
+ later Azure CLI / Service Principal auth phase; the current manual-auth
40
+ connectors don't need them yet. storage_account/container and
41
+ sql_server/sql_database cover the two connector types this tool
42
+ actually talks to today (Blob Storage and Azure SQL Database).
43
+ """
44
+
45
+ tenant_id: Optional[str] = None
46
+ subscription_id: Optional[str] = None
47
+
48
+ storage_account: Optional[str] = None
49
+ container: Optional[str] = None
50
+
51
+ sql_server: Optional[str] = None
52
+ sql_database: Optional[str] = None
53
+
54
+
55
+ class DatabricksConfig(BaseModel):
56
+ """Non-secret Databricks connection details."""
57
+
58
+ workspace_url: Optional[str] = None
59
+ http_path: Optional[str] = None
60
+
61
+
62
+ class TableRef(BaseModel):
63
+ """
64
+ Reference to a table, schema, or whole catalog.
65
+
66
+ catalog is required. schema_name/table are optional: leaving
67
+ schema_name unset means "compare every schema common to both
68
+ catalogs"; leaving table unset (with schema_name set) means "compare
69
+ every table common to both sides of that schema". CatalogValidator's
70
+ compare_schemas/compare_tables perform this discovery internally
71
+ whenever the corresponding restriction is left unset.
72
+ """
73
+
74
+ catalog: Optional[str] = None
75
+ schema_name: Optional[str] = Field(default=None, alias="schema")
76
+ table: Optional[str] = None
77
+
78
+ model_config = {"populate_by_name": True}
79
+
80
+
81
+ class BlobSourceConfig(BaseModel):
82
+ """
83
+ Non-secret scoping for an Azure Blob Storage source (source_type =
84
+ azure_blob). The storage account/container credentials themselves
85
+ live on AzureConfig (storage_account/container, shared with any other
86
+ use of the same storage account); this section only scopes WHICH
87
+ blobs within that container are treated as comparison sources.
88
+ """
89
+
90
+ container: Optional[str] = None
91
+ folder_prefix: Optional[str] = Field(
92
+ default=None,
93
+ description=(
94
+ "Optional path prefix to restrict blob discovery to (e.g. "
95
+ "'validation/2024/'). If unset, the whole container is scanned."
96
+ ),
97
+ )
98
+ file_pattern: Optional[str] = Field(
99
+ default=None,
100
+ description=(
101
+ "Optional glob pattern to restrict blob discovery to (e.g. "
102
+ "'*.csv' or '*.parquet'). If unset, every supported file "
103
+ "extension (.csv/.txt/.xlsx/.xls/.parquet) is considered."
104
+ ),
105
+ )
106
+ blob_path: Optional[str] = Field(
107
+ default=None,
108
+ description=(
109
+ "Optional exact path to a single source blob (e.g. "
110
+ "'n8ndirectory/file_example_XLSX_100.csv'). If set together "
111
+ "with target_table.table, that exact blob is compared "
112
+ "directly against that exact table - bypassing filename-to-"
113
+ "table-name discovery entirely, even if the names don't "
114
+ "match. If unset, folder_prefix/file_pattern-based discovery "
115
+ "across multiple blobs applies as usual."
116
+ ),
117
+ )
118
+
119
+
120
+ class SqlSourceConfig(BaseModel):
121
+ """
122
+ Non-secret scoping for an Azure SQL Database source (source_type =
123
+ azure_sql). Server/database themselves live on AzureConfig
124
+ (sql_server/sql_database); this section scopes which schema/table
125
+ within that database are compared, same optional-means-"compare all"
126
+ convention as TableRef.
127
+ """
128
+
129
+ schema_name: Optional[str] = Field(default=None, alias="schema")
130
+ table: Optional[str] = None
131
+
132
+ model_config = {"populate_by_name": True}
133
+
134
+
135
+ class ValidatorConfig(BaseModel):
136
+ """
137
+ Top-level, non-secret configuration persisted to config.yaml.
138
+
139
+ source_type selects which of source_table/blob_source/sql_source is
140
+ read by `validate` - only the section matching source_type is
141
+ meaningful; the other two may be present (e.g. left over from
142
+ switching source types in the wizard) but are ignored. target_table
143
+ is always a Databricks catalog/schema/table ref regardless of
144
+ source_type, since every comparison path targets Databricks.
145
+ """
146
+
147
+ source_type: SourceType = Field(default=SourceType.DATABRICKS)
148
+
149
+ azure: AzureConfig = Field(default_factory=AzureConfig)
150
+ databricks: DatabricksConfig = Field(default_factory=DatabricksConfig)
151
+
152
+ source_table: TableRef = Field(default_factory=TableRef)
153
+ target_table: TableRef = Field(default_factory=TableRef)
154
+
155
+ primary_key: Optional[List[str]] = Field(
156
+ default=None,
157
+ description=(
158
+ "Optional primary/business key column(s) for the single named "
159
+ "table in source_table/target_table (only meaningful when both "
160
+ "are set to a specific table, not left blank for a catalog-"
161
+ "wide sweep). When set, row-level comparison uses this key "
162
+ "instead of falling back to a synthetic ROW_NUMBER() match - "
163
+ "cheaper and more reliable, and avoids the full-table sort "
164
+ "the row-number fallback needs on large tables. If unset, the "
165
+ "row-number fallback is used as before."
166
+ ),
167
+ )
168
+
169
+ blob_source: BlobSourceConfig = Field(default_factory=BlobSourceConfig)
170
+ sql_source: SqlSourceConfig = Field(default_factory=SqlSourceConfig)
171
+
172
+ validations: List[ValidationType] = Field(
173
+ default_factory=lambda: [
174
+ ValidationType.CATALOG,
175
+ ValidationType.SCHEMA,
176
+ ValidationType.COLUMN,
177
+ ValidationType.ROW,
178
+ ]
179
+ )
@@ -0,0 +1 @@
1
+ """Connectors package: I/O-only clients for Azure and Databricks (no comparison logic)."""