databricks-tellr 0.3.13.dev15__tar.gz → 0.3.13.dev17__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: databricks-tellr
3
- Version: 0.3.13.dev15
3
+ Version: 0.3.13.dev17
4
4
  Summary: Tellr deployment tooling for Databricks Apps
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -27,8 +27,6 @@ env:
27
27
  value: "${LAKEBASE_INSTANCE}"
28
28
  - name: LAKEBASE_SCHEMA
29
29
  value: "${LAKEBASE_SCHEMA}"
30
- - name: GOOGLE_OAUTH_ENCRYPTION_KEY
31
- value: "${GOOGLE_OAUTH_ENCRYPTION_KEY}"
32
30
  - name: LAKEBASE_TYPE
33
31
  value: "${LAKEBASE_TYPE}"
34
32
  - name: LAKEBASE_PG_HOST
@@ -69,3 +67,4 @@ env:
69
67
  # Set to "0" if you want to force-disable the fast path entirely.
70
68
  - name: HUASHU_PIPELINE_ENABLED
71
69
  value: "1"
70
+ $ENCRYPTION_KEY_BLOCK
@@ -30,6 +30,8 @@ from databricks.sdk.service.apps import (
30
30
  from databricks.sdk.service.database import DatabaseInstance
31
31
  from databricks.sdk.service.workspace import ImportFormat
32
32
 
33
+ from databricks_tellr.identifiers import validate_client_id, validate_schema_name
34
+
33
35
  # Autoscaling imports (Lakebase next-gen)
34
36
  try:
35
37
  from databricks.sdk.service.postgres import (
@@ -195,7 +197,6 @@ def create(
195
197
  client: WorkspaceClient | None = None,
196
198
  profile: str | None = None,
197
199
  config_yaml_path: str | None = None,
198
- encryption_key: str | None = None,
199
200
  mlflow_tracing: dict[str, str] | None = None,
200
201
  ) -> dict[str, Any]:
201
202
  """Deploy Tellr to Databricks Apps.
@@ -225,7 +226,6 @@ def create(
225
226
  client: External WorkspaceClient (optional)
226
227
  profile: Databricks CLI profile name (optional)
227
228
  config_yaml_path: Path to deployment config YAML (mutually exclusive with other args)
228
- encryption_key: Fernet key for Google OAuth encryption. Auto-generated if not provided.
229
229
  mlflow_tracing: Optional overrides for UC tracing env vars in generated ``app.yaml``.
230
230
  Keys: ``MLFLOW_TRACING_SQL_WAREHOUSE_ID``, ``TELLR_MLFLOW_UC_CATALOG``,
231
231
  ``TELLR_MLFLOW_UC_SCHEMA``, ``TELLR_MLFLOW_UC_TABLE_PREFIX``. With
@@ -258,7 +258,6 @@ def create(
258
258
  profile=profile,
259
259
  config_yaml_path=config_yaml_path,
260
260
  seed_databricks_defaults=False,
261
- encryption_key=encryption_key,
262
261
  mlflow_tracing=mlflow_tracing,
263
262
  )
264
263
 
@@ -288,8 +287,9 @@ def update(
288
287
  reset_database: If True, drop and recreate the schema (tables recreated on app startup)
289
288
  client: External WorkspaceClient (optional)
290
289
  profile: Databricks CLI profile name (optional)
291
- encryption_key: Fernet key for Google OAuth encryption. If not provided, the
292
- existing key is read from the deployed app.yaml to preserve encrypted data.
290
+ encryption_key: Override for the legacy Fernet key to carry forward
291
+ into the regenerated app.yaml (boot then migrates it into the
292
+ encryption_keys table). Default: read from the deployed app.yaml.
293
293
  mlflow_tracing: Optional overrides for UC tracing env vars (same keys as ``create``).
294
294
  Values from deployment YAML are not loaded on update; use this argument or
295
295
  ``TELLR_DEPLOY_MLFLOW_*`` environment variables.
@@ -333,7 +333,6 @@ def _create_databricks(
333
333
  profile: str | None = None,
334
334
  config_yaml_path: str | None = None,
335
335
  seed_databricks_defaults: bool = True,
336
- encryption_key: str | None = None,
337
336
  mlflow_tracing: dict[str, str] | None = None,
338
337
  ) -> dict[str, Any]:
339
338
  """Deploy Tellr to Databricks Apps with configurable seeding.
@@ -353,7 +352,6 @@ def _create_databricks(
353
352
  profile: Databricks CLI profile name (optional)
354
353
  config_yaml_path: Path to deployment config YAML (mutually exclusive with other args)
355
354
  seed_databricks_defaults: If True, seed Databricks-specific content on startup
356
- encryption_key: Fernet key for Google OAuth encryption. Auto-generated if not provided.
357
355
  mlflow_tracing: Optional overrides for UC tracing placeholders in ``app.yaml``.
358
356
 
359
357
  Returns:
@@ -417,11 +415,10 @@ def _create_databricks(
417
415
  lakebase_name,
418
416
  schema_name,
419
417
  seed_databricks_defaults=seed_databricks_defaults,
420
- encryption_key=encryption_key,
421
418
  lakebase_result=lakebase_result,
422
419
  mlflow_tracing=mlflow_subs,
423
420
  )
424
- print(" Generated app.yaml (with encryption key)")
421
+ print(" Generated app.yaml")
425
422
 
426
423
  print(f"Uploading to: {app_file_workspace_path}")
427
424
  _upload_files(ws, staging, app_file_workspace_path)
@@ -507,8 +504,9 @@ def _update_databricks(
507
504
  client: External WorkspaceClient (optional)
508
505
  profile: Databricks CLI profile name (optional)
509
506
  seed_databricks_defaults: If True, seed Databricks-specific content on startup
510
- encryption_key: Fernet key for Google OAuth encryption. If not provided, reads
511
- the existing key from the deployed app.yaml to preserve encrypted data.
507
+ encryption_key: Override for the legacy Fernet key to carry forward
508
+ into the regenerated app.yaml (boot then migrates it into the
509
+ encryption_keys table). Default: read from the deployed app.yaml.
512
510
  mlflow_tracing: Optional overrides for UC tracing placeholders in ``app.yaml``.
513
511
 
514
512
  Returns:
@@ -526,7 +524,7 @@ def _update_databricks(
526
524
  overrides=mlflow_tracing,
527
525
  )
528
526
 
529
- # Preserve the existing encryption key so we don't invalidate encrypted data
527
+ # Preserve the existing key so boot can migrate it (carry-forward).
530
528
  if not encryption_key:
531
529
  encryption_key = _read_existing_encryption_key(ws, app_file_workspace_path)
532
530
 
@@ -555,9 +553,9 @@ def _update_databricks(
555
553
  lakebase_name,
556
554
  schema_name,
557
555
  seed_databricks_defaults=seed_databricks_defaults,
558
- encryption_key=encryption_key,
559
556
  lakebase_result=lakebase_result,
560
557
  mlflow_tracing=mlflow_subs,
558
+ encryption_key=encryption_key,
561
559
  )
562
560
  _upload_files(ws, staging, app_file_workspace_path)
563
561
  print(" Files updated")
@@ -660,6 +658,11 @@ def _check_breaking_migrations(
660
658
  Raises:
661
659
  DeploymentError: If the user declines the migration.
662
660
  """
661
+ # MEDIUM-5: schema_name is interpolated into the row-count DDL below
662
+ # (identifiers can't be parameterized); validate before any interpolation,
663
+ # matching the other DDL sites in this module.
664
+ validate_schema_name(schema_name)
665
+
663
666
  try:
664
667
  conn, _ = _get_lakebase_connection(ws, lakebase_name, lakebase_result=lakebase_result)
665
668
  except Exception as e:
@@ -733,29 +736,34 @@ def _check_breaking_migrations(
733
736
  def _read_existing_encryption_key(
734
737
  ws: WorkspaceClient, workspace_path: str
735
738
  ) -> str | None:
736
- """Read the GOOGLE_OAUTH_ENCRYPTION_KEY from an existing deployed app.yaml.
739
+ """Read GOOGLE_OAUTH_ENCRYPTION_KEY from an existing deployed app.yaml.
737
740
 
738
- This preserves the encryption key across updates so that previously encrypted
739
- credentials and tokens remain decryptable.
741
+ Returns the key string, or None when app.yaml is readable but carries no
742
+ key entry (already migrated, or a fresh-era install).
740
743
 
741
- Returns:
742
- The encryption key string, or None if not found.
744
+ Raises DeploymentError when app.yaml cannot be downloaded or parsed:
745
+ silently treating an unreadable app.yaml as "no key" would skip the
746
+ CRITICAL-3 migration and let the new code generate a fresh key,
747
+ orphaning all existing ciphertext.
743
748
  """
744
749
  try:
745
750
  resp = ws.workspace.download(f"{workspace_path}/app.yaml")
746
751
  raw = resp.read() if hasattr(resp, "read") else resp
747
752
  content = raw.decode("utf-8") if isinstance(raw, bytes) else str(raw)
748
-
749
753
  existing = yaml.safe_load(content)
750
- for env_entry in existing.get("env", []):
751
- if env_entry.get("name") == "GOOGLE_OAUTH_ENCRYPTION_KEY":
752
- key = env_entry.get("value")
753
- if key:
754
- print(" Preserved existing encryption key from deployed app.yaml")
755
- return key
756
754
  except Exception as e:
757
- logger.warning("Could not read existing encryption key from app.yaml: %s", e)
755
+ raise DeploymentError(
756
+ f"Could not read the deployed app.yaml at {workspace_path}/app.yaml "
757
+ f"({e}). Aborting: the update must know whether a legacy encryption "
758
+ f"key needs relocating before it overwrites app.yaml."
759
+ ) from e
758
760
 
761
+ for env_entry in (existing or {}).get("env", []):
762
+ if env_entry.get("name") == "GOOGLE_OAUTH_ENCRYPTION_KEY":
763
+ key = env_entry.get("value")
764
+ if key:
765
+ print(" Found legacy encryption key in deployed app.yaml")
766
+ return key
759
767
  return None
760
768
 
761
769
 
@@ -1277,19 +1285,22 @@ def _write_app_yaml(
1277
1285
  lakebase_name: str,
1278
1286
  schema_name: str,
1279
1287
  seed_databricks_defaults: bool = False,
1280
- encryption_key: str | None = None,
1281
1288
  lakebase_result: dict[str, Any] | None = None,
1282
1289
  mlflow_tracing: dict[str, str] | None = None,
1290
+ encryption_key: str | None = None,
1283
1291
  ) -> None:
1284
1292
  """Generate app.yaml with environment variables.
1285
1293
 
1294
+ The Fernet encryption key is written ONLY when ``encryption_key`` is
1295
+ provided (the one-time legacy→table carry-forward). Boot seeds the
1296
+ encryption_keys table from it and then scrubs it from app.yaml. When
1297
+ ``encryption_key`` is None the app.yaml is keyless (steady state).
1298
+
1286
1299
  Args:
1287
1300
  staging_dir: Directory to write the app.yaml file
1288
1301
  lakebase_name: Lakebase instance name
1289
1302
  schema_name: Schema name
1290
1303
  seed_databricks_defaults: If True, include Databricks-specific content seeding
1291
- encryption_key: Fernet encryption key for Google OAuth credentials/tokens.
1292
- Auto-generated if not provided.
1293
1304
  lakebase_result: Result dict from _get_or_create_lakebase() with type info.
1294
1305
  mlflow_tracing: Resolved template keys for UC tracing (four entries). If
1295
1306
  omitted, values are taken only from ``TELLR_DEPLOY_MLFLOW_*`` env vars.
@@ -1300,12 +1311,6 @@ def _write_app_yaml(
1300
1311
  else:
1301
1312
  init_call = "init_database()"
1302
1313
 
1303
- if not encryption_key:
1304
- from cryptography.fernet import Fernet
1305
-
1306
- encryption_key = Fernet.generate_key().decode()
1307
- logger.info("Auto-generated GOOGLE_OAUTH_ENCRYPTION_KEY for deployment")
1308
-
1309
1314
  # Determine lakebase type info for env vars
1310
1315
  lakebase_type = (lakebase_result or {}).get("type", "provisioned")
1311
1316
  lakebase_pg_host = (lakebase_result or {}).get("host", "")
@@ -1315,12 +1320,20 @@ def _write_app_yaml(
1315
1320
  if mlflow_tracing is None:
1316
1321
  mlflow_tracing = _mlflow_substitutions_for_app_yaml()
1317
1322
 
1323
+ if encryption_key:
1324
+ key_block = (
1325
+ " - name: GOOGLE_OAUTH_ENCRYPTION_KEY\n"
1326
+ f' value: "{encryption_key}"'
1327
+ )
1328
+ else:
1329
+ key_block = ""
1330
+
1318
1331
  template_content = _load_template("app.yaml.template")
1319
1332
  content = Template(template_content).substitute(
1333
+ ENCRYPTION_KEY_BLOCK=key_block,
1320
1334
  LAKEBASE_INSTANCE=lakebase_name,
1321
1335
  LAKEBASE_SCHEMA=schema_name,
1322
1336
  INIT_DATABASE_CALL=init_call,
1323
- GOOGLE_OAUTH_ENCRYPTION_KEY=encryption_key,
1324
1337
  LAKEBASE_TYPE=lakebase_type,
1325
1338
  LAKEBASE_PG_HOST=lakebase_pg_host,
1326
1339
  LAKEBASE_PROJECT_ID=lakebase_project_id,
@@ -1567,6 +1580,9 @@ def _setup_database_schema(
1567
1580
  print(" Warning: Could not get app client ID - schema setup skipped")
1568
1581
  return
1569
1582
 
1583
+ validate_schema_name(schema_name)
1584
+ validate_client_id(client_id)
1585
+
1570
1586
  conn, _ = _get_lakebase_connection(ws, lakebase_name, lakebase_result=lakebase_result)
1571
1587
 
1572
1588
  try:
@@ -1597,6 +1613,10 @@ def _reset_schema(
1597
1613
  """
1598
1614
  client_id = _get_app_client_id(app)
1599
1615
 
1616
+ validate_schema_name(schema_name)
1617
+ if client_id:
1618
+ validate_client_id(client_id)
1619
+
1600
1620
  conn, _ = _get_lakebase_connection(ws, lakebase_name, lakebase_result=lakebase_result)
1601
1621
 
1602
1622
  try:
@@ -1624,6 +1644,9 @@ def _grant_schema_permissions(cur: Any, schema_name: str, client_id: str) -> Non
1624
1644
 
1625
1645
  This function only grants schema/table permissions — it does NOT create roles.
1626
1646
  """
1647
+ validate_schema_name(schema_name)
1648
+ validate_client_id(client_id)
1649
+
1627
1650
  # Verify the role exists before granting
1628
1651
  cur.execute(
1629
1652
  "SELECT 1 FROM pg_roles WHERE rolname = %s", (client_id,)
@@ -0,0 +1,37 @@
1
+ """Validation for identifiers interpolated into Postgres DDL (SDR-4437 MEDIUM-5).
2
+
3
+ Both inputs are config/platform-derived today (schema_name from deploy
4
+ config, client_id from the app SP), so this is hardening against future
5
+ user-derived values, not a live injection.
6
+
7
+ Lives only in this (deploy-tool) distribution: the app distribution has no
8
+ DDL-identifier interpolation site once the dead setup_lakebase_schema is
9
+ removed, so there is no counterpart to keep in sync.
10
+ """
11
+
12
+ import re
13
+
14
+ _SCHEMA_RE = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
15
+ # UUID (App SP client id) OR all-digits (str(service_principal_id) fallback
16
+ # in _get_app_client_id). Both are injection-safe charsets.
17
+ _CLIENT_ID_RE = re.compile(
18
+ r"^(?:[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}"
19
+ r"-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}|[0-9]+)$"
20
+ )
21
+
22
+
23
+ def validate_schema_name(schema: str) -> str:
24
+ """Return *schema* if it is a safe Postgres schema identifier; else raise."""
25
+ if not isinstance(schema, str) or not _SCHEMA_RE.match(schema):
26
+ raise ValueError(f"Invalid Postgres schema name: {schema!r}")
27
+ return schema
28
+
29
+
30
+ def validate_client_id(client_id: str) -> str:
31
+ """Return *client_id* if it is a UUID or numeric SP id; else raise."""
32
+ if not isinstance(client_id, str) or not _CLIENT_ID_RE.match(client_id):
33
+ raise ValueError(
34
+ f"Invalid service-principal client id (expected UUID or numeric id): "
35
+ f"{client_id!r}"
36
+ )
37
+ return client_id
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: databricks-tellr
3
- Version: 0.3.13.dev15
3
+ Version: 0.3.13.dev17
4
4
  Summary: Tellr deployment tooling for Databricks Apps
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -2,6 +2,7 @@ README.md
2
2
  pyproject.toml
3
3
  databricks_tellr/__init__.py
4
4
  databricks_tellr/deploy.py
5
+ databricks_tellr/identifiers.py
5
6
  databricks_tellr.egg-info/PKG-INFO
6
7
  databricks_tellr.egg-info/SOURCES.txt
7
8
  databricks_tellr.egg-info/dependency_links.txt
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "databricks-tellr"
7
- version = "0.3.13.dev15"
7
+ version = "0.3.13.dev17"
8
8
  description = "Tellr deployment tooling for Databricks Apps"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"