continuo-python-runtime 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- continuo_python_runtime/__init__.py +1 -0
- continuo_python_runtime/cli.py +142 -0
- continuo_python_runtime/conform.py +220 -0
- continuo_python_runtime/context.py +62 -0
- continuo_python_runtime/contract/__init__.py +1 -0
- continuo_python_runtime/contract/loader.py +363 -0
- continuo_python_runtime/contract/merge.py +80 -0
- continuo_python_runtime/contract/model.py +39 -0
- continuo_python_runtime/contract/paths.py +38 -0
- continuo_python_runtime/errors.py +36 -0
- continuo_python_runtime/harness.py +208 -0
- continuo_python_runtime/hashing.py +34 -0
- continuo_python_runtime/lint.py +295 -0
- continuo_python_runtime/types.py +208 -0
- continuo_python_runtime-0.1.0.dist-info/METADATA +193 -0
- continuo_python_runtime-0.1.0.dist-info/RECORD +18 -0
- continuo_python_runtime-0.1.0.dist-info/WHEEL +4 -0
- continuo_python_runtime-0.1.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,363 @@
|
|
|
1
|
+
"""Contract v1 loader and validator.
|
|
2
|
+
|
|
3
|
+
Parses YAML contract files into validated :class:`~continuo_python_runtime
|
|
4
|
+
.contract.model.Node` instances, raising :class:`ContractError` for any
|
|
5
|
+
structural or semantic violation.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
import yaml
|
|
14
|
+
|
|
15
|
+
from continuo_python_runtime.contract.model import (
|
|
16
|
+
CRITICALITIES,
|
|
17
|
+
EXTRA_COLUMNS_POLICIES,
|
|
18
|
+
Column,
|
|
19
|
+
Node,
|
|
20
|
+
)
|
|
21
|
+
from continuo_python_runtime.errors import ContractError
|
|
22
|
+
from continuo_python_runtime.types import parse_sql_type
|
|
23
|
+
|
|
24
|
+
_ALLOWED_KEYS = {
|
|
25
|
+
"schema",
|
|
26
|
+
"table",
|
|
27
|
+
"description",
|
|
28
|
+
"owner",
|
|
29
|
+
"schedule",
|
|
30
|
+
"criticality",
|
|
31
|
+
"script",
|
|
32
|
+
"extra_columns",
|
|
33
|
+
"reads",
|
|
34
|
+
"output_columns",
|
|
35
|
+
"content_hash",
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
_REQUIRED_STRING_FIELDS = ("schema", "table", "owner", "schedule", "script")
|
|
39
|
+
|
|
40
|
+
_ALLOWED_OUTPUT_COLUMN_KEYS = {"name", "type", "nullable"}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _mask_string_literals(sql: str) -> str | None:
|
|
44
|
+
"""Return ``sql`` with the contents of single/double-quoted string
|
|
45
|
+
literals removed, so a ``;`` scan doesn't trip on one embedded in a
|
|
46
|
+
literal (e.g. ``split_part(tags, ';', 1)``).
|
|
47
|
+
|
|
48
|
+
Handles doubled ``''`` escapes inside single-quoted strings. The quote
|
|
49
|
+
delimiters themselves are kept; only the interior characters are
|
|
50
|
+
dropped. The result is not valid SQL - it exists solely for the
|
|
51
|
+
semicolon check below.
|
|
52
|
+
|
|
53
|
+
Returns ``None`` if a single- or double-quoted literal is left
|
|
54
|
+
unterminated (no closing quote before the end of the string): silently
|
|
55
|
+
truncating in that case would drop everything after the opening quote
|
|
56
|
+
-- including any ``;`` it was hiding -- and let a multi-statement read
|
|
57
|
+
through undetected.
|
|
58
|
+
"""
|
|
59
|
+
result = []
|
|
60
|
+
i = 0
|
|
61
|
+
n = len(sql)
|
|
62
|
+
while i < n:
|
|
63
|
+
ch = sql[i]
|
|
64
|
+
if ch == "'":
|
|
65
|
+
result.append(ch)
|
|
66
|
+
i += 1
|
|
67
|
+
closed = False
|
|
68
|
+
while i < n:
|
|
69
|
+
if sql[i] == "'":
|
|
70
|
+
if i + 1 < n and sql[i + 1] == "'":
|
|
71
|
+
# doubled '' escape: part of the literal, drop both
|
|
72
|
+
i += 2
|
|
73
|
+
continue
|
|
74
|
+
result.append(sql[i])
|
|
75
|
+
i += 1
|
|
76
|
+
closed = True
|
|
77
|
+
break
|
|
78
|
+
i += 1
|
|
79
|
+
if not closed:
|
|
80
|
+
return None
|
|
81
|
+
continue
|
|
82
|
+
if ch == '"':
|
|
83
|
+
result.append(ch)
|
|
84
|
+
i += 1
|
|
85
|
+
closed = False
|
|
86
|
+
while i < n:
|
|
87
|
+
if sql[i] == '"':
|
|
88
|
+
result.append(sql[i])
|
|
89
|
+
i += 1
|
|
90
|
+
closed = True
|
|
91
|
+
break
|
|
92
|
+
i += 1
|
|
93
|
+
if not closed:
|
|
94
|
+
return None
|
|
95
|
+
continue
|
|
96
|
+
result.append(ch)
|
|
97
|
+
i += 1
|
|
98
|
+
return "".join(result)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _strip_leading_sql_comments(sql: str) -> str:
|
|
102
|
+
"""Strip leading whitespace, ``-- ...`` line comments, and ``/* ... */``
|
|
103
|
+
block comments from the start of ``sql``.
|
|
104
|
+
|
|
105
|
+
An unterminated ``/*`` block comment (no closing ``*/``) consumes the
|
|
106
|
+
rest of the string, so callers see an empty statement rather than a
|
|
107
|
+
crash. Bounded by ``len(sql) + 1`` iterations (each iteration strictly
|
|
108
|
+
shrinks ``text``) with an explicit trailing return, so every path
|
|
109
|
+
returns a definite ``str`` rather than relying on an unbounded
|
|
110
|
+
``while True`` loop that a type checker can't prove always exits via
|
|
111
|
+
``return``.
|
|
112
|
+
"""
|
|
113
|
+
text = sql
|
|
114
|
+
for _ in range(len(sql) + 1):
|
|
115
|
+
stripped = text.lstrip()
|
|
116
|
+
if stripped.startswith("--"):
|
|
117
|
+
newline = stripped.find("\n")
|
|
118
|
+
text = stripped[newline + 1 :] if newline != -1 else ""
|
|
119
|
+
continue
|
|
120
|
+
if stripped.startswith("/*"):
|
|
121
|
+
end = stripped.find("*/")
|
|
122
|
+
text = stripped[end + 2 :] if end != -1 else ""
|
|
123
|
+
continue
|
|
124
|
+
return stripped
|
|
125
|
+
return text.lstrip()
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _validate_read_shape(label: str, name: str, sql: str) -> None:
|
|
129
|
+
"""Enforce the §13.1 read shape: a single SELECT/WITH statement.
|
|
130
|
+
|
|
131
|
+
The read, after stripping whitespace and leading SQL comments, must
|
|
132
|
+
start with ``select``, ``with``, or a leading ``(`` (a parenthesized
|
|
133
|
+
SELECT), case-insensitive. It must contain no ``;`` except one optional
|
|
134
|
+
trailing semicolon; the semicolon scan is literal-aware, ignoring ``;``
|
|
135
|
+
characters inside single/double-quoted string literals. An unterminated
|
|
136
|
+
string literal is itself rejected -- it can otherwise hide arbitrary
|
|
137
|
+
trailing SQL (including a ``;``) from the scan.
|
|
138
|
+
Schema-qualification is left to the control plane's sqlglot validation.
|
|
139
|
+
|
|
140
|
+
Raises:
|
|
141
|
+
ContractError: If the shape is violated, naming the read.
|
|
142
|
+
"""
|
|
143
|
+
stripped = sql.strip()
|
|
144
|
+
body = stripped[:-1] if stripped.endswith(";") else stripped
|
|
145
|
+
masked = _mask_string_literals(body)
|
|
146
|
+
if masked is None:
|
|
147
|
+
raise ContractError(
|
|
148
|
+
f"{label}: read '{name}' has an unterminated string literal"
|
|
149
|
+
)
|
|
150
|
+
if ";" in masked:
|
|
151
|
+
raise ContractError(
|
|
152
|
+
f"{label}: 'reads.{name}' must be a single SQL statement "
|
|
153
|
+
f"(only one optional trailing semicolon allowed)"
|
|
154
|
+
)
|
|
155
|
+
lowered = _strip_leading_sql_comments(body).lower()
|
|
156
|
+
if not (
|
|
157
|
+
lowered.startswith("select")
|
|
158
|
+
or lowered.startswith("with")
|
|
159
|
+
or lowered.startswith("(")
|
|
160
|
+
):
|
|
161
|
+
raise ContractError(
|
|
162
|
+
f"{label}: 'reads.{name}' must start with SELECT or WITH"
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
class _StrictLoader(yaml.SafeLoader):
|
|
167
|
+
"""SafeLoader that rejects duplicate mapping keys instead of keeping the last."""
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _strict_construct_mapping(
|
|
171
|
+
loader: _StrictLoader, node: yaml.MappingNode, deep: bool = False
|
|
172
|
+
):
|
|
173
|
+
seen = set()
|
|
174
|
+
for key_node, _ in node.value:
|
|
175
|
+
key = loader.construct_object(key_node, deep=deep)
|
|
176
|
+
if key in seen:
|
|
177
|
+
raise ContractError(f"duplicate key {key!r} at {key_node.start_mark}")
|
|
178
|
+
seen.add(key)
|
|
179
|
+
return yaml.SafeLoader.construct_mapping(loader, node, deep=deep)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
_StrictLoader.add_constructor(
|
|
183
|
+
yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG, _strict_construct_mapping
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _node_label(raw: dict[str, Any], source: str) -> str:
|
|
188
|
+
"""Build a `source (schema.table)`-style label for error messages."""
|
|
189
|
+
schema = raw.get("schema")
|
|
190
|
+
table = raw.get("table")
|
|
191
|
+
if isinstance(schema, str) and schema and isinstance(table, str) and table:
|
|
192
|
+
return f"{source} ({schema}.{table})"
|
|
193
|
+
return source
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def parse_node(raw: dict[str, Any], source: str) -> Node:
|
|
197
|
+
"""Validate a single raw mapping and build a :class:`Node`.
|
|
198
|
+
|
|
199
|
+
``source`` is the originating filename; it appears in every error message.
|
|
200
|
+
"""
|
|
201
|
+
if not isinstance(raw, dict):
|
|
202
|
+
raise ContractError(
|
|
203
|
+
f"{source}: node must be a mapping, got {type(raw).__name__}"
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
label = _node_label(raw, source)
|
|
207
|
+
|
|
208
|
+
unknown = set(raw) - _ALLOWED_KEYS
|
|
209
|
+
if unknown:
|
|
210
|
+
raise ContractError(f"{label}: unknown key(s) {sorted(unknown)}")
|
|
211
|
+
|
|
212
|
+
for field in _REQUIRED_STRING_FIELDS:
|
|
213
|
+
value = raw.get(field)
|
|
214
|
+
if not isinstance(value, str) or not value.strip():
|
|
215
|
+
raise ContractError(
|
|
216
|
+
f"{label}: required field '{field}' must be a non-empty string"
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
schema = raw["schema"]
|
|
220
|
+
table = raw["table"]
|
|
221
|
+
owner = raw["owner"]
|
|
222
|
+
schedule = raw["schedule"]
|
|
223
|
+
script = raw["script"]
|
|
224
|
+
|
|
225
|
+
criticality = raw.get("criticality")
|
|
226
|
+
if not isinstance(criticality, str) or criticality not in CRITICALITIES:
|
|
227
|
+
raise ContractError(
|
|
228
|
+
f"{label}: 'criticality' must be one of {sorted(CRITICALITIES)}, got {criticality!r}"
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
extra_columns = raw.get("extra_columns", "raise")
|
|
232
|
+
if (
|
|
233
|
+
not isinstance(extra_columns, str)
|
|
234
|
+
or extra_columns not in EXTRA_COLUMNS_POLICIES
|
|
235
|
+
):
|
|
236
|
+
raise ContractError(
|
|
237
|
+
f"{label}: 'extra_columns' must be one of {sorted(EXTRA_COLUMNS_POLICIES)}, "
|
|
238
|
+
f"got {extra_columns!r}"
|
|
239
|
+
)
|
|
240
|
+
|
|
241
|
+
reads = raw.get("reads")
|
|
242
|
+
if not isinstance(reads, dict) or not reads:
|
|
243
|
+
raise ContractError(
|
|
244
|
+
f"{label}: 'reads' must be a non-empty mapping of name -> SQL"
|
|
245
|
+
)
|
|
246
|
+
for name, sql in reads.items():
|
|
247
|
+
if not isinstance(name, str) or not name.strip():
|
|
248
|
+
raise ContractError(
|
|
249
|
+
f"{label}: 'reads' name {name!r} must be a non-empty string"
|
|
250
|
+
)
|
|
251
|
+
if not isinstance(sql, str) or not sql.strip():
|
|
252
|
+
raise ContractError(
|
|
253
|
+
f"{label}: 'reads.{name}' must be a non-empty SQL string"
|
|
254
|
+
)
|
|
255
|
+
_validate_read_shape(label, name, sql)
|
|
256
|
+
|
|
257
|
+
raw_columns = raw.get("output_columns")
|
|
258
|
+
if not isinstance(raw_columns, list) or not raw_columns:
|
|
259
|
+
raise ContractError(f"{label}: 'output_columns' must be a non-empty list")
|
|
260
|
+
|
|
261
|
+
seen_names: set[str] = set()
|
|
262
|
+
columns: list[Column] = []
|
|
263
|
+
for entry in raw_columns:
|
|
264
|
+
if not isinstance(entry, dict):
|
|
265
|
+
raise ContractError(
|
|
266
|
+
f"{label}: each 'output_columns' entry must be a mapping, got {entry!r}"
|
|
267
|
+
)
|
|
268
|
+
unknown_col_keys = set(entry) - _ALLOWED_OUTPUT_COLUMN_KEYS
|
|
269
|
+
if unknown_col_keys:
|
|
270
|
+
raise ContractError(
|
|
271
|
+
f"{label}: output column has unknown key(s) {sorted(unknown_col_keys)}"
|
|
272
|
+
)
|
|
273
|
+
name = entry.get("name")
|
|
274
|
+
if not isinstance(name, str) or not name.strip():
|
|
275
|
+
raise ContractError(f"{label}: output column missing non-empty 'name'")
|
|
276
|
+
col_type = entry.get("type")
|
|
277
|
+
if not isinstance(col_type, str):
|
|
278
|
+
raise ContractError(
|
|
279
|
+
f"{label}: output column '{name}' has unsupported 'type' {col_type!r}"
|
|
280
|
+
)
|
|
281
|
+
try:
|
|
282
|
+
parse_sql_type(col_type)
|
|
283
|
+
except ContractError as e:
|
|
284
|
+
raise ContractError(
|
|
285
|
+
f"{label}: output column '{name}' has unsupported {str(e)}"
|
|
286
|
+
) from e
|
|
287
|
+
nullable = entry.get("nullable", True)
|
|
288
|
+
if not isinstance(nullable, bool):
|
|
289
|
+
raise ContractError(
|
|
290
|
+
f"{label}: output column '{name}' has non-boolean 'nullable' {nullable!r}"
|
|
291
|
+
)
|
|
292
|
+
if name in seen_names:
|
|
293
|
+
raise ContractError(f"{label}: duplicate column '{name}' in output_columns")
|
|
294
|
+
seen_names.add(name)
|
|
295
|
+
columns.append(Column(name=name, type=col_type, nullable=nullable))
|
|
296
|
+
|
|
297
|
+
description = raw.get("description", "")
|
|
298
|
+
if not isinstance(description, str):
|
|
299
|
+
raise ContractError(f"{label}: 'description' must be a string")
|
|
300
|
+
|
|
301
|
+
content_hash = raw.get("content_hash")
|
|
302
|
+
if content_hash is not None and not isinstance(content_hash, str):
|
|
303
|
+
raise ContractError(f"{label}: 'content_hash' must be a string")
|
|
304
|
+
|
|
305
|
+
return Node(
|
|
306
|
+
schema=schema,
|
|
307
|
+
table=table,
|
|
308
|
+
owner=owner,
|
|
309
|
+
schedule=schedule,
|
|
310
|
+
criticality=criticality,
|
|
311
|
+
script=script,
|
|
312
|
+
reads=dict(reads),
|
|
313
|
+
output_columns=tuple(columns),
|
|
314
|
+
description=description,
|
|
315
|
+
extra_columns=extra_columns,
|
|
316
|
+
content_hash=content_hash,
|
|
317
|
+
)
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def load_contract_dir(path: Path) -> list[Node]:
|
|
321
|
+
"""Load and validate every `*.yml`/`*.yaml` contract file under ``path``.
|
|
322
|
+
|
|
323
|
+
Raises `ContractError` if no nodes are found, if any file's document is
|
|
324
|
+
malformed, or if two nodes across files share the same `(schema, table)`.
|
|
325
|
+
"""
|
|
326
|
+
files = sorted(path.glob("*.yml")) + sorted(path.glob("*.yaml"))
|
|
327
|
+
|
|
328
|
+
nodes: list[Node] = []
|
|
329
|
+
relation_sources: dict[str, str] = {}
|
|
330
|
+
|
|
331
|
+
for file in files:
|
|
332
|
+
text = file.read_text()
|
|
333
|
+
try:
|
|
334
|
+
document = yaml.load(text, Loader=_StrictLoader) # noqa: S506 — SafeLoader subclass
|
|
335
|
+
except ContractError as exc:
|
|
336
|
+
raise ContractError(f"{file.name}: {exc}") from None
|
|
337
|
+
except yaml.YAMLError as exc:
|
|
338
|
+
raise ContractError(f"{file.name}: invalid YAML: {exc}") from None
|
|
339
|
+
if document is None:
|
|
340
|
+
document = {}
|
|
341
|
+
if not isinstance(document, dict):
|
|
342
|
+
raise ContractError(
|
|
343
|
+
f"{file.name}: contract document must be a mapping, "
|
|
344
|
+
f"got {type(document).__name__}"
|
|
345
|
+
)
|
|
346
|
+
raw_nodes = document.get("nodes")
|
|
347
|
+
if not isinstance(raw_nodes, list):
|
|
348
|
+
raise ContractError(f"{file.name}: 'nodes' must be a list")
|
|
349
|
+
|
|
350
|
+
for raw_node in raw_nodes:
|
|
351
|
+
node = parse_node(raw_node, file.name)
|
|
352
|
+
existing_source = relation_sources.get(node.relation)
|
|
353
|
+
if existing_source is not None:
|
|
354
|
+
raise ContractError(
|
|
355
|
+
f"duplicate node {node.relation} (in {existing_source} and {file.name})"
|
|
356
|
+
)
|
|
357
|
+
relation_sources[node.relation] = file.name
|
|
358
|
+
nodes.append(node)
|
|
359
|
+
|
|
360
|
+
if not nodes:
|
|
361
|
+
raise ContractError(f"no contract files found in {path}")
|
|
362
|
+
|
|
363
|
+
return nodes
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""Contract v1 merger: wire contract builder."""
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
import yaml
|
|
6
|
+
|
|
7
|
+
from continuo_python_runtime.contract.loader import load_contract_dir
|
|
8
|
+
from continuo_python_runtime.contract.model import CONTRACT_VERSION, Node
|
|
9
|
+
from continuo_python_runtime.contract.paths import resolve_script_path
|
|
10
|
+
from continuo_python_runtime.hashing import content_hash
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def node_entry(node: Node) -> dict:
|
|
14
|
+
"""Convert a Node to its wire form dict.
|
|
15
|
+
|
|
16
|
+
Returns the node as a dict with all fields, output_columns as list of dicts,
|
|
17
|
+
and NO content_hash. The nullable field is always present in output_columns.
|
|
18
|
+
"""
|
|
19
|
+
return {
|
|
20
|
+
"schema": node.schema,
|
|
21
|
+
"table": node.table,
|
|
22
|
+
"owner": node.owner,
|
|
23
|
+
"schedule": node.schedule,
|
|
24
|
+
"criticality": node.criticality,
|
|
25
|
+
"script": node.script,
|
|
26
|
+
"reads": node.reads,
|
|
27
|
+
"output_columns": [
|
|
28
|
+
{
|
|
29
|
+
"name": col.name,
|
|
30
|
+
"type": col.type,
|
|
31
|
+
"nullable": col.nullable,
|
|
32
|
+
}
|
|
33
|
+
for col in node.output_columns
|
|
34
|
+
],
|
|
35
|
+
"description": node.description,
|
|
36
|
+
"extra_columns": node.extra_columns,
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def build_wire_contract(contract_dir: Path, repo_root: Path, service: str) -> dict:
|
|
41
|
+
"""Build and return a wire contract document.
|
|
42
|
+
|
|
43
|
+
Loads contracts from contract_dir, resolves script paths against repo_root,
|
|
44
|
+
computes content hashes, and returns a contract document sorted by relation.
|
|
45
|
+
|
|
46
|
+
Raises ContractError if any script file is missing.
|
|
47
|
+
"""
|
|
48
|
+
nodes = load_contract_dir(contract_dir)
|
|
49
|
+
|
|
50
|
+
wire_nodes = []
|
|
51
|
+
for node in nodes:
|
|
52
|
+
entry = node_entry(node)
|
|
53
|
+
|
|
54
|
+
script_path = resolve_script_path(node.script, repo_root, context=node.relation)
|
|
55
|
+
|
|
56
|
+
# Read script bytes and compute hash
|
|
57
|
+
script_bytes = script_path.read_bytes()
|
|
58
|
+
hash_value = content_hash(entry, script_bytes)
|
|
59
|
+
entry["content_hash"] = hash_value
|
|
60
|
+
|
|
61
|
+
wire_nodes.append(entry)
|
|
62
|
+
|
|
63
|
+
# Sort by relation (schema.table)
|
|
64
|
+
wire_nodes.sort(key=lambda entry: f"{entry['schema']}.{entry['table']}")
|
|
65
|
+
|
|
66
|
+
return {
|
|
67
|
+
"contract_version": CONTRACT_VERSION,
|
|
68
|
+
"service": service,
|
|
69
|
+
"nodes": wire_nodes,
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def write_wire_contract(doc: dict, out: Path) -> None:
|
|
74
|
+
"""Write a wire contract document to a YAML file.
|
|
75
|
+
|
|
76
|
+
Creates ``out``'s parent directory (and any missing ancestors) first, so
|
|
77
|
+
callers don't need to pre-create the output directory.
|
|
78
|
+
"""
|
|
79
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
80
|
+
out.write_text(yaml.safe_dump(doc, sort_keys=False))
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Contract v1 model."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
|
|
5
|
+
# Module-level constants
|
|
6
|
+
CRITICALITIES = frozenset({"REGULATORY", "CORE", "SECONDARY"})
|
|
7
|
+
EXTRA_COLUMNS_POLICIES = frozenset({"raise", "warn"})
|
|
8
|
+
CONTRACT_VERSION = 1
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True)
|
|
12
|
+
class Column:
|
|
13
|
+
"""A column definition in a table."""
|
|
14
|
+
|
|
15
|
+
name: str
|
|
16
|
+
type: str
|
|
17
|
+
nullable: bool = True
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class Node:
|
|
22
|
+
"""A node (task/process) definition."""
|
|
23
|
+
|
|
24
|
+
schema: str
|
|
25
|
+
table: str
|
|
26
|
+
owner: str
|
|
27
|
+
schedule: str
|
|
28
|
+
criticality: str
|
|
29
|
+
script: str
|
|
30
|
+
reads: dict[str, str]
|
|
31
|
+
output_columns: tuple[Column, ...]
|
|
32
|
+
description: str = ""
|
|
33
|
+
extra_columns: str = "raise"
|
|
34
|
+
content_hash: str | None = None
|
|
35
|
+
|
|
36
|
+
@property
|
|
37
|
+
def relation(self) -> str:
|
|
38
|
+
"""Return the fully qualified table name."""
|
|
39
|
+
return f"{self.schema}.{self.table}"
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Shared script-path resolution used by both the merger and the harness."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from continuo_python_runtime.errors import ContractError
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def resolve_script_path(script: str, repo_root: Path, *, context: str) -> Path:
|
|
11
|
+
"""Resolve ``script`` (relative to ``repo_root``) and validate it.
|
|
12
|
+
|
|
13
|
+
Rejects absolute paths, paths that escape ``repo_root`` once resolved,
|
|
14
|
+
and paths that don't point to an existing file.
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
script: The script path as declared in the contract, relative to
|
|
18
|
+
``repo_root``.
|
|
19
|
+
repo_root: The repository root the script path is resolved against.
|
|
20
|
+
context: A label prefixed to every error message (e.g. a node's
|
|
21
|
+
relation or a node id) to identify which node's script failed.
|
|
22
|
+
|
|
23
|
+
Raises:
|
|
24
|
+
ContractError: If the path is absolute, escapes the repository
|
|
25
|
+
root, or does not resolve to an existing file.
|
|
26
|
+
"""
|
|
27
|
+
if Path(script).is_absolute():
|
|
28
|
+
raise ContractError(
|
|
29
|
+
f"{context}: script path {script!r} must be relative to the repository root"
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
script_path = (repo_root / script).resolve()
|
|
33
|
+
if not script_path.is_relative_to(repo_root.resolve()):
|
|
34
|
+
raise ContractError(f"{context}: script path {script!r} escapes the repository root")
|
|
35
|
+
if not script_path.is_file():
|
|
36
|
+
raise ContractError(f"{context}: script not found: {script}")
|
|
37
|
+
|
|
38
|
+
return script_path
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Deterministic error taxonomy for the harness.
|
|
2
|
+
|
|
3
|
+
The sentinel result block's ``message`` starts with ``<ErrorClass>: `` so the
|
|
4
|
+
remediation classifier can key off it without parsing free text.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class HarnessError(Exception):
|
|
9
|
+
"""Base for all runtime failures the harness converts to a sentinel block."""
|
|
10
|
+
|
|
11
|
+
@property
|
|
12
|
+
def error_class(self) -> str:
|
|
13
|
+
return type(self).__name__
|
|
14
|
+
|
|
15
|
+
def sentinel_message(self) -> str:
|
|
16
|
+
return f"{self.error_class}: {self}"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ContractError(HarnessError):
|
|
20
|
+
"""Contract missing, invalid, node not found, or script missing."""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class ReadError(HarnessError):
|
|
24
|
+
"""Unknown read name, or a declared read failed at the warehouse."""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class ScriptError(HarnessError):
|
|
28
|
+
"""run() raised, has the wrong signature, or returned a non-Arrow-convertible value."""
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class ConformError(HarnessError):
|
|
32
|
+
"""Structural mismatch, strict-cast failure, or VARCHAR overflow."""
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class LoadError(HarnessError):
|
|
36
|
+
"""DDL or INSERT failure at the warehouse during the write."""
|