modelable 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of modelable might be problematic. Click here for more details.
- modelable/__init__.py +1 -0
- modelable/__main__.py +3 -0
- modelable/_pydantic_py314_compat.py +31 -0
- modelable/cli.py +41 -0
- modelable/commands/__init__.py +1 -0
- modelable/commands/apicurio.py +84 -0
- modelable/commands/codegen.py +241 -0
- modelable/commands/common.py +43 -0
- modelable/commands/compile.py +237 -0
- modelable/commands/create.py +164 -0
- modelable/commands/diff.py +82 -0
- modelable/commands/graph.py +53 -0
- modelable/commands/llm.py +564 -0
- modelable/commands/lsp.py +15 -0
- modelable/commands/runtime.py +37 -0
- modelable/commands/scenario.py +104 -0
- modelable/commands/spec.py +197 -0
- modelable/commands/workspace.py +240 -0
- modelable/compat/__init__.py +11 -0
- modelable/compat/checker.py +179 -0
- modelable/compat/diff.py +169 -0
- modelable/compiler/__init__.py +3 -0
- modelable/compiler/compiler.py +19 -0
- modelable/compiler/workspace.py +346 -0
- modelable/diagnostics/__init__.py +3 -0
- modelable/diagnostics/model.py +27 -0
- modelable/emitters/__init__.py +0 -0
- modelable/emitters/base.py +22 -0
- modelable/emitters/csharp.py +245 -0
- modelable/emitters/dbt_yaml.py +290 -0
- modelable/emitters/diagnostics.py +25 -0
- modelable/emitters/fhir.py +694 -0
- modelable/emitters/fhir_validator.py +36 -0
- modelable/emitters/go.py +334 -0
- modelable/emitters/java.py +264 -0
- modelable/emitters/json_schema.py +458 -0
- modelable/emitters/markdown.py +252 -0
- modelable/emitters/odcs.py +355 -0
- modelable/emitters/openlineage.py +315 -0
- modelable/emitters/openmetadata.py +258 -0
- modelable/emitters/python.py +282 -0
- modelable/emitters/rust.py +643 -0
- modelable/emitters/shapes.py +261 -0
- modelable/emitters/sql.py +266 -0
- modelable/emitters/targets.py +141 -0
- modelable/emitters/typescript.py +352 -0
- modelable/expressions/__init__.py +0 -0
- modelable/expressions/cel.py +547 -0
- modelable/governance/__init__.py +3 -0
- modelable/governance/checker.py +271 -0
- modelable/governance/por.py +46 -0
- modelable/grammar/__init__.py +1 -0
- modelable/grammar/modelable.lark +257 -0
- modelable/graph/__init__.py +5 -0
- modelable/graph/export.py +442 -0
- modelable/llm/__init__.py +43 -0
- modelable/llm/chat.py +255 -0
- modelable/llm/config.py +87 -0
- modelable/llm/context.py +194 -0
- modelable/llm/engine.py +976 -0
- modelable/llm/importers.py +1077 -0
- modelable/llm/provenance.py +84 -0
- modelable/llm/providers.py +182 -0
- modelable/llm/qa.py +126 -0
- modelable/llm/recommendations.py +33 -0
- modelable/llm/redaction.py +19 -0
- modelable/llm/render.py +279 -0
- modelable/llm/update_plan.py +101 -0
- modelable/llm/validation_help.py +10 -0
- modelable/lsp/__init__.py +3 -0
- modelable/lsp/__main__.py +4 -0
- modelable/lsp/code_actions.py +210 -0
- modelable/lsp/completion.py +480 -0
- modelable/lsp/definition.py +343 -0
- modelable/lsp/diagnostics.py +31 -0
- modelable/lsp/document_symbols.py +197 -0
- modelable/lsp/federation.py +261 -0
- modelable/lsp/folding.py +33 -0
- modelable/lsp/formatting.py +64 -0
- modelable/lsp/highlight.py +30 -0
- modelable/lsp/hover.py +370 -0
- modelable/lsp/inlay_hints.py +158 -0
- modelable/lsp/references.py +511 -0
- modelable/lsp/rename.py +564 -0
- modelable/lsp/semantic_tokens.py +412 -0
- modelable/lsp/server.py +370 -0
- modelable/lsp/workspace.py +83 -0
- modelable/lsp/workspace_symbols.py +104 -0
- modelable/parser/__init__.py +94 -0
- modelable/parser/ir.py +451 -0
- modelable/parser/parse.py +47 -0
- modelable/parser/transformer.py +798 -0
- modelable/parser/wire.py +68 -0
- modelable/planner/__init__.py +0 -0
- modelable/planner/lineage.py +91 -0
- modelable/planner/planner.py +134 -0
- modelable/planner/plans.py +122 -0
- modelable/py.typed +0 -0
- modelable/registry/__init__.py +9 -0
- modelable/registry/apicurio.py +166 -0
- modelable/registry/base.py +18 -0
- modelable/registry/factory.py +18 -0
- modelable/registry/index.py +419 -0
- modelable/registry/local.py +26 -0
- modelable/registry/oci.py +22 -0
- modelable/registry/resolver.py +213 -0
- modelable/registry/schema.sql +119 -0
- modelable/registry/signature.py +26 -0
- modelable/release.py +125 -0
- modelable/runtime/__init__.py +5 -0
- modelable/runtime/adapter/__init__.py +17 -0
- modelable/runtime/adapter/base.py +18 -0
- modelable/runtime/adapter/postgres.py +82 -0
- modelable/specs/__init__.py +23 -0
- modelable/specs/tracking.py +220 -0
- modelable/validation/__init__.py +3 -0
- modelable/validation/semantic.py +659 -0
- modelable-1.0.0.dist-info/METADATA +61 -0
- modelable-1.0.0.dist-info/RECORD +122 -0
- modelable-1.0.0.dist-info/WHEEL +4 -0
- modelable-1.0.0.dist-info/entry_points.txt +2 -0
- modelable-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,547 @@
|
|
|
1
|
+
"""CEL subset tokenizer, parser, validator, and lineage extractor for Modelable."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
|
|
8
|
+
# ── Tokens ────────────────────────────────────────────────────────────────────
|
|
9
|
+
|
|
10
|
+
_TOKEN_SPEC: list[tuple[str, str]] = [
|
|
11
|
+
("STRING", r'"(?:[^"\\]|\\.)*"|\'(?:[^\'\\]|\\.)*\''),
|
|
12
|
+
("FLOAT", r"\d+\.\d+"),
|
|
13
|
+
("INT", r"\d+"),
|
|
14
|
+
("AND", r"&&"),
|
|
15
|
+
("OR", r"\|\|"),
|
|
16
|
+
("NEQ", r"!="),
|
|
17
|
+
("LTE", r"<="),
|
|
18
|
+
("GTE", r">="),
|
|
19
|
+
("EQ", r"=="),
|
|
20
|
+
("BANG", r"!"),
|
|
21
|
+
("PLUS", r"\+"),
|
|
22
|
+
("MINUS", r"-"),
|
|
23
|
+
("STAR", r"\*"),
|
|
24
|
+
("SLASH", r"/"),
|
|
25
|
+
("PERCENT", r"%"),
|
|
26
|
+
("LT", r"<"),
|
|
27
|
+
("GT", r">"),
|
|
28
|
+
("QUESTION", r"\?"),
|
|
29
|
+
("COLON", r":"),
|
|
30
|
+
("DOT", r"\."),
|
|
31
|
+
("LPAREN", r"\("),
|
|
32
|
+
("RPAREN", r"\)"),
|
|
33
|
+
("LBRACE", r"\{"),
|
|
34
|
+
("RBRACE", r"\}"),
|
|
35
|
+
("LBRACKET", r"\["),
|
|
36
|
+
("RBRACKET", r"\]"),
|
|
37
|
+
("COMMA", r","),
|
|
38
|
+
("IDENT", r"[A-Za-z_][A-Za-z0-9_]*"),
|
|
39
|
+
("WS", r"\s+"),
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
_MASTER_RE = re.compile("|".join(f"(?P<{name}>{pattern})" for name, pattern in _TOKEN_SPEC))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass(frozen=True)
|
|
46
|
+
class Token:
|
|
47
|
+
type: str
|
|
48
|
+
value: str
|
|
49
|
+
pos: int
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class CelParseError(Exception):
|
|
53
|
+
def __init__(self, msg: str, pos: int = -1) -> None:
|
|
54
|
+
super().__init__(msg)
|
|
55
|
+
self.pos = pos
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _tokenize(expr: str) -> list[Token]:
|
|
59
|
+
# Strip inline comments before tokenizing
|
|
60
|
+
comment_pos = expr.find("//")
|
|
61
|
+
if comment_pos != -1:
|
|
62
|
+
expr = expr[:comment_pos]
|
|
63
|
+
tokens: list[Token] = []
|
|
64
|
+
for m in _MASTER_RE.finditer(expr):
|
|
65
|
+
kind = m.lastgroup
|
|
66
|
+
assert kind is not None
|
|
67
|
+
if kind == "WS":
|
|
68
|
+
continue
|
|
69
|
+
value = m.group()
|
|
70
|
+
pos = m.start()
|
|
71
|
+
if kind == "IDENT":
|
|
72
|
+
if value == "true":
|
|
73
|
+
kind = "TRUE"
|
|
74
|
+
elif value == "false":
|
|
75
|
+
kind = "FALSE"
|
|
76
|
+
elif value == "null":
|
|
77
|
+
kind = "NULL"
|
|
78
|
+
elif value == "in":
|
|
79
|
+
kind = "IN"
|
|
80
|
+
elif value == "where":
|
|
81
|
+
kind = "WHERE"
|
|
82
|
+
tokens.append(Token(type=kind, value=value, pos=pos))
|
|
83
|
+
tokens.append(Token(type="EOF", value="", pos=len(expr)))
|
|
84
|
+
return tokens
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
# ── AST ───────────────────────────────────────────────────────────────────────
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@dataclass
|
|
91
|
+
class Literal:
|
|
92
|
+
value: str | int | float | bool | None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@dataclass
|
|
96
|
+
class FieldRef:
|
|
97
|
+
"""alias.field reference against a declared source or join."""
|
|
98
|
+
|
|
99
|
+
alias: str
|
|
100
|
+
field: str
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
@dataclass
|
|
104
|
+
class RuntimeRef:
|
|
105
|
+
"""request.x, auth.x, or params.x."""
|
|
106
|
+
|
|
107
|
+
namespace: str
|
|
108
|
+
name: str
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
@dataclass
|
|
112
|
+
class UnaryOp:
|
|
113
|
+
op: str
|
|
114
|
+
expr: CelExpr
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
@dataclass
|
|
118
|
+
class BinaryOp:
|
|
119
|
+
op: str
|
|
120
|
+
left: CelExpr
|
|
121
|
+
right: CelExpr
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
@dataclass
|
|
125
|
+
class TernaryOp:
|
|
126
|
+
cond: CelExpr
|
|
127
|
+
then_: CelExpr
|
|
128
|
+
else_: CelExpr
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@dataclass
|
|
132
|
+
class FunctionCall:
|
|
133
|
+
name: str
|
|
134
|
+
args: list[CelExpr] = field(default_factory=list)
|
|
135
|
+
where_filter: CelExpr | None = None
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
@dataclass
|
|
139
|
+
class ListLiteral:
|
|
140
|
+
items: list[CelExpr] = field(default_factory=list)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
@dataclass
|
|
144
|
+
class WildcardRef:
|
|
145
|
+
"""alias.* — selects all fields of a source alias."""
|
|
146
|
+
|
|
147
|
+
alias: str
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
@dataclass
|
|
151
|
+
class ObjectLiteral:
|
|
152
|
+
"""{ key: expr, ... } — inline record/map literal."""
|
|
153
|
+
|
|
154
|
+
pairs: list[tuple[str, CelExpr]] = field(default_factory=list)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
CelExpr = (
|
|
158
|
+
Literal
|
|
159
|
+
| FieldRef
|
|
160
|
+
| RuntimeRef
|
|
161
|
+
| UnaryOp
|
|
162
|
+
| BinaryOp
|
|
163
|
+
| TernaryOp
|
|
164
|
+
| FunctionCall
|
|
165
|
+
| ListLiteral
|
|
166
|
+
| WildcardRef
|
|
167
|
+
| ObjectLiteral
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
# ── Parser ────────────────────────────────────────────────────────────────────
|
|
171
|
+
|
|
172
|
+
_RUNTIME_NAMESPACES = frozenset({"request", "auth", "params", "env"})
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
class _Parser:
|
|
176
|
+
def __init__(self, tokens: list[Token]) -> None:
|
|
177
|
+
self._tokens = tokens
|
|
178
|
+
self._pos = 0
|
|
179
|
+
|
|
180
|
+
def _peek(self) -> Token:
|
|
181
|
+
return self._tokens[self._pos]
|
|
182
|
+
|
|
183
|
+
def _consume(self, expected: str | None = None) -> Token:
|
|
184
|
+
tok = self._tokens[self._pos]
|
|
185
|
+
if expected and tok.type != expected:
|
|
186
|
+
raise CelParseError(
|
|
187
|
+
f"expected {expected} but got {tok.type!r} ({tok.value!r})",
|
|
188
|
+
tok.pos,
|
|
189
|
+
)
|
|
190
|
+
self._pos += 1
|
|
191
|
+
return tok
|
|
192
|
+
|
|
193
|
+
def _at(self, *types: str) -> bool:
|
|
194
|
+
return self._peek().type in types
|
|
195
|
+
|
|
196
|
+
def parse(self) -> CelExpr:
|
|
197
|
+
expr = self._ternary()
|
|
198
|
+
if not self._at("EOF"):
|
|
199
|
+
tok = self._peek()
|
|
200
|
+
raise CelParseError(f"unexpected token {tok.value!r}", tok.pos)
|
|
201
|
+
return expr
|
|
202
|
+
|
|
203
|
+
def _ternary(self) -> CelExpr:
|
|
204
|
+
cond = self._or()
|
|
205
|
+
if self._at("QUESTION"):
|
|
206
|
+
self._consume()
|
|
207
|
+
then_ = self._or()
|
|
208
|
+
self._consume("COLON")
|
|
209
|
+
else_ = self._ternary()
|
|
210
|
+
return TernaryOp(cond=cond, then_=then_, else_=else_)
|
|
211
|
+
return cond
|
|
212
|
+
|
|
213
|
+
def _or(self) -> CelExpr:
|
|
214
|
+
left = self._and()
|
|
215
|
+
while self._at("OR"):
|
|
216
|
+
self._consume()
|
|
217
|
+
right = self._and()
|
|
218
|
+
left = BinaryOp(op="||", left=left, right=right)
|
|
219
|
+
return left
|
|
220
|
+
|
|
221
|
+
def _and(self) -> CelExpr:
|
|
222
|
+
left = self._not()
|
|
223
|
+
while self._at("AND"):
|
|
224
|
+
self._consume()
|
|
225
|
+
right = self._not()
|
|
226
|
+
left = BinaryOp(op="&&", left=left, right=right)
|
|
227
|
+
return left
|
|
228
|
+
|
|
229
|
+
def _not(self) -> CelExpr:
|
|
230
|
+
if self._at("BANG"):
|
|
231
|
+
self._consume()
|
|
232
|
+
return UnaryOp(op="!", expr=self._not())
|
|
233
|
+
return self._comparison()
|
|
234
|
+
|
|
235
|
+
def _comparison(self) -> CelExpr:
|
|
236
|
+
left = self._add()
|
|
237
|
+
if self._at("EQ", "NEQ", "LT", "LTE", "GT", "GTE", "IN"):
|
|
238
|
+
op = self._consume().value
|
|
239
|
+
right = self._add()
|
|
240
|
+
return BinaryOp(op=op, left=left, right=right)
|
|
241
|
+
return left
|
|
242
|
+
|
|
243
|
+
def _add(self) -> CelExpr:
|
|
244
|
+
left = self._mul()
|
|
245
|
+
while self._at("PLUS", "MINUS"):
|
|
246
|
+
op = self._consume().value
|
|
247
|
+
right = self._mul()
|
|
248
|
+
left = BinaryOp(op=op, left=left, right=right)
|
|
249
|
+
return left
|
|
250
|
+
|
|
251
|
+
def _mul(self) -> CelExpr:
|
|
252
|
+
left = self._unary()
|
|
253
|
+
while self._at("STAR", "SLASH", "PERCENT"):
|
|
254
|
+
op = self._consume().value
|
|
255
|
+
right = self._unary()
|
|
256
|
+
left = BinaryOp(op=op, left=left, right=right)
|
|
257
|
+
return left
|
|
258
|
+
|
|
259
|
+
def _unary(self) -> CelExpr:
|
|
260
|
+
if self._at("MINUS"):
|
|
261
|
+
self._consume()
|
|
262
|
+
return UnaryOp(op="-", expr=self._unary())
|
|
263
|
+
return self._primary()
|
|
264
|
+
|
|
265
|
+
def _primary(self) -> CelExpr:
|
|
266
|
+
tok = self._peek()
|
|
267
|
+
|
|
268
|
+
if tok.type == "LPAREN":
|
|
269
|
+
self._consume()
|
|
270
|
+
expr = self._ternary()
|
|
271
|
+
self._consume("RPAREN")
|
|
272
|
+
return expr
|
|
273
|
+
|
|
274
|
+
if tok.type == "LBRACKET":
|
|
275
|
+
return self._list()
|
|
276
|
+
|
|
277
|
+
if tok.type == "LBRACE":
|
|
278
|
+
return self._object()
|
|
279
|
+
|
|
280
|
+
if tok.type == "TRUE":
|
|
281
|
+
self._consume()
|
|
282
|
+
return Literal(value=True)
|
|
283
|
+
|
|
284
|
+
if tok.type == "FALSE":
|
|
285
|
+
self._consume()
|
|
286
|
+
return Literal(value=False)
|
|
287
|
+
|
|
288
|
+
if tok.type == "NULL":
|
|
289
|
+
self._consume()
|
|
290
|
+
return Literal(value=None)
|
|
291
|
+
|
|
292
|
+
if tok.type == "STRING":
|
|
293
|
+
self._consume()
|
|
294
|
+
raw = tok.value[1:-1]
|
|
295
|
+
unescaped = raw.replace('\\"', '"').replace("\\'", "'").replace("\\\\", "\\")
|
|
296
|
+
return Literal(value=unescaped)
|
|
297
|
+
|
|
298
|
+
if tok.type == "FLOAT":
|
|
299
|
+
self._consume()
|
|
300
|
+
return Literal(value=float(tok.value))
|
|
301
|
+
|
|
302
|
+
if tok.type == "INT":
|
|
303
|
+
self._consume()
|
|
304
|
+
return Literal(value=int(tok.value))
|
|
305
|
+
|
|
306
|
+
if tok.type == "IDENT":
|
|
307
|
+
return self._ident_or_call()
|
|
308
|
+
|
|
309
|
+
raise CelParseError(f"unexpected token {tok.value!r}", tok.pos)
|
|
310
|
+
|
|
311
|
+
def _ident_or_call(self) -> CelExpr:
|
|
312
|
+
name_tok = self._consume("IDENT")
|
|
313
|
+
name = name_tok.value
|
|
314
|
+
|
|
315
|
+
if self._at("LPAREN"):
|
|
316
|
+
self._consume()
|
|
317
|
+
args: list[CelExpr] = []
|
|
318
|
+
while not self._at("RPAREN", "EOF"):
|
|
319
|
+
# Skip named argument label: "name: value" → consume name and colon
|
|
320
|
+
if (
|
|
321
|
+
self._at("IDENT")
|
|
322
|
+
and self._pos + 1 < len(self._tokens)
|
|
323
|
+
and self._tokens[self._pos + 1].type == "COLON"
|
|
324
|
+
):
|
|
325
|
+
self._consume("IDENT")
|
|
326
|
+
self._consume("COLON")
|
|
327
|
+
args.append(self._ternary())
|
|
328
|
+
if self._at("COMMA"):
|
|
329
|
+
self._consume()
|
|
330
|
+
self._consume("RPAREN")
|
|
331
|
+
where_filter = None
|
|
332
|
+
if self._at("WHERE"):
|
|
333
|
+
self._consume()
|
|
334
|
+
where_filter = self._ternary()
|
|
335
|
+
return FunctionCall(name=name, args=args, where_filter=where_filter)
|
|
336
|
+
|
|
337
|
+
if self._at("DOT"):
|
|
338
|
+
self._consume()
|
|
339
|
+
if self._at("STAR"):
|
|
340
|
+
self._consume()
|
|
341
|
+
return WildcardRef(alias=name)
|
|
342
|
+
field_tok = self._consume("IDENT")
|
|
343
|
+
field_name = field_tok.value
|
|
344
|
+
if name in _RUNTIME_NAMESPACES:
|
|
345
|
+
return RuntimeRef(namespace=name, name=field_name)
|
|
346
|
+
return FieldRef(alias=name, field=field_name)
|
|
347
|
+
|
|
348
|
+
# Bare identifier — not valid in MVP CEL (alias.field required)
|
|
349
|
+
return FieldRef(alias="", field=name)
|
|
350
|
+
|
|
351
|
+
def _list(self) -> ListLiteral:
|
|
352
|
+
self._consume("LBRACKET")
|
|
353
|
+
items: list[CelExpr] = []
|
|
354
|
+
while not self._at("RBRACKET", "EOF"):
|
|
355
|
+
items.append(self._ternary())
|
|
356
|
+
if self._at("COMMA"):
|
|
357
|
+
self._consume()
|
|
358
|
+
self._consume("RBRACKET")
|
|
359
|
+
return ListLiteral(items=items)
|
|
360
|
+
|
|
361
|
+
def _object(self) -> ObjectLiteral:
|
|
362
|
+
self._consume("LBRACE")
|
|
363
|
+
pairs: list[tuple[str, CelExpr]] = []
|
|
364
|
+
while not self._at("RBRACE", "EOF"):
|
|
365
|
+
key_tok = self._consume("IDENT")
|
|
366
|
+
self._consume("COLON")
|
|
367
|
+
value = self._ternary()
|
|
368
|
+
pairs.append((key_tok.value, value))
|
|
369
|
+
if self._at("COMMA"):
|
|
370
|
+
self._consume()
|
|
371
|
+
self._consume("RBRACE")
|
|
372
|
+
return ObjectLiteral(pairs=pairs)
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def parse_cel(expression: str) -> tuple[CelExpr | None, list[str]]:
|
|
376
|
+
"""Parse a CEL expression string. Returns (ast, parse_errors)."""
|
|
377
|
+
try:
|
|
378
|
+
tokens = _tokenize(expression)
|
|
379
|
+
return _Parser(tokens).parse(), []
|
|
380
|
+
except CelParseError as exc:
|
|
381
|
+
return None, [f"CEL001: parse error: {exc}"]
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
# ── Validation ────────────────────────────────────────────────────────────────
|
|
385
|
+
|
|
386
|
+
_SCALAR_FUNCTIONS = frozenset(
|
|
387
|
+
{
|
|
388
|
+
"lower",
|
|
389
|
+
"upper",
|
|
390
|
+
"trim",
|
|
391
|
+
"contains",
|
|
392
|
+
"startsWith",
|
|
393
|
+
"endsWith",
|
|
394
|
+
"slice",
|
|
395
|
+
"date",
|
|
396
|
+
"daysBetween",
|
|
397
|
+
"date_diff",
|
|
398
|
+
"truncate",
|
|
399
|
+
"coalesce",
|
|
400
|
+
"toString",
|
|
401
|
+
"toDecimal",
|
|
402
|
+
"hashHmacSha256",
|
|
403
|
+
"hmac_sha256",
|
|
404
|
+
"now",
|
|
405
|
+
"today",
|
|
406
|
+
"decimal",
|
|
407
|
+
"round",
|
|
408
|
+
"collect",
|
|
409
|
+
}
|
|
410
|
+
)
|
|
411
|
+
|
|
412
|
+
_AGGREGATE_FUNCTIONS = frozenset({"count", "sum", "min", "max", "avg", "countif", "count_distinct", "mode"})
|
|
413
|
+
|
|
414
|
+
_NON_DETERMINISTIC_FUNCTIONS = frozenset({"random", "uuid", "currentUser"})
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
@dataclass
|
|
418
|
+
class CelContext:
|
|
419
|
+
"""Validation context built from a projection version's sources."""
|
|
420
|
+
|
|
421
|
+
source_fields: dict[str, set[str]]
|
|
422
|
+
has_group_by: bool
|
|
423
|
+
fqn: str
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
@dataclass
|
|
427
|
+
class CelValidationResult:
|
|
428
|
+
errors: list[str]
|
|
429
|
+
field_refs: list[tuple[str, str]]
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def validate_cel_expr(expr: CelExpr, context: CelContext) -> CelValidationResult:
|
|
433
|
+
"""Validate a parsed CEL expression against a projection context."""
|
|
434
|
+
errors: list[str] = []
|
|
435
|
+
refs: list[tuple[str, str]] = []
|
|
436
|
+
_walk(expr, context, errors, refs)
|
|
437
|
+
return CelValidationResult(errors=errors, field_refs=refs)
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def _walk(
|
|
441
|
+
expr: CelExpr,
|
|
442
|
+
ctx: CelContext,
|
|
443
|
+
errors: list[str],
|
|
444
|
+
refs: list[tuple[str, str]],
|
|
445
|
+
) -> None:
|
|
446
|
+
if isinstance(expr, Literal):
|
|
447
|
+
return
|
|
448
|
+
|
|
449
|
+
if isinstance(expr, FieldRef):
|
|
450
|
+
if expr.alias == "":
|
|
451
|
+
errors.append(
|
|
452
|
+
f"CEL002: {ctx.fqn}: bare identifier '{expr.field}' is not allowed — use alias.field notation"
|
|
453
|
+
)
|
|
454
|
+
return
|
|
455
|
+
if expr.alias not in ctx.source_fields:
|
|
456
|
+
errors.append(f"CEL002: {ctx.fqn}: unknown alias '{expr.alias}'")
|
|
457
|
+
return
|
|
458
|
+
if expr.field not in ctx.source_fields[expr.alias]:
|
|
459
|
+
errors.append(f"CEL002: {ctx.fqn}: unknown field '{expr.alias}.{expr.field}'")
|
|
460
|
+
return
|
|
461
|
+
refs.append((expr.alias, expr.field))
|
|
462
|
+
return
|
|
463
|
+
|
|
464
|
+
if isinstance(expr, RuntimeRef):
|
|
465
|
+
# Phase 1: accept all request/auth/params references without declaration check
|
|
466
|
+
return
|
|
467
|
+
|
|
468
|
+
if isinstance(expr, UnaryOp):
|
|
469
|
+
_walk(expr.expr, ctx, errors, refs)
|
|
470
|
+
return
|
|
471
|
+
|
|
472
|
+
if isinstance(expr, BinaryOp):
|
|
473
|
+
_walk(expr.left, ctx, errors, refs)
|
|
474
|
+
_walk(expr.right, ctx, errors, refs)
|
|
475
|
+
return
|
|
476
|
+
|
|
477
|
+
if isinstance(expr, TernaryOp):
|
|
478
|
+
_walk(expr.cond, ctx, errors, refs)
|
|
479
|
+
_walk(expr.then_, ctx, errors, refs)
|
|
480
|
+
_walk(expr.else_, ctx, errors, refs)
|
|
481
|
+
return
|
|
482
|
+
|
|
483
|
+
if isinstance(expr, FunctionCall):
|
|
484
|
+
name = expr.name
|
|
485
|
+
if name in _NON_DETERMINISTIC_FUNCTIONS:
|
|
486
|
+
errors.append(f"CEL007: {ctx.fqn}: non-deterministic function '{name}' is not allowed")
|
|
487
|
+
elif name not in _SCALAR_FUNCTIONS and name not in _AGGREGATE_FUNCTIONS:
|
|
488
|
+
errors.append(f"CEL005: {ctx.fqn}: unsupported function '{name}'")
|
|
489
|
+
# max/min with 2+ args act as scalar greatest/least, not as row aggregates
|
|
490
|
+
is_scalar_max_min = name in ("max", "min") and len(expr.args) > 1
|
|
491
|
+
if name in _AGGREGATE_FUNCTIONS and not ctx.has_group_by and not is_scalar_max_min:
|
|
492
|
+
errors.append(f"CEL006: {ctx.fqn}: aggregate function '{name}' used in projection without group by")
|
|
493
|
+
for arg in expr.args:
|
|
494
|
+
_walk(arg, ctx, errors, refs)
|
|
495
|
+
if expr.where_filter is not None:
|
|
496
|
+
_walk(expr.where_filter, ctx, errors, refs)
|
|
497
|
+
return
|
|
498
|
+
|
|
499
|
+
if isinstance(expr, ListLiteral):
|
|
500
|
+
for item in expr.items:
|
|
501
|
+
_walk(item, ctx, errors, refs)
|
|
502
|
+
return
|
|
503
|
+
|
|
504
|
+
if isinstance(expr, WildcardRef):
|
|
505
|
+
if expr.alias and expr.alias not in ctx.source_fields:
|
|
506
|
+
errors.append(f"CEL002: {ctx.fqn}: unknown alias '{expr.alias}'")
|
|
507
|
+
return
|
|
508
|
+
|
|
509
|
+
if isinstance(expr, ObjectLiteral):
|
|
510
|
+
for _, value in expr.pairs:
|
|
511
|
+
_walk(value, ctx, errors, refs)
|
|
512
|
+
return
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
# ── Lineage extraction ────────────────────────────────────────────────────────
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
def extract_field_refs(expr: CelExpr) -> list[tuple[str, str]]:
|
|
519
|
+
"""Collect all (alias, field_name) pairs referenced in a CEL expression."""
|
|
520
|
+
refs: list[tuple[str, str]] = []
|
|
521
|
+
_collect_refs(expr, refs)
|
|
522
|
+
return refs
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
def _collect_refs(expr: CelExpr, refs: list[tuple[str, str]]) -> None:
|
|
526
|
+
if isinstance(expr, FieldRef) and expr.alias:
|
|
527
|
+
refs.append((expr.alias, expr.field))
|
|
528
|
+
elif isinstance(expr, BinaryOp):
|
|
529
|
+
_collect_refs(expr.left, refs)
|
|
530
|
+
_collect_refs(expr.right, refs)
|
|
531
|
+
elif isinstance(expr, UnaryOp):
|
|
532
|
+
_collect_refs(expr.expr, refs)
|
|
533
|
+
elif isinstance(expr, TernaryOp):
|
|
534
|
+
_collect_refs(expr.cond, refs)
|
|
535
|
+
_collect_refs(expr.then_, refs)
|
|
536
|
+
_collect_refs(expr.else_, refs)
|
|
537
|
+
elif isinstance(expr, FunctionCall):
|
|
538
|
+
for arg in expr.args:
|
|
539
|
+
_collect_refs(arg, refs)
|
|
540
|
+
if expr.where_filter is not None:
|
|
541
|
+
_collect_refs(expr.where_filter, refs)
|
|
542
|
+
elif isinstance(expr, ListLiteral):
|
|
543
|
+
for item in expr.items:
|
|
544
|
+
_collect_refs(item, refs)
|
|
545
|
+
elif isinstance(expr, ObjectLiteral):
|
|
546
|
+
for _, value in expr.pairs:
|
|
547
|
+
_collect_refs(value, refs)
|