schematalog-core 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- schematalog/common/__init__.py +1 -0
- schematalog/common/avroform/__init__.py +15 -0
- schematalog/common/avroform/exceptions.py +2 -0
- schematalog/common/avroform/to_avro.py +208 -0
- schematalog/common/avroform/to_json_schema.py +124 -0
- schematalog/common/logging.py +141 -0
- schematalog/common/models.py +12 -0
- schematalog/common/validation.py +80 -0
- schematalog/domain/__init__.py +8 -0
- schematalog/domain/exceptions.py +13 -0
- schematalog/domain/schema.py +465 -0
- schematalog/testing/__init__.py +12 -0
- schematalog/testing/conformance.py +259 -0
- schematalog/testing/example_schema.json +66 -0
- schematalog/testing/samples.py +28 -0
- schematalog_core-0.1.0.dist-info/METADATA +69 -0
- schematalog_core-0.1.0.dist-info/RECORD +19 -0
- schematalog_core-0.1.0.dist-info/WHEEL +4 -0
- schematalog_core-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Layer-neutral utilities shared across the application (no dependency on other layers)."""
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Convert between JSON Schema and Apache Avro schemas.
|
|
2
|
+
|
|
3
|
+
A small, dependency-free converter covering the common subset of JSON Schema:
|
|
4
|
+
objects/records, primitives, arrays, enums, nullable fields (unions), nested
|
|
5
|
+
records, string formats that map to Avro logical types, and internal ``$ref``
|
|
6
|
+
pointers (``#/$defs/...``, inlined before conversion). Constructs with no clean Avro
|
|
7
|
+
equivalent (``oneOf``/``anyOf``/``allOf``, external or recursive ``$ref``) raise
|
|
8
|
+
:class:`AvroConversionError` rather than emitting an invalid schema.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from .exceptions import AvroConversionError
|
|
12
|
+
from .to_avro import to_avro
|
|
13
|
+
from .to_json_schema import to_json_schema
|
|
14
|
+
|
|
15
|
+
__all__ = ["AvroConversionError", "to_avro", "to_json_schema"]
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
"""JSON Schema -> Avro schema."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from .exceptions import AvroConversionError
|
|
7
|
+
|
|
8
|
+
AvroType = str | dict[str, Any] | list[Any]
|
|
9
|
+
|
|
10
|
+
# JSON Schema primitive `type` -> Avro primitive type.
|
|
11
|
+
_PRIMITIVES = {
|
|
12
|
+
"boolean": "boolean",
|
|
13
|
+
"integer": "long",
|
|
14
|
+
"number": "double",
|
|
15
|
+
"null": "null",
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
# JSON Schema string `format` -> (Avro base type, Avro logicalType).
|
|
19
|
+
_STRING_FORMATS = {
|
|
20
|
+
"date": ("int", "date"),
|
|
21
|
+
"time": ("int", "time-millis"),
|
|
22
|
+
"date-time": ("long", "timestamp-millis"),
|
|
23
|
+
"uuid": ("string", "uuid"),
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
# JSON Schema keywords we cannot faithfully express in Avro. (`$ref` is handled
|
|
27
|
+
# separately: internal pointers are inlined before conversion, see `_resolve_refs`.)
|
|
28
|
+
_UNSUPPORTED = ("oneOf", "anyOf", "allOf", "not")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def to_avro(
|
|
32
|
+
json_schema: dict[str, Any], *, name: str = "Record", namespace: str = ""
|
|
33
|
+
) -> AvroType:
|
|
34
|
+
"""Convert a JSON Schema into an Avro schema.
|
|
35
|
+
|
|
36
|
+
Args:
|
|
37
|
+
json_schema: The JSON Schema to convert.
|
|
38
|
+
name: Name for the root Avro record/enum (a schema ``title`` overrides it).
|
|
39
|
+
namespace: Optional Avro namespace for named types.
|
|
40
|
+
|
|
41
|
+
Returns:
|
|
42
|
+
The Avro schema as a JSON-compatible structure.
|
|
43
|
+
|
|
44
|
+
Raises:
|
|
45
|
+
AvroConversionError: If the schema uses unsupported constructs.
|
|
46
|
+
"""
|
|
47
|
+
return _convert(_resolve_refs(json_schema, json_schema, ()), name, namespace, set())
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _resolve_refs(schema: Any, root: dict, stack: tuple[str, ...]) -> Any:
|
|
51
|
+
"""Inline internal ``$ref`` pointers (``#/...``) against the document root.
|
|
52
|
+
|
|
53
|
+
Avro has no reference mechanism, so a fragment ref is expanded in place before
|
|
54
|
+
conversion - e.g. ``{"$ref": "#/$defs/country"}`` becomes the ``country`` subschema.
|
|
55
|
+
External refs (anything not ``#/...``) and cyclic refs cannot be expressed and raise.
|
|
56
|
+
"""
|
|
57
|
+
if isinstance(schema, list):
|
|
58
|
+
return [_resolve_refs(item, root, stack) for item in schema]
|
|
59
|
+
if not isinstance(schema, dict):
|
|
60
|
+
return schema
|
|
61
|
+
ref = schema.get("$ref")
|
|
62
|
+
if isinstance(ref, str):
|
|
63
|
+
return _resolve_ref(schema, ref, root, stack)
|
|
64
|
+
return {key: _resolve_refs(value, root, stack) for key, value in schema.items()}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _resolve_ref(schema: dict, ref: str, root: dict, stack: tuple[str, ...]) -> Any:
|
|
68
|
+
"""Expand a single ``$ref`` node, overlaying any sibling keys onto the target."""
|
|
69
|
+
if not ref.startswith("#/"):
|
|
70
|
+
raise AvroConversionError(f"Cannot resolve external $ref: {ref!r}.")
|
|
71
|
+
if ref in stack:
|
|
72
|
+
raise AvroConversionError(f"Cannot express recursive $ref in Avro: {ref!r}.")
|
|
73
|
+
target = _resolve_refs(_deref(ref, root), root, (*stack, ref))
|
|
74
|
+
# JSON Schema allows keys beside `$ref`; overlay them onto the resolved object.
|
|
75
|
+
siblings = {key: value for key, value in schema.items() if key != "$ref"}
|
|
76
|
+
if siblings and isinstance(target, dict):
|
|
77
|
+
return {**target, **_resolve_refs(siblings, root, stack)}
|
|
78
|
+
return target
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _deref(ref: str, root: dict) -> Any:
|
|
82
|
+
"""Resolve a ``#/a/b`` JSON Pointer against ``root`` (with ~0/~1 unescaping)."""
|
|
83
|
+
node: Any = root
|
|
84
|
+
for raw in ref[2:].split("/"): # drop the leading '#/'
|
|
85
|
+
token = raw.replace("~1", "/").replace("~0", "~")
|
|
86
|
+
if not isinstance(node, dict) or token not in node:
|
|
87
|
+
raise AvroConversionError(f"Cannot resolve $ref: {ref!r}.")
|
|
88
|
+
node = node[token]
|
|
89
|
+
return node
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _convert(schema: Any, name: str, namespace: str, seen: set[str]) -> AvroType:
|
|
93
|
+
if not isinstance(schema, dict):
|
|
94
|
+
raise AvroConversionError(f"Expected a schema object, got {type(schema).__name__}.")
|
|
95
|
+
for keyword in _UNSUPPORTED:
|
|
96
|
+
if keyword in schema:
|
|
97
|
+
raise AvroConversionError(f"Unsupported JSON Schema keyword: {keyword!r}.")
|
|
98
|
+
|
|
99
|
+
if "enum" in schema:
|
|
100
|
+
return _enum(schema, name, namespace, seen)
|
|
101
|
+
|
|
102
|
+
json_type = schema.get("type")
|
|
103
|
+
if isinstance(json_type, list):
|
|
104
|
+
return _union(json_type, schema, name, namespace, seen)
|
|
105
|
+
if json_type == "object":
|
|
106
|
+
return _object(schema, name, namespace, seen)
|
|
107
|
+
if json_type == "array":
|
|
108
|
+
return _array(schema, name, namespace, seen)
|
|
109
|
+
if json_type == "string":
|
|
110
|
+
return _string(schema)
|
|
111
|
+
if json_type in _PRIMITIVES:
|
|
112
|
+
return _PRIMITIVES[json_type]
|
|
113
|
+
raise AvroConversionError(f"Unsupported or missing JSON Schema type: {json_type!r}.")
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _object(schema: dict, name: str, namespace: str, seen: set[str]) -> AvroType:
|
|
117
|
+
additional = schema.get("additionalProperties")
|
|
118
|
+
if not schema.get("properties") and isinstance(additional, dict):
|
|
119
|
+
return {"type": "map", "values": _convert(additional, name, namespace, seen)}
|
|
120
|
+
|
|
121
|
+
record_name = _unique(_pascal(schema.get("title") or name), seen)
|
|
122
|
+
required = set(schema.get("required", []))
|
|
123
|
+
fields = []
|
|
124
|
+
for prop_name, prop_schema in schema.get("properties", {}).items():
|
|
125
|
+
field_type = _convert(prop_schema, prop_name, namespace, seen)
|
|
126
|
+
field: dict[str, Any] = {"name": prop_name, "type": field_type}
|
|
127
|
+
if isinstance(prop_schema, dict) and prop_schema.get("description"):
|
|
128
|
+
field["doc"] = prop_schema["description"]
|
|
129
|
+
if prop_name not in required:
|
|
130
|
+
field["type"] = _nullable(field_type)
|
|
131
|
+
field["default"] = None
|
|
132
|
+
fields.append(field)
|
|
133
|
+
|
|
134
|
+
record: dict[str, Any] = {"type": "record", "name": record_name, "fields": fields}
|
|
135
|
+
if namespace:
|
|
136
|
+
record["namespace"] = namespace
|
|
137
|
+
if schema.get("description"):
|
|
138
|
+
record["doc"] = schema["description"]
|
|
139
|
+
return record
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _array(schema: dict, name: str, namespace: str, seen: set[str]) -> AvroType:
|
|
143
|
+
items = schema.get("items")
|
|
144
|
+
if not isinstance(items, dict):
|
|
145
|
+
raise AvroConversionError("Array schema must declare an 'items' object.")
|
|
146
|
+
return {"type": "array", "items": _convert(items, f"{_pascal(name)}Item", namespace, seen)}
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _string(schema: dict) -> AvroType:
|
|
150
|
+
fmt = schema.get("format")
|
|
151
|
+
if isinstance(fmt, str) and fmt in _STRING_FORMATS:
|
|
152
|
+
base, logical = _STRING_FORMATS[fmt]
|
|
153
|
+
return {"type": base, "logicalType": logical}
|
|
154
|
+
return "string"
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _enum(schema: dict, name: str, namespace: str, seen: set[str]) -> AvroType:
|
|
158
|
+
symbols = schema["enum"]
|
|
159
|
+
valid = all(isinstance(s, str) and _is_avro_name(s) for s in symbols)
|
|
160
|
+
if valid and len(set(symbols)) == len(symbols):
|
|
161
|
+
enum: dict[str, Any] = {
|
|
162
|
+
"type": "enum",
|
|
163
|
+
"name": _unique(_pascal(name), seen),
|
|
164
|
+
"symbols": list(symbols),
|
|
165
|
+
}
|
|
166
|
+
if namespace:
|
|
167
|
+
enum["namespace"] = namespace
|
|
168
|
+
return enum
|
|
169
|
+
# Enums Avro can't represent (non-string or non-identifier values) degrade to string.
|
|
170
|
+
return "string"
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _union(types: list, schema: dict, name: str, namespace: str, seen: set[str]) -> AvroType:
|
|
174
|
+
base = {key: value for key, value in schema.items() if key not in ("type", "enum")}
|
|
175
|
+
parts: list[AvroType] = []
|
|
176
|
+
for json_type in types:
|
|
177
|
+
member = _convert({**base, "type": json_type}, name, namespace, seen)
|
|
178
|
+
if member not in parts:
|
|
179
|
+
parts.append(member)
|
|
180
|
+
# Avro convention: null first so a `null` default is valid.
|
|
181
|
+
if "null" in parts:
|
|
182
|
+
parts = ["null", *(part for part in parts if part != "null")]
|
|
183
|
+
return parts
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _nullable(avro_type: AvroType) -> AvroType:
|
|
187
|
+
if isinstance(avro_type, list):
|
|
188
|
+
return avro_type if "null" in avro_type else ["null", *avro_type]
|
|
189
|
+
return ["null", avro_type]
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def _pascal(value: str) -> str:
|
|
193
|
+
parts = re.split(r"[^0-9a-zA-Z]+", value)
|
|
194
|
+
name = "".join(part[:1].upper() + part[1:] for part in parts if part)
|
|
195
|
+
return name or "Record"
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _unique(name: str, seen: set[str]) -> str:
|
|
199
|
+
candidate, index = name, 1
|
|
200
|
+
while candidate in seen:
|
|
201
|
+
index += 1
|
|
202
|
+
candidate = f"{name}{index}"
|
|
203
|
+
seen.add(candidate)
|
|
204
|
+
return candidate
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _is_avro_name(value: str) -> bool:
|
|
208
|
+
return bool(re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", value))
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""Avro schema -> JSON Schema."""
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from .exceptions import AvroConversionError
|
|
6
|
+
|
|
7
|
+
JsonSchema = dict[str, Any]
|
|
8
|
+
|
|
9
|
+
# Avro primitive type -> JSON Schema.
|
|
10
|
+
_PRIMITIVES = {
|
|
11
|
+
"string": {"type": "string"},
|
|
12
|
+
"bytes": {"type": "string"},
|
|
13
|
+
"int": {"type": "integer"},
|
|
14
|
+
"long": {"type": "integer"},
|
|
15
|
+
"float": {"type": "number"},
|
|
16
|
+
"double": {"type": "number"},
|
|
17
|
+
"boolean": {"type": "boolean"},
|
|
18
|
+
"null": {"type": "null"},
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
# Avro logicalType -> JSON Schema.
|
|
22
|
+
_LOGICAL_TYPES = {
|
|
23
|
+
"date": {"type": "string", "format": "date"},
|
|
24
|
+
"time-millis": {"type": "string", "format": "time"},
|
|
25
|
+
"timestamp-millis": {"type": "string", "format": "date-time"},
|
|
26
|
+
"uuid": {"type": "string", "format": "uuid"},
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def to_json_schema(avro_schema: Any) -> JsonSchema:
|
|
31
|
+
"""Convert an Avro schema into a JSON Schema."""
|
|
32
|
+
return _convert(avro_schema)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _convert(avro: Any) -> JsonSchema:
|
|
36
|
+
if isinstance(avro, str):
|
|
37
|
+
return _primitive(avro)
|
|
38
|
+
if isinstance(avro, list):
|
|
39
|
+
return _union(avro)
|
|
40
|
+
if isinstance(avro, dict):
|
|
41
|
+
return _named(avro)
|
|
42
|
+
raise AvroConversionError(f"Unexpected Avro schema node: {avro!r}.")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _primitive(name: str) -> JsonSchema:
|
|
46
|
+
try:
|
|
47
|
+
return dict(_PRIMITIVES[name])
|
|
48
|
+
except KeyError:
|
|
49
|
+
raise AvroConversionError(f"Unknown Avro type: {name!r}.") from None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _named(avro: dict) -> JsonSchema:
|
|
53
|
+
logical = avro.get("logicalType")
|
|
54
|
+
if logical in _LOGICAL_TYPES:
|
|
55
|
+
return dict(_LOGICAL_TYPES[logical])
|
|
56
|
+
|
|
57
|
+
avro_type = avro.get("type")
|
|
58
|
+
if avro_type == "record":
|
|
59
|
+
return _record(avro)
|
|
60
|
+
if avro_type == "enum":
|
|
61
|
+
return {"type": "string", "enum": list(avro["symbols"])}
|
|
62
|
+
if avro_type == "array":
|
|
63
|
+
return {"type": "array", "items": _convert(avro["items"])}
|
|
64
|
+
if avro_type == "map":
|
|
65
|
+
return {"type": "object", "additionalProperties": _convert(avro["values"])}
|
|
66
|
+
if avro_type == "fixed":
|
|
67
|
+
return {"type": "string"}
|
|
68
|
+
if isinstance(avro_type, str):
|
|
69
|
+
return _primitive(avro_type)
|
|
70
|
+
if isinstance(avro_type, (list, dict)):
|
|
71
|
+
return _convert(avro_type)
|
|
72
|
+
raise AvroConversionError(f"Unsupported Avro schema: {avro!r}.")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _record(avro: dict) -> JsonSchema:
|
|
76
|
+
properties: dict[str, Any] = {}
|
|
77
|
+
required: list[str] = []
|
|
78
|
+
for field in avro.get("fields", []):
|
|
79
|
+
field_type = field["type"]
|
|
80
|
+
optional = _is_nullable(field_type) or "default" in field
|
|
81
|
+
prop = _convert(_strip_null(field_type))
|
|
82
|
+
if field.get("doc"):
|
|
83
|
+
prop = {**prop, "description": field["doc"]}
|
|
84
|
+
properties[field["name"]] = prop
|
|
85
|
+
if not optional:
|
|
86
|
+
required.append(field["name"])
|
|
87
|
+
|
|
88
|
+
result: JsonSchema = {"type": "object", "properties": properties}
|
|
89
|
+
if required:
|
|
90
|
+
result["required"] = required
|
|
91
|
+
if avro.get("name"):
|
|
92
|
+
result["title"] = avro["name"]
|
|
93
|
+
if avro.get("doc"):
|
|
94
|
+
result["description"] = avro["doc"]
|
|
95
|
+
return result
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _union(members: list) -> JsonSchema:
|
|
99
|
+
non_null = [member for member in members if member != "null"]
|
|
100
|
+
nullable = "null" in members
|
|
101
|
+
if len(non_null) == 1:
|
|
102
|
+
schema = _convert(non_null[0])
|
|
103
|
+
return _with_null(schema) if nullable else schema
|
|
104
|
+
options = [_convert(member) for member in non_null]
|
|
105
|
+
if nullable:
|
|
106
|
+
options.append({"type": "null"})
|
|
107
|
+
return {"anyOf": options}
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _with_null(schema: JsonSchema) -> JsonSchema:
|
|
111
|
+
if set(schema) == {"type"} and isinstance(schema["type"], str):
|
|
112
|
+
return {"type": [schema["type"], "null"]}
|
|
113
|
+
return {"anyOf": [schema, {"type": "null"}]}
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _is_nullable(field_type: Any) -> bool:
|
|
117
|
+
return isinstance(field_type, list) and "null" in field_type
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _strip_null(field_type: Any) -> Any:
|
|
121
|
+
if not isinstance(field_type, list):
|
|
122
|
+
return field_type
|
|
123
|
+
non_null = [member for member in field_type if member != "null"]
|
|
124
|
+
return non_null[0] if len(non_null) == 1 else non_null
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
"""Structured logging: unclogger (JSON via structlog) plus a redaction safety net.
|
|
2
|
+
|
|
3
|
+
This is the one place logging is configured. The policy:
|
|
4
|
+
|
|
5
|
+
- `get_logger(name)` returns an `unclogger.Unclogger` that emits JSON; use it for
|
|
6
|
+
*our* events (presentation/application, and domain where a rule needs a trace).
|
|
7
|
+
We deliberately do not reroute uvicorn/FastAPI's own access logs through here -
|
|
8
|
+
they already log requests; duplicating them adds noise, not signal.
|
|
9
|
+
- `LogContext` is a typed mapping of the identifier fields that may be bound to the
|
|
10
|
+
context. The type checker rejects unknown keys, so a misspelled or contents-bearing
|
|
11
|
+
field is a type error rather than a silent leak.
|
|
12
|
+
- `bind_context` / `clear_context` bind those identifiers (request id) at a
|
|
13
|
+
request entry point so every downstream log inherits them without
|
|
14
|
+
the call site repeating them.
|
|
15
|
+
- A `sanitary.StructlogSanitizer` processor redacts credential-shaped keys and values
|
|
16
|
+
as a last-resort net. It is not a license to log secrets - it is the backstop for
|
|
17
|
+
when one slips through.
|
|
18
|
+
|
|
19
|
+
`common` is layer-neutral, so `configure_logging` takes `debug` as an argument rather
|
|
20
|
+
than importing `wiring.config`; the composition root passes `settings.DEBUG`.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import logging
|
|
24
|
+
import re
|
|
25
|
+
from typing import Final, TypedDict, Unpack
|
|
26
|
+
|
|
27
|
+
from sanitary import StructlogSanitizer
|
|
28
|
+
import unclogger
|
|
29
|
+
|
|
30
|
+
REDACTED: Final = "[REDACTED]"
|
|
31
|
+
|
|
32
|
+
# Value patterns are the primary net: they catch a credential-shaped *value* even
|
|
33
|
+
# under an innocuous field name. Matched against string values.
|
|
34
|
+
_FORBIDDEN_VALUE_PATTERNS: Final = (
|
|
35
|
+
re.compile(r"\bBearer\s+[A-Za-z0-9._~+/=-]{8,}", re.IGNORECASE), # bearer token
|
|
36
|
+
re.compile(r"\beyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+"), # JWT (PropelAuth)
|
|
37
|
+
re.compile(r"://[^/\s:@]+:[^/\s@]+@"), # credentials in a URL/DSN userinfo
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
# Field names are the secondary fallback for a sensitive value logged under an obvious
|
|
41
|
+
# name. Matched case-insensitively by exact name (not substring).
|
|
42
|
+
_FORBIDDEN_KEYS: Final = frozenset(
|
|
43
|
+
{
|
|
44
|
+
"password",
|
|
45
|
+
"secret",
|
|
46
|
+
"token",
|
|
47
|
+
"access_token",
|
|
48
|
+
"session_token",
|
|
49
|
+
"pending_token",
|
|
50
|
+
"api_key",
|
|
51
|
+
"apikey",
|
|
52
|
+
"authorization",
|
|
53
|
+
"cookie",
|
|
54
|
+
"set-cookie",
|
|
55
|
+
"credentials",
|
|
56
|
+
}
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
# Noisy third-party loggers pinned to WARNING; raise only when actively debugging.
|
|
60
|
+
# Uvicorn is deliberately absent - we leave its access/error logs as-is.
|
|
61
|
+
_THIRD_PARTY_LOGGERS: Final = (
|
|
62
|
+
"sqlalchemy",
|
|
63
|
+
"sqlalchemy.engine",
|
|
64
|
+
"asyncio",
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
# `unknown_objects="deny"` is the runtime backstop: any object reaching the sanitizer
|
|
68
|
+
# without a `__sanitary_context__` hook is masked wholesale rather than walked via
|
|
69
|
+
# `vars()` (which would leak its every attribute). Scalars pass through untouched.
|
|
70
|
+
_sanitizer: Final = StructlogSanitizer(
|
|
71
|
+
keys=_FORBIDDEN_KEYS,
|
|
72
|
+
patterns=_FORBIDDEN_VALUE_PATTERNS,
|
|
73
|
+
replacement=REDACTED,
|
|
74
|
+
message=REDACTED,
|
|
75
|
+
unknown_objects="deny",
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class LogContext(TypedDict, total=False):
|
|
80
|
+
"""Identifier fields that may be bound to the logging context.
|
|
81
|
+
|
|
82
|
+
Only identifiers belong here, never contents. Binding via `bind_context` is
|
|
83
|
+
type-checked against this mapping, so an unknown or misspelled key is a type
|
|
84
|
+
error rather than a silently-leaked field.
|
|
85
|
+
|
|
86
|
+
Deliberately minimal (request correlation only): richer per-request attributes
|
|
87
|
+
and spans are left to the planned OpenTelemetry instrumentation rather than
|
|
88
|
+
hand-bound here.
|
|
89
|
+
"""
|
|
90
|
+
|
|
91
|
+
request_id: str
|
|
92
|
+
method: str
|
|
93
|
+
path: str
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def bind_context(**fields: Unpack[LogContext]) -> None:
|
|
97
|
+
"""Bind identifier fields to the logging context for all downstream logs.
|
|
98
|
+
|
|
99
|
+
Call at a request entry point so every log line emitted while handling the
|
|
100
|
+
request inherits the identifiers without repeating them.
|
|
101
|
+
"""
|
|
102
|
+
unclogger.context_bind(**fields)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def clear_context(*keys: str) -> None:
|
|
106
|
+
"""Clear bound context fields, or all of them if no keys are named."""
|
|
107
|
+
unclogger.context_clear(*keys)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
_configured = False
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def configure_logging(*, debug: bool = False) -> None:
|
|
114
|
+
"""Configure process-wide structured logging. Idempotent; safe to call repeatedly.
|
|
115
|
+
|
|
116
|
+
Registers the redaction processor, pins noisy third-party loggers to WARNING, and
|
|
117
|
+
sets the root level from `debug`. Does not touch uvicorn's loggers.
|
|
118
|
+
|
|
119
|
+
Args:
|
|
120
|
+
debug: When true, emit DEBUG-level logs; otherwise INFO.
|
|
121
|
+
"""
|
|
122
|
+
global _configured
|
|
123
|
+
if _configured:
|
|
124
|
+
return
|
|
125
|
+
unclogger.add_processors(_sanitizer)
|
|
126
|
+
for name in _THIRD_PARTY_LOGGERS:
|
|
127
|
+
logging.getLogger(name).setLevel(logging.WARNING)
|
|
128
|
+
unclogger.set_level(logging.DEBUG if debug else logging.INFO)
|
|
129
|
+
_configured = True
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def get_logger(name: str) -> unclogger.Unclogger:
|
|
133
|
+
"""Return a structured logger.
|
|
134
|
+
|
|
135
|
+
Args:
|
|
136
|
+
name: The logger name; by convention `__name__`.
|
|
137
|
+
|
|
138
|
+
Returns:
|
|
139
|
+
An `Unclogger` usable like a standard library logger, emitting JSON.
|
|
140
|
+
"""
|
|
141
|
+
return unclogger.get_logger(name)
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
from pydantic import BaseModel, ConfigDict
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class FrozenModel(BaseModel):
|
|
5
|
+
"""Base class for immutable, frozen Pydantic models.
|
|
6
|
+
|
|
7
|
+
Layer-neutral implementation base. Domain layers wrap this with semantic
|
|
8
|
+
aliases (e.g. `ValueObject` in `domain/schema.py`); presentation DTOs can
|
|
9
|
+
inherit directly. Mirrors gapmap's `common.models.FrozenModel`.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
model_config = ConfigDict(frozen=True)
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
from jsonschema.exceptions import SchemaError
|
|
2
|
+
from jsonschema.validators import validator_for
|
|
3
|
+
|
|
4
|
+
# Accepted JSON Schema metaschemas, ordered oldest to newest. `infer_metaschema`
|
|
5
|
+
# tries them newest-first (i.e. reversed), so the order matters.
|
|
6
|
+
META_SCHEMAS: list[str] = [
|
|
7
|
+
"https://json-schema.org/draft-04/schema",
|
|
8
|
+
"https://json-schema.org/draft-06/schema",
|
|
9
|
+
"https://json-schema.org/draft-07/schema",
|
|
10
|
+
"https://json-schema.org/draft/2019-09/schema",
|
|
11
|
+
"https://json-schema.org/draft/2020-12/schema",
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class IncompatibleSchemaError(Exception):
|
|
16
|
+
"""Raised if a schema does not conform to any of the accepted meta schemas."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def preprocess_schema(schema: dict) -> dict:
|
|
20
|
+
schema = validate_metaschema(schema)
|
|
21
|
+
schema = convert_openapi_nullable(schema)
|
|
22
|
+
schema.pop("$id", None)
|
|
23
|
+
return schema
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def validate_metaschema(schema: dict) -> dict:
|
|
27
|
+
"""Validate schema against declared metaschema.
|
|
28
|
+
|
|
29
|
+
If the schema does not declare the `$schema` keyword, infers the metaschema
|
|
30
|
+
and inserts the keyword.
|
|
31
|
+
"""
|
|
32
|
+
if "$schema" in schema:
|
|
33
|
+
validator = validator_for(schema)
|
|
34
|
+
try:
|
|
35
|
+
validator.check_schema(schema)
|
|
36
|
+
except SchemaError:
|
|
37
|
+
return infer_metaschema(schema)
|
|
38
|
+
else:
|
|
39
|
+
return schema
|
|
40
|
+
else:
|
|
41
|
+
return infer_metaschema(schema)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def infer_metaschema(schema: dict) -> dict:
|
|
45
|
+
"""Infers and declares the correct metaschema for the schema.
|
|
46
|
+
|
|
47
|
+
The metaschema is inferred by checking the schema against each
|
|
48
|
+
of the supported metaschema URLs, starting with the latest.
|
|
49
|
+
The first valid metaschema is inserted as the value of the
|
|
50
|
+
`$schema` keyword.
|
|
51
|
+
"""
|
|
52
|
+
for metaschema_url in reversed(META_SCHEMAS):
|
|
53
|
+
schema["$schema"] = metaschema_url
|
|
54
|
+
try:
|
|
55
|
+
validator_for(schema).check_schema(schema)
|
|
56
|
+
except SchemaError:
|
|
57
|
+
pass
|
|
58
|
+
else:
|
|
59
|
+
return schema
|
|
60
|
+
raise IncompatibleSchemaError
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def convert_openapi_nullable(schema: dict) -> dict:
|
|
64
|
+
"""Converts the OpenAPI-style nullable to a null type.
|
|
65
|
+
|
|
66
|
+
Prior to version 3.1, OpenAPI did not recognise the `null` type,
|
|
67
|
+
instead using a separate keyword `nullable = true` to declare
|
|
68
|
+
values that can be `null`. This functions removes this keyword
|
|
69
|
+
and adds `"null"` to the list of declared types for that field.
|
|
70
|
+
|
|
71
|
+
If the schema contains `properties`, it is applied recursively.
|
|
72
|
+
"""
|
|
73
|
+
if schema.pop("nullable", ...) is True:
|
|
74
|
+
if isinstance(schema["type"], str):
|
|
75
|
+
schema["type"] = [schema["type"]]
|
|
76
|
+
if "null" not in schema["type"]:
|
|
77
|
+
schema["type"].append("null")
|
|
78
|
+
for prop_name, prop in schema.get("properties", {}).items():
|
|
79
|
+
schema["properties"][prop_name] = convert_openapi_nullable(prop)
|
|
80
|
+
return schema
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""The domain: what a schema is, and the contract a storage backend implements.
|
|
2
|
+
|
|
3
|
+
`__version__` is this distribution's own. It is deliberately separate from the
|
|
4
|
+
application's: the contract a backend codes against changes on its own schedule, and a
|
|
5
|
+
backend pinning it should not be dragged along by a UI release.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
class SchemaConflictError(Exception):
|
|
2
|
+
"""Signifies that the schema version being stored already exists."""
|
|
3
|
+
|
|
4
|
+
def __init__(self, schema):
|
|
5
|
+
super().__init__(f"Conflict: Schema `{schema.name} v{schema.version}` already exists.")
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class UnknownSchemaError(Exception):
|
|
9
|
+
"""Signifies that the schema version being requested does not exist."""
|
|
10
|
+
|
|
11
|
+
def __init__(self, schema_name, version=""):
|
|
12
|
+
version = f" v{version}" if version else ""
|
|
13
|
+
super().__init__(f"Unknown schema: `{schema_name}{version}`.")
|