api-response-cleaner 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api_response_cleaner/__init__.py +8 -0
- api_response_cleaner/cleaner.py +199 -0
- api_response_cleaner/errors.py +39 -0
- api_response_cleaner/normalizer.py +80 -0
- api_response_cleaner/py.typed +1 -0
- api_response_cleaner/schema.py +149 -0
- api_response_cleaner/validator.py +32 -0
- api_response_cleaner-0.1.0.dist-info/METADATA +246 -0
- api_response_cleaner-0.1.0.dist-info/RECORD +12 -0
- api_response_cleaner-0.1.0.dist-info/WHEEL +5 -0
- api_response_cleaner-0.1.0.dist-info/licenses/LICENSE +21 -0
- api_response_cleaner-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
from .cleaner import Cleaner, CleanResult
|
|
2
|
+
from .schema import Schema, Field
|
|
3
|
+
from .errors import ErrorCode, FieldError, SchemaError, CleaningError
|
|
4
|
+
|
|
5
|
+
__version__ = "0.1.0"
|
|
6
|
+
|
|
7
|
+
def clean(data: dict, schema: Schema, aliases: dict = None, extra: str = "ignore") -> CleanResult:
|
|
8
|
+
return Cleaner(schema, aliases, extra).clean(data)
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import copy
|
|
3
|
+
from typing import Any, Dict, List, Optional, Tuple
|
|
4
|
+
from .errors import ErrorCode, FieldError, CleaningError, SchemaError
|
|
5
|
+
from .schema import Schema, Field
|
|
6
|
+
from .normalizer import normalize
|
|
7
|
+
from .validator import validate_constraints
|
|
8
|
+
|
|
9
|
+
class CleanResult:
|
|
10
|
+
def __init__(self, data: dict, errors: List[FieldError]):
|
|
11
|
+
self.data = data
|
|
12
|
+
self.errors = errors
|
|
13
|
+
self.ok = len(self.errors) == 0
|
|
14
|
+
|
|
15
|
+
def raise_for_errors(self):
|
|
16
|
+
if not self.ok:
|
|
17
|
+
raise CleaningError(self.errors)
|
|
18
|
+
|
|
19
|
+
class Cleaner:
|
|
20
|
+
def __init__(self, schema: Schema, aliases: Optional[Dict[str, List[str]]] = None, extra: str = "ignore"):
|
|
21
|
+
self.extra = extra
|
|
22
|
+
self.fields = {}
|
|
23
|
+
self.aliases_map = aliases or {}
|
|
24
|
+
|
|
25
|
+
self._all_known_names = set()
|
|
26
|
+
|
|
27
|
+
for k, v in schema.items():
|
|
28
|
+
field = Field(k, v)
|
|
29
|
+
if k in self.aliases_map:
|
|
30
|
+
field.aliases = self.aliases_map[k] + field.aliases
|
|
31
|
+
self.fields[k] = field
|
|
32
|
+
self._all_known_names.add(k)
|
|
33
|
+
|
|
34
|
+
alias_to_canon = {}
|
|
35
|
+
for k, f in self.fields.items():
|
|
36
|
+
for a in f.aliases:
|
|
37
|
+
if a in self._all_known_names:
|
|
38
|
+
raise SchemaError(f"Alias {a} for field {k} collides with field name")
|
|
39
|
+
if a in alias_to_canon:
|
|
40
|
+
raise SchemaError(f"Alias {a} is duplicated across fields")
|
|
41
|
+
alias_to_canon[a] = k
|
|
42
|
+
self._all_known_names.add(a)
|
|
43
|
+
|
|
44
|
+
for k in self.aliases_map:
|
|
45
|
+
if k not in self.fields:
|
|
46
|
+
raise SchemaError(f"aliases entry names unknown field: {k}")
|
|
47
|
+
|
|
48
|
+
def clean(self, data: Any) -> CleanResult:
|
|
49
|
+
if isinstance(data, (str, bytes)):
|
|
50
|
+
try:
|
|
51
|
+
data = json.loads(data)
|
|
52
|
+
except Exception:
|
|
53
|
+
return CleanResult({}, [FieldError("", ErrorCode.invalid_input, "Invalid JSON")])
|
|
54
|
+
|
|
55
|
+
if not isinstance(data, dict):
|
|
56
|
+
return CleanResult({}, [FieldError("", ErrorCode.invalid_input, "Top-level input must be an object")])
|
|
57
|
+
|
|
58
|
+
errors = []
|
|
59
|
+
result_data, _ = self._clean_object(data, self.fields, "", self.extra, errors)
|
|
60
|
+
|
|
61
|
+
return CleanResult(result_data, errors)
|
|
62
|
+
|
|
63
|
+
def _clean_object(self, input_obj: dict, fields: Dict[str, Field],
|
|
64
|
+
path_prefix: str, extra: str, errors: List[FieldError]) -> Tuple[dict, bool]:
|
|
65
|
+
out = {}
|
|
66
|
+
has_error = False
|
|
67
|
+
consumed_keys = set()
|
|
68
|
+
|
|
69
|
+
for name, field in fields.items():
|
|
70
|
+
path = f"{path_prefix}.{name}" if path_prefix else name
|
|
71
|
+
|
|
72
|
+
val = None
|
|
73
|
+
is_present = False
|
|
74
|
+
is_null = False
|
|
75
|
+
|
|
76
|
+
candidates = [name] + field.aliases
|
|
77
|
+
for cand in candidates:
|
|
78
|
+
if cand in input_obj:
|
|
79
|
+
consumed_keys.add(cand)
|
|
80
|
+
if input_obj[cand] is not None:
|
|
81
|
+
val = input_obj[cand]
|
|
82
|
+
is_present = True
|
|
83
|
+
break
|
|
84
|
+
else:
|
|
85
|
+
is_present = True
|
|
86
|
+
is_null = True
|
|
87
|
+
|
|
88
|
+
if not is_present:
|
|
89
|
+
if field.required:
|
|
90
|
+
errors.append(FieldError(path, ErrorCode.missing_required, "Missing required field"))
|
|
91
|
+
has_error = True
|
|
92
|
+
elif field.has_default:
|
|
93
|
+
out[name] = copy.deepcopy(field.default)
|
|
94
|
+
continue
|
|
95
|
+
|
|
96
|
+
if val is None and is_null:
|
|
97
|
+
if not field.nullable:
|
|
98
|
+
errors.append(FieldError(path, ErrorCode.null_not_allowed, "Null not allowed"))
|
|
99
|
+
has_error = True
|
|
100
|
+
else:
|
|
101
|
+
out[name] = None
|
|
102
|
+
continue
|
|
103
|
+
|
|
104
|
+
norm_val, err_code = normalize(val, field.type_str, field.strip)
|
|
105
|
+
if err_code:
|
|
106
|
+
errors.append(FieldError(path, err_code, "Conversion failed"))
|
|
107
|
+
has_error = True
|
|
108
|
+
continue
|
|
109
|
+
|
|
110
|
+
# Nested schema
|
|
111
|
+
if field.schema is not None:
|
|
112
|
+
inner_fields = {}
|
|
113
|
+
for ik, iv in field.schema.items():
|
|
114
|
+
inner_fields[ik] = Field(ik, iv)
|
|
115
|
+
|
|
116
|
+
inner_known = set(inner_fields.keys())
|
|
117
|
+
for ik, inner_f in inner_fields.items():
|
|
118
|
+
for a in inner_f.aliases:
|
|
119
|
+
if a in inner_known:
|
|
120
|
+
raise SchemaError(f"Alias {a} for field {ik} collides")
|
|
121
|
+
inner_known.add(a)
|
|
122
|
+
|
|
123
|
+
inner_out, inner_has_err = self._clean_object(norm_val, inner_fields, path, extra, errors)
|
|
124
|
+
if inner_has_err:
|
|
125
|
+
has_error = True
|
|
126
|
+
else:
|
|
127
|
+
out[name] = inner_out
|
|
128
|
+
continue
|
|
129
|
+
|
|
130
|
+
# List
|
|
131
|
+
if field.type_str == "list" and field.items is not None:
|
|
132
|
+
item_field = Field("item", field.items, is_items=True)
|
|
133
|
+
list_out = []
|
|
134
|
+
list_has_err = False
|
|
135
|
+
|
|
136
|
+
for i, item_val in enumerate(norm_val):
|
|
137
|
+
item_path = f"{path}[{i}]"
|
|
138
|
+
|
|
139
|
+
if item_val is None:
|
|
140
|
+
if not item_field.nullable:
|
|
141
|
+
errors.append(FieldError(item_path, ErrorCode.null_not_allowed, "Null not allowed in list"))
|
|
142
|
+
list_has_err = True
|
|
143
|
+
else:
|
|
144
|
+
list_out.append(None)
|
|
145
|
+
continue
|
|
146
|
+
|
|
147
|
+
item_norm, item_err = normalize(item_val, item_field.type_str, item_field.strip)
|
|
148
|
+
if item_err:
|
|
149
|
+
errors.append(FieldError(item_path, item_err, "List item conversion failed"))
|
|
150
|
+
list_has_err = True
|
|
151
|
+
continue
|
|
152
|
+
|
|
153
|
+
if item_field.schema is not None:
|
|
154
|
+
inner_fields = {ik: Field(ik, iv) for ik, iv in item_field.schema.items()}
|
|
155
|
+
inner_out, inner_err = self._clean_object(item_norm, inner_fields, item_path, extra, errors)
|
|
156
|
+
if inner_err:
|
|
157
|
+
list_has_err = True
|
|
158
|
+
else:
|
|
159
|
+
list_out.append(inner_out)
|
|
160
|
+
else:
|
|
161
|
+
item_errs = validate_constraints(
|
|
162
|
+
item_norm, item_field.type_str, item_field.min, item_field.max,
|
|
163
|
+
item_field.min_length, item_field.max_length, item_field.choices
|
|
164
|
+
)
|
|
165
|
+
if item_errs:
|
|
166
|
+
for c_err, c_msg in item_errs:
|
|
167
|
+
errors.append(FieldError(item_path, c_err, c_msg))
|
|
168
|
+
list_has_err = True
|
|
169
|
+
else:
|
|
170
|
+
list_out.append(item_norm)
|
|
171
|
+
|
|
172
|
+
if list_has_err:
|
|
173
|
+
has_error = True
|
|
174
|
+
continue
|
|
175
|
+
else:
|
|
176
|
+
norm_val = list_out
|
|
177
|
+
|
|
178
|
+
errs = validate_constraints(
|
|
179
|
+
norm_val, field.type_str, field.min, field.max,
|
|
180
|
+
field.min_length, field.max_length, field.choices
|
|
181
|
+
)
|
|
182
|
+
if errs:
|
|
183
|
+
for c_err, c_msg in errs:
|
|
184
|
+
errors.append(FieldError(path, c_err, c_msg))
|
|
185
|
+
has_error = True
|
|
186
|
+
else:
|
|
187
|
+
out[name] = norm_val
|
|
188
|
+
|
|
189
|
+
# Handle extra fields
|
|
190
|
+
for k, v in input_obj.items():
|
|
191
|
+
if k not in consumed_keys:
|
|
192
|
+
if extra == "error":
|
|
193
|
+
epath = f"{path_prefix}.{k}" if path_prefix else k
|
|
194
|
+
errors.append(FieldError(epath, ErrorCode.unexpected_field, "Unexpected field"))
|
|
195
|
+
has_error = True
|
|
196
|
+
elif extra == "keep":
|
|
197
|
+
out[k] = copy.deepcopy(v)
|
|
198
|
+
|
|
199
|
+
return out, has_error
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
from enum import Enum
|
|
2
|
+
|
|
3
|
+
class ErrorCode(str, Enum):
|
|
4
|
+
missing_required = "missing_required"
|
|
5
|
+
null_not_allowed = "null_not_allowed"
|
|
6
|
+
conversion_failed = "conversion_failed"
|
|
7
|
+
not_in_choices = "not_in_choices"
|
|
8
|
+
too_small = "too_small"
|
|
9
|
+
too_large = "too_large"
|
|
10
|
+
too_short = "too_short"
|
|
11
|
+
too_long = "too_long"
|
|
12
|
+
unexpected_field = "unexpected_field"
|
|
13
|
+
invalid_input = "invalid_input"
|
|
14
|
+
|
|
15
|
+
class FieldError:
|
|
16
|
+
def __init__(self, field: str, code: ErrorCode, message: str):
|
|
17
|
+
self.field = field
|
|
18
|
+
self.code = code
|
|
19
|
+
self.message = message
|
|
20
|
+
|
|
21
|
+
def to_dict(self) -> dict:
|
|
22
|
+
return {
|
|
23
|
+
"field": self.field,
|
|
24
|
+
"code": self.code.value,
|
|
25
|
+
"message": self.message
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
def __repr__(self) -> str:
|
|
29
|
+
return f"FieldError({self.field!r}, {self.code!r}, {self.message!r})"
|
|
30
|
+
|
|
31
|
+
class SchemaError(Exception):
|
|
32
|
+
"""Raised at schema build time for invalid schema definitions."""
|
|
33
|
+
pass
|
|
34
|
+
|
|
35
|
+
class CleaningError(Exception):
|
|
36
|
+
"""Raised when raise_for_errors() is called and errors exist."""
|
|
37
|
+
def __init__(self, errors: list[FieldError]):
|
|
38
|
+
self.errors = errors
|
|
39
|
+
super().__init__("Cleaning failed")
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
import math
|
|
2
|
+
from typing import Any, Tuple, Optional
|
|
3
|
+
from .errors import ErrorCode
|
|
4
|
+
|
|
5
|
+
def normalize(value: Any, type_str: str, strip: bool = True) -> Tuple[Any, Optional[ErrorCode]]:
|
|
6
|
+
if type_str == "object":
|
|
7
|
+
return value, None
|
|
8
|
+
|
|
9
|
+
if type_str == "str":
|
|
10
|
+
if isinstance(value, str):
|
|
11
|
+
return value.strip() if strip else value, None
|
|
12
|
+
if isinstance(value, (int, float)):
|
|
13
|
+
if isinstance(value, bool):
|
|
14
|
+
return None, ErrorCode.conversion_failed
|
|
15
|
+
if isinstance(value, float) and not math.isfinite(value):
|
|
16
|
+
return None, ErrorCode.conversion_failed
|
|
17
|
+
val = str(value)
|
|
18
|
+
return val.strip() if strip else val, None
|
|
19
|
+
return None, ErrorCode.conversion_failed
|
|
20
|
+
|
|
21
|
+
if type_str == "int":
|
|
22
|
+
if isinstance(value, bool):
|
|
23
|
+
return None, ErrorCode.conversion_failed
|
|
24
|
+
if isinstance(value, int):
|
|
25
|
+
return value, None
|
|
26
|
+
if isinstance(value, float):
|
|
27
|
+
if math.isfinite(value) and value.is_integer():
|
|
28
|
+
return int(value), None
|
|
29
|
+
return None, ErrorCode.conversion_failed
|
|
30
|
+
if isinstance(value, str):
|
|
31
|
+
try:
|
|
32
|
+
f_val = float(value)
|
|
33
|
+
if math.isfinite(f_val) and f_val.is_integer():
|
|
34
|
+
return int(f_val), None
|
|
35
|
+
except ValueError:
|
|
36
|
+
pass
|
|
37
|
+
return None, ErrorCode.conversion_failed
|
|
38
|
+
|
|
39
|
+
if type_str == "float":
|
|
40
|
+
if isinstance(value, bool):
|
|
41
|
+
return None, ErrorCode.conversion_failed
|
|
42
|
+
if isinstance(value, (int, float)):
|
|
43
|
+
if not math.isfinite(value):
|
|
44
|
+
return None, ErrorCode.conversion_failed
|
|
45
|
+
return float(value), None
|
|
46
|
+
if isinstance(value, str):
|
|
47
|
+
try:
|
|
48
|
+
f_val = float(value)
|
|
49
|
+
if not math.isfinite(f_val):
|
|
50
|
+
return None, ErrorCode.conversion_failed
|
|
51
|
+
return f_val, None
|
|
52
|
+
except ValueError:
|
|
53
|
+
pass
|
|
54
|
+
return None, ErrorCode.conversion_failed
|
|
55
|
+
|
|
56
|
+
if type_str == "bool":
|
|
57
|
+
if isinstance(value, bool):
|
|
58
|
+
return value, None
|
|
59
|
+
if isinstance(value, int) and not isinstance(value, bool):
|
|
60
|
+
if value in (0, 1):
|
|
61
|
+
return bool(value), None
|
|
62
|
+
if isinstance(value, str):
|
|
63
|
+
v_lower = value.strip().lower()
|
|
64
|
+
if v_lower in ("true", "yes", "y", "on", "1"):
|
|
65
|
+
return True, None
|
|
66
|
+
if v_lower in ("false", "no", "n", "off", "0"):
|
|
67
|
+
return False, None
|
|
68
|
+
return None, ErrorCode.conversion_failed
|
|
69
|
+
|
|
70
|
+
if type_str == "list":
|
|
71
|
+
if isinstance(value, (list, tuple)):
|
|
72
|
+
return list(value), None
|
|
73
|
+
return None, ErrorCode.conversion_failed
|
|
74
|
+
|
|
75
|
+
if type_str == "dict":
|
|
76
|
+
if isinstance(value, dict):
|
|
77
|
+
return value, None
|
|
78
|
+
return None, ErrorCode.conversion_failed
|
|
79
|
+
|
|
80
|
+
return None, ErrorCode.conversion_failed
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
import math
|
|
2
|
+
from typing import Any, Dict, List, Optional, Union
|
|
3
|
+
from .errors import SchemaError
|
|
4
|
+
from .normalizer import normalize
|
|
5
|
+
from .validator import validate_constraints
|
|
6
|
+
|
|
7
|
+
ALLOWED_KEYS = {
|
|
8
|
+
"type", "required", "default", "nullable", "aliases", "strip",
|
|
9
|
+
"choices", "min", "max", "min_length", "max_length", "schema", "items"
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
TYPE_MAP = {
|
|
13
|
+
str: "str", int: "int", float: "float", bool: "bool",
|
|
14
|
+
list: "list", dict: "dict", object: "object"
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
class Field:
|
|
18
|
+
def __init__(self, name: str, spec: Any, is_items: bool = False):
|
|
19
|
+
self.name = name
|
|
20
|
+
|
|
21
|
+
if isinstance(spec, type) and spec in TYPE_MAP:
|
|
22
|
+
spec = {"type": spec}
|
|
23
|
+
elif not isinstance(spec, dict):
|
|
24
|
+
raise SchemaError(f"Field {name} spec must be a dict or a supported type")
|
|
25
|
+
|
|
26
|
+
for k in spec:
|
|
27
|
+
if k not in ALLOWED_KEYS:
|
|
28
|
+
raise SchemaError(f"Unknown spec key {k} in field {name}")
|
|
29
|
+
|
|
30
|
+
type_val = spec.get("type", dict if "schema" in spec else None)
|
|
31
|
+
if type_val not in TYPE_MAP and type_val not in TYPE_MAP.values():
|
|
32
|
+
raise SchemaError(f"Unsupported type {type_val} in field {name}")
|
|
33
|
+
|
|
34
|
+
self.type_str = type_val if isinstance(type_val, str) else TYPE_MAP[type_val]
|
|
35
|
+
self.strip = spec.get("strip", True)
|
|
36
|
+
|
|
37
|
+
if is_items:
|
|
38
|
+
for bad_key in ("required", "default", "aliases"):
|
|
39
|
+
if bad_key in spec:
|
|
40
|
+
raise SchemaError(f"items spec cannot contain {bad_key}")
|
|
41
|
+
self.required = False
|
|
42
|
+
self.aliases = []
|
|
43
|
+
self.has_default = False
|
|
44
|
+
self.default = None
|
|
45
|
+
else:
|
|
46
|
+
self.required = spec.get("required", False)
|
|
47
|
+
self.aliases = spec.get("aliases", [])
|
|
48
|
+
if not isinstance(self.aliases, list):
|
|
49
|
+
raise SchemaError(f"aliases must be a list for field {name}")
|
|
50
|
+
self.has_default = "default" in spec
|
|
51
|
+
self.default = spec.get("default")
|
|
52
|
+
|
|
53
|
+
self.nullable = spec.get("nullable", False)
|
|
54
|
+
|
|
55
|
+
self.min = spec.get("min")
|
|
56
|
+
self.max = spec.get("max")
|
|
57
|
+
self.min_length = spec.get("min_length")
|
|
58
|
+
self.max_length = spec.get("max_length")
|
|
59
|
+
self.choices = spec.get("choices")
|
|
60
|
+
self.schema = spec.get("schema")
|
|
61
|
+
self.items = spec.get("items")
|
|
62
|
+
|
|
63
|
+
if self.schema is not None:
|
|
64
|
+
for ik, iv in self.schema.items():
|
|
65
|
+
Field(ik, iv)
|
|
66
|
+
|
|
67
|
+
if self.items is not None:
|
|
68
|
+
Field("item", self.items, is_items=True)
|
|
69
|
+
|
|
70
|
+
self._validate_constraints()
|
|
71
|
+
self._normalize_choices()
|
|
72
|
+
self._check_default()
|
|
73
|
+
|
|
74
|
+
def _validate_constraints(self):
|
|
75
|
+
if self.required and self.has_default:
|
|
76
|
+
raise SchemaError(f"Field {self.name} cannot be both required and have a default")
|
|
77
|
+
|
|
78
|
+
if self.min is not None or self.max is not None:
|
|
79
|
+
if self.type_str not in ("int", "float"):
|
|
80
|
+
raise SchemaError(f"min/max constraints used on unsupported type {self.type_str}")
|
|
81
|
+
if isinstance(self.min, bool) or isinstance(self.max, bool):
|
|
82
|
+
raise SchemaError("min/max cannot be bool")
|
|
83
|
+
if self.min is not None and not math.isfinite(self.min):
|
|
84
|
+
raise SchemaError("min must be finite")
|
|
85
|
+
if self.max is not None and not math.isfinite(self.max):
|
|
86
|
+
raise SchemaError("max must be finite")
|
|
87
|
+
if self.min is not None and self.max is not None and self.min > self.max:
|
|
88
|
+
raise SchemaError("min > max")
|
|
89
|
+
|
|
90
|
+
if self.min_length is not None or self.max_length is not None:
|
|
91
|
+
if self.type_str not in ("str", "list"):
|
|
92
|
+
raise SchemaError(f"min_length/max_length constraints used on unsupported type {self.type_str}")
|
|
93
|
+
if isinstance(self.min_length, bool) or isinstance(self.max_length, bool):
|
|
94
|
+
raise SchemaError("min_length/max_length cannot be bool")
|
|
95
|
+
if self.min_length is not None and (not isinstance(self.min_length, int) or self.min_length < 0):
|
|
96
|
+
raise SchemaError("min_length must be non-negative int")
|
|
97
|
+
if self.max_length is not None and (not isinstance(self.max_length, int) or self.max_length < 0):
|
|
98
|
+
raise SchemaError("max_length must be non-negative int")
|
|
99
|
+
if self.min_length is not None and self.max_length is not None and self.min_length > self.max_length:
|
|
100
|
+
raise SchemaError("min_length > max_length")
|
|
101
|
+
|
|
102
|
+
if self.choices is not None:
|
|
103
|
+
if self.type_str in ("list", "dict"):
|
|
104
|
+
raise SchemaError(f"choices constraint used on unsupported type {self.type_str}")
|
|
105
|
+
if isinstance(self.choices, (str, bytes)) or not hasattr(self.choices, '__iter__') or not self.choices:
|
|
106
|
+
raise SchemaError("choices must be a non-empty iterable (not str/bytes)")
|
|
107
|
+
|
|
108
|
+
def _normalize_choices(self):
|
|
109
|
+
if self.choices is not None:
|
|
110
|
+
normalized_choices = []
|
|
111
|
+
for c in self.choices:
|
|
112
|
+
val, err = normalize(c, self.type_str, self.strip)
|
|
113
|
+
if err:
|
|
114
|
+
raise SchemaError(f"Choice {c} fails normalization for type {self.type_str}")
|
|
115
|
+
normalized_choices.append(val)
|
|
116
|
+
self.choices = normalized_choices
|
|
117
|
+
|
|
118
|
+
def _check_default(self):
|
|
119
|
+
if not self.has_default:
|
|
120
|
+
return
|
|
121
|
+
|
|
122
|
+
if self.default is None:
|
|
123
|
+
return
|
|
124
|
+
|
|
125
|
+
if self.schema is not None:
|
|
126
|
+
if not isinstance(self.default, dict):
|
|
127
|
+
raise SchemaError("default for nested schema must be a dict")
|
|
128
|
+
return
|
|
129
|
+
|
|
130
|
+
if self.type_str == "list":
|
|
131
|
+
if not isinstance(self.default, list):
|
|
132
|
+
raise SchemaError("default for list field must be a list")
|
|
133
|
+
return
|
|
134
|
+
|
|
135
|
+
# Scalar default checking
|
|
136
|
+
val, err = normalize(self.default, self.type_str, self.strip)
|
|
137
|
+
if err:
|
|
138
|
+
raise SchemaError(f"Scalar default fails normalization: {err.name if hasattr(err, 'name') else err}")
|
|
139
|
+
|
|
140
|
+
errs = validate_constraints(
|
|
141
|
+
val, self.type_str, self.min, self.max,
|
|
142
|
+
self.min_length, self.max_length, self.choices
|
|
143
|
+
)
|
|
144
|
+
if errs:
|
|
145
|
+
raise SchemaError("Scalar default fails constraints")
|
|
146
|
+
|
|
147
|
+
self.default = val
|
|
148
|
+
|
|
149
|
+
Schema = Dict[str, Any]
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
from typing import Any, List, Tuple
|
|
2
|
+
from .errors import ErrorCode
|
|
3
|
+
|
|
4
|
+
def validate_constraints(
|
|
5
|
+
value: Any,
|
|
6
|
+
type_str: str,
|
|
7
|
+
min_val: Any = None,
|
|
8
|
+
max_val: Any = None,
|
|
9
|
+
min_length: Any = None,
|
|
10
|
+
max_length: Any = None,
|
|
11
|
+
choices: Any = None
|
|
12
|
+
) -> List[Tuple[ErrorCode, str]]:
|
|
13
|
+
errs = []
|
|
14
|
+
|
|
15
|
+
if choices is not None:
|
|
16
|
+
if value not in choices:
|
|
17
|
+
errs.append((ErrorCode.not_in_choices, f"Value not in choices"))
|
|
18
|
+
|
|
19
|
+
if type_str in ("int", "float"):
|
|
20
|
+
if min_val is not None and value < min_val:
|
|
21
|
+
errs.append((ErrorCode.too_small, f"Value is less than {min_val}"))
|
|
22
|
+
if max_val is not None and value > max_val:
|
|
23
|
+
errs.append((ErrorCode.too_large, f"Value is greater than {max_val}"))
|
|
24
|
+
|
|
25
|
+
if type_str in ("str", "list"):
|
|
26
|
+
length = len(value)
|
|
27
|
+
if min_length is not None and length < min_length:
|
|
28
|
+
errs.append((ErrorCode.too_short, f"Length {length} is less than {min_length}"))
|
|
29
|
+
if max_length is not None and length > max_length:
|
|
30
|
+
errs.append((ErrorCode.too_long, f"Length {length} is greater than {max_length}"))
|
|
31
|
+
|
|
32
|
+
return errs
|
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: api-response-cleaner
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Turns inconsistent API responses into validated Python data using a developer-defined schema.
|
|
5
|
+
Author: Narala Vamsi Krishna
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/VamsiKrishnaNarala/api-response-cleaner
|
|
8
|
+
Project-URL: Repository, https://github.com/VamsiKrishnaNarala/api-response-cleaner
|
|
9
|
+
Project-URL: Issues, https://github.com/VamsiKrishnaNarala/api-response-cleaner/issues
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Requires-Python: >=3.11
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Provides-Extra: dev
|
|
19
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
20
|
+
Requires-Dist: build; extra == "dev"
|
|
21
|
+
Requires-Dist: twine; extra == "dev"
|
|
22
|
+
Dynamic: license-file
|
|
23
|
+
|
|
24
|
+
# api-response-cleaner
|
|
25
|
+
|
|
26
|
+
Turns inconsistent API responses into validated Python data using a developer-defined schema. Zero dependencies.
|
|
27
|
+
|
|
28
|
+
## Why Use This Library? (Use Cases)
|
|
29
|
+
|
|
30
|
+
- **[Standardize third-party APIs](#example-a--standardizing-apis)**: Normalize inconsistent field types and response formats into a predictable output schema.
|
|
31
|
+
- **[Handle legacy and inconsistent data](#example-b--legacy-field-names-and-defaults)**: Support field aliases, optional fields, defaults, and explicit missing/null behavior.
|
|
32
|
+
- **Lightweight alternative to Pydantic**: A smaller, standard-library-runtime-dependency-free option for dictionary-based cleaning and validation when you don't need arbitrary object models.
|
|
33
|
+
- **[Validate nested JSON safely](#example-c--nested-data)**: Perform nested schema validation and error reporting without manual chains of `.get()` calls.
|
|
34
|
+
|
|
35
|
+
## Installation
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
pip install api-response-cleaner
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Quick Start
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from api_response_cleaner import clean
|
|
45
|
+
|
|
46
|
+
result = clean(
|
|
47
|
+
{"full_name": " Ravi ", "age": "25"},
|
|
48
|
+
schema={"name": {"type": str, "required": True},
|
|
49
|
+
"age": {"type": int, "required": True}},
|
|
50
|
+
aliases={"name": ["full_name"]},
|
|
51
|
+
)
|
|
52
|
+
print(result.data) # {"name": "Ravi", "age": 25}
|
|
53
|
+
print(result.errors) # []
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## Schema Reference
|
|
57
|
+
|
|
58
|
+
Define fields as dictionaries. Supported keys:
|
|
59
|
+
- `type`: `str`, `int`, `float`, `bool`, `list`, `dict`, `object`
|
|
60
|
+
- `required`: boolean, default `False`
|
|
61
|
+
- `default`: fallback value if key absent
|
|
62
|
+
- `nullable`: boolean, default `False`
|
|
63
|
+
- `aliases`: list of alternative keys
|
|
64
|
+
- `strip`: boolean, default `True` (for strings)
|
|
65
|
+
- `choices`: iterable of allowed values
|
|
66
|
+
- `min`, `max`: bounds for `int` and `float`
|
|
67
|
+
- `min_length`, `max_length`: length bounds for `str` and `list`
|
|
68
|
+
- `schema`: dictionary of sub-fields (implies `type` dict)
|
|
69
|
+
- `items`: type or spec for list elements
|
|
70
|
+
|
|
71
|
+
Shorthand: `{"age": int}` is equivalent to `{"age": {"type": int}}`.
|
|
72
|
+
|
|
73
|
+
## Missing and Null Decision Table
|
|
74
|
+
|
|
75
|
+
| Input state | optional | required | default=7 | nullable=True |
|
|
76
|
+
|-----------------------------|----------------|-----------------------|-----------|-----------------------|
|
|
77
|
+
| key absent | key omitted | missing_required | 7 | key omitted |
|
|
78
|
+
| key present, value null | key omitted | null_not_allowed | 7 | None (null preserved) |
|
|
79
|
+
| key present, value "25" | 25 | 25 | 25 | 25 |
|
|
80
|
+
| key present, value "" | conversion_failed in all four columns |
|
|
81
|
+
|
|
82
|
+
Note: `nullable=True` + `required=True` with key absent gives `missing_required`.
|
|
83
|
+
|
|
84
|
+
## Alias and Nested Rules
|
|
85
|
+
|
|
86
|
+
- Aliases are tried in order if canonical name is missing.
|
|
87
|
+
- If canonical name and alias are both present and non-null, canonical wins.
|
|
88
|
+
- Alias keys are consumed; they are not reported as extra or unexpected.
|
|
89
|
+
- Errors are reported on canonical paths.
|
|
90
|
+
- Nested schema values must be objects (dict).
|
|
91
|
+
- Inner errors are collected and reported with full dotted paths (e.g. `address.city`).
|
|
92
|
+
- If an inner error occurs, the whole nested key is omitted from result data.
|
|
93
|
+
|
|
94
|
+
## Structured Defaults Warning
|
|
95
|
+
|
|
96
|
+
A structured default (`list` or `dict`) is **trusted**. It is returned as-is (deep copied) and its contents are NOT normalized, constraint-checked, or mapped for aliases. Structured defaults must already be in their final valid form.
|
|
97
|
+
|
|
98
|
+
## Constraints
|
|
99
|
+
|
|
100
|
+
| Constraint | Supported Types |
|
|
101
|
+
|---------------------------|-------------------------|
|
|
102
|
+
| `min`, `max` | `int`, `float` |
|
|
103
|
+
| `min_length`, `max_length`| `str`, `list` |
|
|
104
|
+
| `choices` | `str`, `int`, `float`, `bool`, `object` |
|
|
105
|
+
|
|
106
|
+
## Conversion Rules
|
|
107
|
+
|
|
108
|
+
- `str`: Accepts strings, ints, finite floats. Strips whitespace by default.
|
|
109
|
+
- `int`: Accepts ints, whole floats (e.g., `25.0`), and numeric strings.
|
|
110
|
+
- `float`: Accepts finite floats, ints, and numeric strings.
|
|
111
|
+
- `bool`: Accepts booleans, ints `0` and `1`, strings `true`, `false`, `yes`, `no`, `on`, `off`, `1`, `0`.
|
|
112
|
+
- `list`: Accepts lists and tuples.
|
|
113
|
+
- `dict`: Accepts dicts.
|
|
114
|
+
- `object`: Accepts anything.
|
|
115
|
+
|
|
116
|
+
## Error Codes
|
|
117
|
+
|
|
118
|
+
- `missing_required`
|
|
119
|
+
- `null_not_allowed`
|
|
120
|
+
- `conversion_failed`
|
|
121
|
+
- `not_in_choices`
|
|
122
|
+
- `too_small`
|
|
123
|
+
- `too_large`
|
|
124
|
+
- `too_short`
|
|
125
|
+
- `too_long`
|
|
126
|
+
- `unexpected_field`
|
|
127
|
+
- `invalid_input`
|
|
128
|
+
|
|
129
|
+
## Examples
|
|
130
|
+
|
|
131
|
+
### Example A — Standardizing APIs
|
|
132
|
+
|
|
133
|
+
Normalize inconsistent payloads into a predictable shape.
|
|
134
|
+
|
|
135
|
+
```python
|
|
136
|
+
from api_response_cleaner import clean
|
|
137
|
+
|
|
138
|
+
schema = {
|
|
139
|
+
"name": {"type": str, "aliases": ["full_name"], "required": True},
|
|
140
|
+
"age": {"type": int, "nullable": True}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
api_a = {"name": "Ravi", "age": 25}
|
|
144
|
+
api_b = {"name": "Ravi", "age": "25"}
|
|
145
|
+
api_c = {"full_name": "Ravi", "age": None}
|
|
146
|
+
|
|
147
|
+
for api, payload in zip(["A", "B", "C"], [api_a, api_b, api_c]):
|
|
148
|
+
res = clean(payload, schema)
|
|
149
|
+
print(f"API {api}: data={res.data}, errors={res.errors}")
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
*Note that API C's `age=None` is handled exactly according to the schema (preserved because `nullable=True`). It is never silently converted to a zero value or a default unless explicitly configured.*
|
|
153
|
+
|
|
154
|
+
### Example B — Legacy field names and defaults
|
|
155
|
+
|
|
156
|
+
Define canonical fields while supporting old aliases and providing sensible defaults.
|
|
157
|
+
|
|
158
|
+
```python
|
|
159
|
+
from api_response_cleaner import clean
|
|
160
|
+
|
|
161
|
+
schema = {
|
|
162
|
+
"theme": {"type": str, "aliases": ["ui_theme", "color_mode"], "default": "light"}
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
# 1. Canonical key works normally
|
|
166
|
+
print(clean({"theme": "dark"}, schema).data)
|
|
167
|
+
# -> {"theme": "dark"}
|
|
168
|
+
|
|
169
|
+
# 2. Alias is seamlessly mapped to the canonical key
|
|
170
|
+
print(clean({"ui_theme": "dark"}, schema).data)
|
|
171
|
+
# -> {"theme": "dark"}
|
|
172
|
+
|
|
173
|
+
# 3. Missing key falls back to the default
|
|
174
|
+
print(clean({}, schema).data)
|
|
175
|
+
# -> {"theme": "light"}
|
|
176
|
+
|
|
177
|
+
# 4. Explicit null ignores the default unless the field is nullable
|
|
178
|
+
res = clean({"theme": None}, schema)
|
|
179
|
+
print(res.errors[0].code)
|
|
180
|
+
# -> ErrorCode.null_not_allowed
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
### Example C — Nested data
|
|
184
|
+
|
|
185
|
+
Safely parse nested user/address data without manual `.get()` chains. Invalid nested objects are omitted entirely instead of returned partially cleaned.
|
|
186
|
+
|
|
187
|
+
```python
|
|
188
|
+
from api_response_cleaner import Cleaner
|
|
189
|
+
|
|
190
|
+
cleaner = Cleaner({
|
|
191
|
+
"address": {
|
|
192
|
+
"schema": {
|
|
193
|
+
"city": {"type": str, "required": True},
|
|
194
|
+
"zip": int
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
})
|
|
198
|
+
|
|
199
|
+
# Valid payload
|
|
200
|
+
res_valid = cleaner.clean({"address": {"city": "NY", "zip": "10001"}})
|
|
201
|
+
print(res_valid.data)
|
|
202
|
+
# -> {"address": {"city": "NY", "zip": 10001}}
|
|
203
|
+
|
|
204
|
+
# Invalid payload
|
|
205
|
+
res_invalid = cleaner.clean({"address": {"zip": "abc"}})
|
|
206
|
+
print("Errors:", [(e.field, e.code.value) for e in res_invalid.errors])
|
|
207
|
+
# -> Errors: [('address.city', 'missing_required'), ('address.zip', 'conversion_failed')]
|
|
208
|
+
print("Data:", res_invalid.data)
|
|
209
|
+
# -> Data: {}
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
### Example D — List validation
|
|
213
|
+
|
|
214
|
+
Validate lists of typed items. Bad elements are reported by their original index, and the whole list is omitted from data if any element is invalid.
|
|
215
|
+
|
|
216
|
+
```python
|
|
217
|
+
from api_response_cleaner import clean
|
|
218
|
+
|
|
219
|
+
schema = {
|
|
220
|
+
"scores": {"type": list, "items": int}
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
res = clean({"scores": ["1", "x", None, "3"]}, schema)
|
|
224
|
+
print("Errors:", [(e.field, e.code.value) for e in res.errors])
|
|
225
|
+
# -> Errors: [('scores[1]', 'conversion_failed'), ('scores[2]', 'null_not_allowed')]
|
|
226
|
+
print("Data:", res.data)
|
|
227
|
+
# -> Data: {}
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
## Development Commands
|
|
231
|
+
|
|
232
|
+
```bash
|
|
233
|
+
pip install -e .[dev]
|
|
234
|
+
pytest
|
|
235
|
+
python -m build
|
|
236
|
+
twine check dist/*
|
|
237
|
+
```
|
|
238
|
+
|
|
239
|
+
## Releasing
|
|
240
|
+
|
|
241
|
+
- [ ] pytest passes on Python 3.11 and the newest installed Python
|
|
242
|
+
- [ ] python -m build produces one sdist and one wheel
|
|
243
|
+
- [ ] twine check dist/* passes
|
|
244
|
+
- [ ] the wheel installs into a clean virtual environment and the quick-start example runs
|
|
245
|
+
- [ ] [project.urls], author and license holder are filled in
|
|
246
|
+
- [ ] the package name is confirmed available on PyPI, then upload to TestPyPI first
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
api_response_cleaner/__init__.py,sha256=H_fbzmliBfgGjk2BMWRuX00vyfnG0PDBF6crCUhxJwo,324
|
|
2
|
+
api_response_cleaner/cleaner.py,sha256=z8GnZKO9lbJDs6j0m967ugjZ4KMU2Bbu8ZtaXbU5zo4,8171
|
|
3
|
+
api_response_cleaner/errors.py,sha256=wrasO5umoUF_pUxX3qkp8RQXbWI4y9AFaq6CXiFBQ80,1184
|
|
4
|
+
api_response_cleaner/normalizer.py,sha256=_WfbetsRXcwcQpgveV5VvHZ0HFPxZzTGvlGRRd7DuKM,2957
|
|
5
|
+
api_response_cleaner/py.typed,sha256=4W8VliAYUP1KY2gLJ_YDy2TmcXYVm-PY7XikQD_bFwA,2
|
|
6
|
+
api_response_cleaner/schema.py,sha256=QeNJdzuBI9Br6P20ScDtaWz0Vcdb3lxp5rx0AVdgM9o,6432
|
|
7
|
+
api_response_cleaner/validator.py,sha256=OepTfdRv-5p513sDzihIDPk2ypi0N_bU7eWOF0aH3ZY,1138
|
|
8
|
+
api_response_cleaner-0.1.0.dist-info/licenses/LICENSE,sha256=QmM4n-igqz_SUin4liyZCmgxdfPdArWHkLErKR_9y1I,1077
|
|
9
|
+
api_response_cleaner-0.1.0.dist-info/METADATA,sha256=flro4mhUHWHnDrQbIiSFsZ39N79X4ENnc6e7XkBKE7I,9055
|
|
10
|
+
api_response_cleaner-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
11
|
+
api_response_cleaner-0.1.0.dist-info/top_level.txt,sha256=18fg4tObVVHg8TYajzpXnoi2H-78OAFtzxmKkp_Gpfw,21
|
|
12
|
+
api_response_cleaner-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Narala Vamsi Krishna
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
api_response_cleaner
|