focus-data-toolkit 0.11.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- focus_data_toolkit/__init__.py +69 -0
- focus_data_toolkit/__main__.py +6 -0
- focus_data_toolkit/_version.py +8 -0
- focus_data_toolkit/cli.py +968 -0
- focus_data_toolkit/context/__init__.py +88 -0
- focus_data_toolkit/context/billing.py +54 -0
- focus_data_toolkit/context/provider.py +90 -0
- focus_data_toolkit/convert/__init__.py +708 -0
- focus_data_toolkit/convert/billing_period.py +65 -0
- focus_data_toolkit/convert/contract_applied.py +235 -0
- focus_data_toolkit/convert/contract_commitment.py +182 -0
- focus_data_toolkit/convert/cost_and_usage.py +179 -0
- focus_data_toolkit/convert/detect.py +39 -0
- focus_data_toolkit/convert/invoice_detail.py +199 -0
- focus_data_toolkit/convert/streaming.py +1030 -0
- focus_data_toolkit/errors.py +145 -0
- focus_data_toolkit/focus_json.py +68 -0
- focus_data_toolkit/generators/__init__.py +61 -0
- focus_data_toolkit/generators/_shim.py +43 -0
- focus_data_toolkit/generators/engine/__init__.py +14 -0
- focus_data_toolkit/generators/engine/context.py +12 -0
- focus_data_toolkit/generators/engine/determinism.py +117 -0
- focus_data_toolkit/generators/engine/json_focus.py +63 -0
- focus_data_toolkit/generators/engine/ladder.py +71 -0
- focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
- focus_data_toolkit/generators/engine/serialize.py +151 -0
- focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
- focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
- focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
- focus_data_toolkit/generators/providers/__init__.py +29 -0
- focus_data_toolkit/generators/providers/aws.py +186 -0
- focus_data_toolkit/generators/providers/azure.py +191 -0
- focus_data_toolkit/generators/providers/gcp.py +194 -0
- focus_data_toolkit/generators/providers/profile.py +123 -0
- focus_data_toolkit/generators/scenarios.py +178 -0
- focus_data_toolkit/generators/versions/__init__.py +17 -0
- focus_data_toolkit/generators/versions/adapter.py +41 -0
- focus_data_toolkit/generators/versions/v1_2.py +111 -0
- focus_data_toolkit/generators/versions/v1_3.py +154 -0
- focus_data_toolkit/io/__init__.py +1 -0
- focus_data_toolkit/io/atomic_writer.py +462 -0
- focus_data_toolkit/io/csv_io.py +128 -0
- focus_data_toolkit/io/parquet_io.py +528 -0
- focus_data_toolkit/io/records.py +92 -0
- focus_data_toolkit/io/row_source.py +117 -0
- focus_data_toolkit/lifecycle.py +342 -0
- focus_data_toolkit/manifest.py +114 -0
- focus_data_toolkit/model/__init__.py +43 -0
- focus_data_toolkit/model/capabilities.py +66 -0
- focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
- focus_data_toolkit/model/focus_1_4_model.json +1913 -0
- focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
- focus_data_toolkit/model/focus_json_keys.py +112 -0
- focus_data_toolkit/model/iso_4217_currencies.json +23 -0
- focus_data_toolkit/model/json_schema_check.py +205 -0
- focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
- focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
- focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
- focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
- focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
- focus_data_toolkit/model/model_provenance.json +58 -0
- focus_data_toolkit/model/validator.py +498 -0
- focus_data_toolkit/modes.py +18 -0
- focus_data_toolkit/official_validator.py +61 -0
- focus_data_toolkit/progress.py +89 -0
- focus_data_toolkit/provenance.py +106 -0
- focus_data_toolkit/py.typed +1 -0
- focus_data_toolkit/runtime.py +243 -0
- focus_data_toolkit/schema/__init__.py +17 -0
- focus_data_toolkit/schema/detection.py +274 -0
- focus_data_toolkit/schema/registry.py +127 -0
- focus_data_toolkit/storage/__init__.py +1 -0
- focus_data_toolkit/storage/external_index.py +99 -0
- focus_data_toolkit/storage/spill.py +150 -0
- focus_data_toolkit/studio/__init__.py +19 -0
- focus_data_toolkit/studio/app.py +467 -0
- focus_data_toolkit/studio/config.py +42 -0
- focus_data_toolkit/studio/frontend/app.js +214 -0
- focus_data_toolkit/studio/frontend/index.html +101 -0
- focus_data_toolkit/studio/frontend/style.css +60 -0
- focus_data_toolkit/studio/jobs.py +142 -0
- focus_data_toolkit/studio/preview.py +32 -0
- focus_data_toolkit/studio/security.py +125 -0
- focus_data_toolkit/studio/server.py +71 -0
- focus_data_toolkit/supplement/__init__.py +50 -0
- focus_data_toolkit/supplement/adapters/__init__.py +21 -0
- focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
- focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
- focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
- focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
- focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
- focus_data_toolkit/supplement/adapters/registry.py +215 -0
- focus_data_toolkit/supplement/apply.py +318 -0
- focus_data_toolkit/supplement/gaps.py +219 -0
- focus_data_toolkit/supplement/kinds.py +118 -0
- focus_data_toolkit/supplement/loader.py +409 -0
- focus_data_toolkit/supplement/spec.py +74 -0
- focus_data_toolkit/supplement/validate.py +215 -0
- focus_data_toolkit/validate/__init__.py +15 -0
- focus_data_toolkit/validate/allocation.py +333 -0
- focus_data_toolkit/validate/bundle.py +254 -0
- focus_data_toolkit/validate/codes.py +93 -0
- focus_data_toolkit/validate/corrections.py +245 -0
- focus_data_toolkit/validate/reconciliation.py +98 -0
- focus_data_toolkit/validate/referential.py +289 -0
- focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
- focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
- focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
- focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
- focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
"""Structured, client-actionable diagnostics.
|
|
2
|
+
|
|
3
|
+
A :class:`Diagnostic` carries everything a user needs to locate and fix a problem in real
|
|
4
|
+
client data: the stable rule code, severity, a human message, the business key of the
|
|
5
|
+
offending record(s), and — where known — the file, dataset, physical line, column, value,
|
|
6
|
+
expected/actual, a fix suggestion, provenance, and the group/join context. It renders to
|
|
7
|
+
both JSON (for machine consumption) and a readable console block, and never degrades to
|
|
8
|
+
``"invalid input"``.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from enum import StrEnum
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class Severity(StrEnum):
|
|
18
|
+
ERROR = "ERROR" # a real conformance/integrity violation
|
|
19
|
+
WARNING = "WARNING" # suspicious, not necessarily invalid
|
|
20
|
+
INFO = "INFO" # informational
|
|
21
|
+
NOT_EXECUTABLE = "NOT_EXECUTABLE" # the check could not run (required data absent)
|
|
22
|
+
NOT_APPLICABLE = "NOT_APPLICABLE" # the check does not apply to this bundle
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
# Severities that represent an actual failure (used to compute a report's ``ok``).
|
|
26
|
+
FAILING_SEVERITIES: frozenset[Severity] = frozenset({Severity.ERROR})
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True)
|
|
30
|
+
class Diagnostic:
|
|
31
|
+
"""One finding, addressable to a specific record / column / rule."""
|
|
32
|
+
|
|
33
|
+
code: str
|
|
34
|
+
severity: Severity
|
|
35
|
+
message: str
|
|
36
|
+
datasets: tuple[str, ...] = ()
|
|
37
|
+
file: str | None = None
|
|
38
|
+
dataset: str | None = None
|
|
39
|
+
line_number: int | None = None
|
|
40
|
+
record_keys: dict[str, str] = field(default_factory=dict)
|
|
41
|
+
column: str | None = None
|
|
42
|
+
value: str | None = None
|
|
43
|
+
expected: str | None = None
|
|
44
|
+
actual: str | None = None
|
|
45
|
+
rule: str | None = None
|
|
46
|
+
suggestion: str | None = None
|
|
47
|
+
provenance: str | None = None
|
|
48
|
+
source: str | None = None
|
|
49
|
+
context: dict[str, str] = field(default_factory=dict)
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def is_failure(self) -> bool:
|
|
53
|
+
return self.severity in FAILING_SEVERITIES
|
|
54
|
+
|
|
55
|
+
def as_dict(self) -> dict:
|
|
56
|
+
"""JSON-serialisable view (omitting empty fields)."""
|
|
57
|
+
out: dict = {
|
|
58
|
+
"rule_id": self.code,
|
|
59
|
+
"severity": self.severity.value,
|
|
60
|
+
"message": self.message,
|
|
61
|
+
}
|
|
62
|
+
if self.datasets:
|
|
63
|
+
out["datasets"] = list(self.datasets)
|
|
64
|
+
if self.dataset:
|
|
65
|
+
out["dataset"] = self.dataset
|
|
66
|
+
if self.file:
|
|
67
|
+
out["file"] = self.file
|
|
68
|
+
if self.line_number is not None:
|
|
69
|
+
out["line_number"] = self.line_number
|
|
70
|
+
if self.record_keys:
|
|
71
|
+
out["record_keys"] = dict(self.record_keys)
|
|
72
|
+
if self.column:
|
|
73
|
+
out["column"] = self.column
|
|
74
|
+
if self.value is not None:
|
|
75
|
+
out["value"] = self.value
|
|
76
|
+
if self.expected is not None:
|
|
77
|
+
out["expected"] = self.expected
|
|
78
|
+
if self.actual is not None:
|
|
79
|
+
out["actual"] = self.actual
|
|
80
|
+
if self.rule:
|
|
81
|
+
out["rule"] = self.rule
|
|
82
|
+
if self.suggestion:
|
|
83
|
+
out["suggestion"] = self.suggestion
|
|
84
|
+
if self.provenance:
|
|
85
|
+
out["provenance"] = self.provenance
|
|
86
|
+
if self.source:
|
|
87
|
+
out["source"] = self.source
|
|
88
|
+
if self.context:
|
|
89
|
+
out["context"] = dict(self.context)
|
|
90
|
+
return out
|
|
91
|
+
|
|
92
|
+
def as_row(self) -> dict[str, str]:
|
|
93
|
+
"""Flat string row for CSV export of large violation sets."""
|
|
94
|
+
return {
|
|
95
|
+
"rule_id": self.code,
|
|
96
|
+
"severity": self.severity.value,
|
|
97
|
+
"datasets": ",".join(self.datasets),
|
|
98
|
+
"dataset": self.dataset or "",
|
|
99
|
+
"file": self.file or "",
|
|
100
|
+
"line_number": "" if self.line_number is None else str(self.line_number),
|
|
101
|
+
"record_keys": ";".join(f"{k}={v}" for k, v in self.record_keys.items()),
|
|
102
|
+
"column": self.column or "",
|
|
103
|
+
"value": "" if self.value is None else self.value,
|
|
104
|
+
"expected": "" if self.expected is None else self.expected,
|
|
105
|
+
"actual": "" if self.actual is None else self.actual,
|
|
106
|
+
"message": self.message,
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
def format(self) -> str:
|
|
110
|
+
"""Readable multi-line console rendering (see module docstring example)."""
|
|
111
|
+
lines = [f"{self.severity.value} {self.code}: {self.message}"]
|
|
112
|
+
loc = []
|
|
113
|
+
if self.file:
|
|
114
|
+
loc.append(f"file={self.file}")
|
|
115
|
+
if self.dataset:
|
|
116
|
+
loc.append(f"dataset={self.dataset}")
|
|
117
|
+
if self.line_number is not None:
|
|
118
|
+
loc.append(f"row {self.line_number}")
|
|
119
|
+
if self.column:
|
|
120
|
+
loc.append(f"column={self.column}")
|
|
121
|
+
if loc:
|
|
122
|
+
lines.append(" " + " ".join(loc))
|
|
123
|
+
for key, val in self.record_keys.items():
|
|
124
|
+
lines.append(f' {key}="{val}"')
|
|
125
|
+
if self.expected is not None or self.actual is not None:
|
|
126
|
+
lines.append(f" expected={self.expected!r} actual={self.actual!r}")
|
|
127
|
+
if self.suggestion:
|
|
128
|
+
lines.append(f" suggestion: {self.suggestion}")
|
|
129
|
+
return "\n".join(lines)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
CSV_FIELDNAMES: tuple[str, ...] = (
|
|
133
|
+
"rule_id",
|
|
134
|
+
"severity",
|
|
135
|
+
"datasets",
|
|
136
|
+
"dataset",
|
|
137
|
+
"file",
|
|
138
|
+
"line_number",
|
|
139
|
+
"record_keys",
|
|
140
|
+
"column",
|
|
141
|
+
"value",
|
|
142
|
+
"expected",
|
|
143
|
+
"actual",
|
|
144
|
+
"message",
|
|
145
|
+
)
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Deterministic JSON serialization for FOCUS JSON-typed columns.
|
|
2
|
+
|
|
3
|
+
FOCUS types several JSON object properties as ``Numeric`` / ``Decimal`` (e.g.
|
|
4
|
+
``AllocatedMethodDetails.AllocatedRatio``, ``ContractApplied.*AppliedCost``).
|
|
5
|
+
Those MUST be serialized as JSON **numbers**, not quoted strings. To keep exact
|
|
6
|
+
decimals (no float rounding) and byte-reproducible output, numeric properties are
|
|
7
|
+
emitted by inserting their exact decimal text as a raw JSON number token rather
|
|
8
|
+
than round-tripping through ``float``.
|
|
9
|
+
|
|
10
|
+
:class:`JsonNumber` tags text that originated from a real JSON number token (use it
|
|
11
|
+
as ``json.loads``'s ``parse_float``/``parse_int`` hook). It lets a parser tell a
|
|
12
|
+
JSON number apart from a quoted numeric string after parsing, and lets ``_encode``
|
|
13
|
+
re-emit such values as raw number literals (preserving custom numeric properties
|
|
14
|
+
across a parse/serialize round-trip).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import json
|
|
20
|
+
import re
|
|
21
|
+
|
|
22
|
+
# JSON number grammar restricted to the FOCUS NumericFormat shape: no leading zeros,
|
|
23
|
+
# E-notation with a negative-only exponent sign, no leading '+'.
|
|
24
|
+
_JSON_NUMBER_RE = re.compile(r"-?(?:0|[1-9]\d*)(?:\.\d+)?(?:E-?\d+)?")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class JsonNumber(str):
|
|
28
|
+
"""Exact text of a value parsed from a JSON **number** token (not a string)."""
|
|
29
|
+
|
|
30
|
+
__slots__ = ()
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def is_json_number_literal(text: str) -> bool:
|
|
34
|
+
"""True if ``text`` is a JSON number literal (FOCUS numeric shape)."""
|
|
35
|
+
return bool(_JSON_NUMBER_RE.fullmatch(text))
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _raw_number(text: str, key: str | None = None) -> str:
|
|
39
|
+
if not is_json_number_literal(text):
|
|
40
|
+
where = f"property {key!r}" if key else "value"
|
|
41
|
+
raise ValueError(f"numeric {where} is not a JSON number: {text!r}")
|
|
42
|
+
return text
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _encode(value: object, numeric_keys: frozenset[str]) -> str:
|
|
46
|
+
if isinstance(value, dict):
|
|
47
|
+
parts = []
|
|
48
|
+
for key, val in value.items():
|
|
49
|
+
if key in numeric_keys and isinstance(val, str):
|
|
50
|
+
parts.append(f"{json.dumps(key)}:{_raw_number(val, key)}")
|
|
51
|
+
else:
|
|
52
|
+
parts.append(f"{json.dumps(key)}:{_encode(val, numeric_keys)}")
|
|
53
|
+
return "{" + ",".join(parts) + "}"
|
|
54
|
+
if isinstance(value, list):
|
|
55
|
+
return "[" + ",".join(_encode(v, numeric_keys) for v in value) + "]"
|
|
56
|
+
if isinstance(value, JsonNumber):
|
|
57
|
+
return _raw_number(value)
|
|
58
|
+
return json.dumps(value)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def dumps_object(obj: dict, *, numeric_keys: frozenset[str] = frozenset()) -> str:
|
|
62
|
+
"""Serialize ``obj`` to compact JSON, emitting ``numeric_keys`` as JSON numbers.
|
|
63
|
+
|
|
64
|
+
``numeric_keys`` applies by key name at any nesting depth, so numeric
|
|
65
|
+
properties inside an ``Elements`` array are emitted unquoted too. Values
|
|
66
|
+
tagged :class:`JsonNumber` are emitted as raw number literals regardless of key.
|
|
67
|
+
"""
|
|
68
|
+
return _encode(obj, numeric_keys)
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Deterministic, provider-realistic FOCUS 1.2/1.3 source generators.
|
|
2
|
+
|
|
3
|
+
Each ``generate_<provider>_focus_<version>`` module is a thin shim binding a provider profile
|
|
4
|
+
to a version adapter (see :mod:`focus_data_toolkit.generators.engine`); a given ``(rows, seed)``
|
|
5
|
+
pair always produces the same CSV. All modules expose ``generate_csv_bytes(rows, seed)``; the
|
|
6
|
+
1.3 modules additionally expose ``generate_contract_commitment_csv_bytes(rows, seed)``.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from importlib import import_module
|
|
12
|
+
from types import ModuleType, SimpleNamespace
|
|
13
|
+
|
|
14
|
+
PROVIDERS: tuple[str, ...] = ("aws", "azure", "gcp")
|
|
15
|
+
FOCUS_VERSIONS: tuple[str, ...] = ("1.2", "1.3")
|
|
16
|
+
|
|
17
|
+
# In-process registry consulted first by get_generator. Lets tests register an extra provider
|
|
18
|
+
# profile (a "fake" cloud) without adding a module file, proving the engine is provider-agnostic.
|
|
19
|
+
_REGISTRY: dict[tuple[str, str], ModuleType | SimpleNamespace] = {}
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def register_generator(provider: str, focus_version: str, api: ModuleType | SimpleNamespace) -> None:
|
|
23
|
+
"""Register an in-process generator (e.g. from ``engine._shim.build_module_api``).
|
|
24
|
+
|
|
25
|
+
``api`` must expose ``generate_csv_bytes`` (and, for a 1.3-style adapter,
|
|
26
|
+
``generate_contract_commitment_csv_bytes``). Test-oriented seam; production code uses the
|
|
27
|
+
six shipped modules via :func:`get_generator`.
|
|
28
|
+
"""
|
|
29
|
+
_REGISTRY[(provider, focus_version)] = api
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def unregister_generator(provider: str, focus_version: str) -> None:
|
|
33
|
+
"""Remove a previously registered in-process generator (no-op if absent)."""
|
|
34
|
+
_REGISTRY.pop((provider, focus_version), None)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def get_generator(provider: str, focus_version: str) -> ModuleType | SimpleNamespace:
|
|
38
|
+
"""Return the generator for ``provider`` and ``focus_version``.
|
|
39
|
+
|
|
40
|
+
Registered in-process generators win over the shipped modules; otherwise the matching
|
|
41
|
+
``generate_<provider>_focus_<version>`` shim module is imported.
|
|
42
|
+
"""
|
|
43
|
+
if (provider, focus_version) in _REGISTRY:
|
|
44
|
+
return _REGISTRY[(provider, focus_version)]
|
|
45
|
+
if provider not in PROVIDERS:
|
|
46
|
+
raise ValueError(f"unknown provider {provider!r}; expected one of {PROVIDERS}")
|
|
47
|
+
if focus_version not in FOCUS_VERSIONS:
|
|
48
|
+
raise ValueError(
|
|
49
|
+
f"unsupported FOCUS version {focus_version!r}; expected one of {FOCUS_VERSIONS}"
|
|
50
|
+
)
|
|
51
|
+
suffix = focus_version.replace(".", "_")
|
|
52
|
+
return import_module(f"focus_data_toolkit.generators.generate_{provider}_focus_{suffix}")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
__all__ = [
|
|
56
|
+
"FOCUS_VERSIONS",
|
|
57
|
+
"PROVIDERS",
|
|
58
|
+
"get_generator",
|
|
59
|
+
"register_generator",
|
|
60
|
+
"unregister_generator",
|
|
61
|
+
]
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Bind a (ProviderProfile, VersionAdapter) pair to the historical per-module public API.
|
|
2
|
+
|
|
3
|
+
Each ``generate_<provider>_focus_<version>.py`` module is a ~10-line shim that calls
|
|
4
|
+
:func:`build_module_api` and publishes the result, so ``COLUMNS`` / ``generate_csv_bytes`` /
|
|
5
|
+
``generate_rows`` / ``main`` / the ``python -m`` entry point keep working unchanged.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from functools import partial
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from focus_data_toolkit.generators.engine import serialize
|
|
14
|
+
from focus_data_toolkit.generators.engine.ladder import generate_rows
|
|
15
|
+
from focus_data_toolkit.generators.providers.profile import ProviderProfile
|
|
16
|
+
from focus_data_toolkit.generators.versions.adapter import VersionAdapter
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def build_module_api(profile: ProviderProfile, adapter: VersionAdapter) -> dict[str, Any]:
|
|
20
|
+
"""Return the names a generator shim module must expose, bound to (profile, adapter).
|
|
21
|
+
|
|
22
|
+
``Any``-valued on purpose: the entries are heterogeneous module attributes (tuples,
|
|
23
|
+
callables, profile objects) that shim modules publish verbatim via ``globals().update``.
|
|
24
|
+
"""
|
|
25
|
+
api: dict[str, Any] = {
|
|
26
|
+
"COLUMNS": adapter.columns,
|
|
27
|
+
"DEFAULT_ROWS": serialize.DEFAULT_ROWS,
|
|
28
|
+
"DEFAULT_SEED": adapter.default_seed,
|
|
29
|
+
"PROFILE": profile,
|
|
30
|
+
"ADAPTER": adapter,
|
|
31
|
+
"generate_rows": partial(generate_rows, profile=profile, adapter=adapter),
|
|
32
|
+
"generate_csv_bytes": partial(serialize.generate_csv_bytes, profile=profile, adapter=adapter),
|
|
33
|
+
"main": partial(serialize.main, profile=profile, adapter=adapter),
|
|
34
|
+
}
|
|
35
|
+
if adapter.contract_commitment_columns is not None:
|
|
36
|
+
api["CONTRACT_COMMITMENT_COLUMNS"] = adapter.contract_commitment_columns
|
|
37
|
+
api["generate_contract_commitment_rows"] = partial(
|
|
38
|
+
serialize.generate_contract_commitment_rows, profile=profile, adapter=adapter
|
|
39
|
+
)
|
|
40
|
+
api["generate_contract_commitment_csv_bytes"] = partial(
|
|
41
|
+
serialize.generate_contract_commitment_csv_bytes, profile=profile, adapter=adapter
|
|
42
|
+
)
|
|
43
|
+
return api
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Provider- and version-agnostic generation engine.
|
|
2
|
+
|
|
3
|
+
The six ``generate_<provider>_focus_<version>`` modules are thin shims that bind a
|
|
4
|
+
:class:`~focus_data_toolkit.generators.providers.profile.ProviderProfile` to a
|
|
5
|
+
:class:`~focus_data_toolkit.generators.versions.adapter.VersionAdapter` and expose the
|
|
6
|
+
public ``generate_csv_bytes`` / ``generate_rows`` / ``main`` API. All shared logic —
|
|
7
|
+
determinism helpers, the FOCUS JSON builders, the row scenarios, the scenario ladder and
|
|
8
|
+
CSV serialization — lives here, defined exactly once.
|
|
9
|
+
|
|
10
|
+
Determinism contract: output is a pure function of the ordered sequence of RNG method
|
|
11
|
+
calls. The engine preserves the historical call order per scenario, and each provider
|
|
12
|
+
callable owns its own draw count, so a given ``(provider, focus_version, rows, seed)``
|
|
13
|
+
reproduces the same bytes as before the refactor.
|
|
14
|
+
"""
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Small value objects threaded through the engine row builders.
|
|
2
|
+
|
|
3
|
+
``ResourceRef`` / ``RowContext`` live in :mod:`focus_data_toolkit.generators.providers.profile`
|
|
4
|
+
next to the :class:`ServiceSpec` they reference, so the import graph stays a one-way street
|
|
5
|
+
(engine -> providers) with no cycle. Re-exported here for the engine-side importers.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from focus_data_toolkit.generators.providers.profile import ResourceRef, RowContext
|
|
11
|
+
|
|
12
|
+
__all__ = ["ResourceRef", "RowContext"]
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""Shared deterministic helpers and constants for the generators.
|
|
2
|
+
|
|
3
|
+
Every value here is identical across all providers and FOCUS versions (verified against
|
|
4
|
+
the six historical generator modules). Rounding rules (``ROUND_HALF_UP`` + the quanta) and
|
|
5
|
+
the billing window live in exactly one place so they can never drift between providers.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import random
|
|
12
|
+
from datetime import UTC, datetime, timedelta
|
|
13
|
+
from decimal import ROUND_HALF_UP, Decimal
|
|
14
|
+
|
|
15
|
+
# --------------------------------------------------------------------------- #
|
|
16
|
+
# Billing window (fixed timestamps -> no clock -> byte-reproducible)
|
|
17
|
+
# --------------------------------------------------------------------------- #
|
|
18
|
+
BILLING_START = datetime(2026, 5, 1, tzinfo=UTC)
|
|
19
|
+
BILLING_END = datetime(2026, 6, 1, tzinfo=UTC)
|
|
20
|
+
PERIOD_DAYS = 28
|
|
21
|
+
PERIOD_HOURS = PERIOD_DAYS * 24
|
|
22
|
+
COMMIT_TERM_DAYS = 365 # 1-year commitment / contract term
|
|
23
|
+
|
|
24
|
+
# --------------------------------------------------------------------------- #
|
|
25
|
+
# Rounding quanta and commercial rates (single source of truth)
|
|
26
|
+
# --------------------------------------------------------------------------- #
|
|
27
|
+
COST_Q = Decimal("0.000001")
|
|
28
|
+
PRICE_Q = Decimal("0.0000000001")
|
|
29
|
+
QTY_Q = Decimal("0.0001")
|
|
30
|
+
EUR_PER_USD = Decimal("0.92")
|
|
31
|
+
COMMIT_RATE = Decimal("0.667") # amortised commitment rate vs on-demand list
|
|
32
|
+
PRIVATE_RATE = Decimal("0.90") # negotiated (contracted) rate vs list for on-demand
|
|
33
|
+
COMMIT_TERM_HOURS = Decimal("8760") # 1-year reservation term
|
|
34
|
+
|
|
35
|
+
# --------------------------------------------------------------------------- #
|
|
36
|
+
# Shared value pools and FOCUS vocabularies
|
|
37
|
+
# --------------------------------------------------------------------------- #
|
|
38
|
+
ENVIRONMENTS = ("prod", "staging", "dev")
|
|
39
|
+
COST_CENTERS = ("cc-1042", "cc-2087", "cc-3110")
|
|
40
|
+
OWNERS = ("team-platform", "team-data", "team-payments")
|
|
41
|
+
|
|
42
|
+
PRICING_CATEGORIES: tuple[str, ...] = ("Standard", "Dynamic", "Committed", "Other")
|
|
43
|
+
# FOCUS-defined SkuPriceDetails property keys (others MUST be x_-prefixed).
|
|
44
|
+
FOCUS_SKU_PRICE_KEYS: frozenset[str] = frozenset(
|
|
45
|
+
{
|
|
46
|
+
"CoreCount",
|
|
47
|
+
"MemorySize",
|
|
48
|
+
"InstanceType",
|
|
49
|
+
"InstanceSeries",
|
|
50
|
+
"OperatingSystem",
|
|
51
|
+
"DiskType",
|
|
52
|
+
"DiskSpace",
|
|
53
|
+
"DiskMaxIops",
|
|
54
|
+
"GpuCount",
|
|
55
|
+
"NetworkMaxIops",
|
|
56
|
+
"NetworkMaxThroughput",
|
|
57
|
+
}
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
HEX_LOWER = "0123456789abcdef"
|
|
61
|
+
HEX_UPPER = "0123456789ABCDEF"
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def q(value: Decimal, quant: Decimal) -> Decimal:
|
|
65
|
+
"""Quantise ``value`` to ``quant`` using banker-free ROUND_HALF_UP (FOCUS rounding)."""
|
|
66
|
+
return value.quantize(quant, rounding=ROUND_HALF_UP)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def s(value: Decimal) -> str:
|
|
70
|
+
"""Render a Decimal as fixed-point text (never scientific notation)."""
|
|
71
|
+
return format(value, "f")
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def iso(dt: datetime) -> str:
|
|
75
|
+
return dt.strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def parse_iso(value: str) -> datetime:
|
|
79
|
+
return datetime.strptime(value, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=UTC)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def hexid(rng: random.Random, width: int, alphabet: str = HEX_LOWER) -> str:
|
|
83
|
+
"""Draw ``width`` characters from ``alphabet`` (default lowercase hex)."""
|
|
84
|
+
return "".join(rng.choice(alphabet) for _ in range(width))
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def period(i: int, granularity: str) -> tuple[str, str]:
|
|
88
|
+
if granularity == "hourly":
|
|
89
|
+
start = BILLING_START + timedelta(hours=i % PERIOD_HOURS)
|
|
90
|
+
return iso(start), iso(start + timedelta(hours=1))
|
|
91
|
+
if granularity == "daily":
|
|
92
|
+
start = BILLING_START + timedelta(days=i % PERIOD_DAYS)
|
|
93
|
+
return iso(start), iso(start + timedelta(days=1))
|
|
94
|
+
return iso(BILLING_START), iso(BILLING_END)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def sku_price_details(spec_sku_details: dict[str, object]) -> str:
|
|
98
|
+
return json.dumps(spec_sku_details, separators=(",", ":"))
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def contract_id_for(commit_id: str) -> str:
|
|
102
|
+
"""Deterministic parent ContractId for a commitment id (shared by both 1.3 datasets)."""
|
|
103
|
+
return f"CONTRACT-{commit_id.rsplit('/', 1)[-1][:12]}"
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def set_currency(
|
|
107
|
+
row: dict[str, str],
|
|
108
|
+
pricing_currency: str,
|
|
109
|
+
list_unit: Decimal,
|
|
110
|
+
contracted_unit: Decimal,
|
|
111
|
+
effective_cost: Decimal,
|
|
112
|
+
) -> None:
|
|
113
|
+
row["PricingCurrency"] = pricing_currency
|
|
114
|
+
fx = EUR_PER_USD if pricing_currency == "EUR" else Decimal("1")
|
|
115
|
+
row["PricingCurrencyListUnitPrice"] = s(q(list_unit * fx, PRICE_Q))
|
|
116
|
+
row["PricingCurrencyContractedUnitPrice"] = s(q(contracted_unit * fx, PRICE_Q))
|
|
117
|
+
row["PricingCurrencyEffectiveCost"] = s(q(effective_cost * fx, COST_Q))
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""Single-source FOCUS JSON builders for the generators.
|
|
2
|
+
|
|
3
|
+
Both the Split Cost Allocation ``AllocatedMethodDetails`` object and the
|
|
4
|
+
``ContractApplied`` object are FOCUS JSON with *Numeric* properties that MUST be emitted
|
|
5
|
+
as JSON numbers (not quoted strings). They are built here, once, on top of
|
|
6
|
+
:func:`focus_data_toolkit.focus_json.dumps_object`, so no generator, provider or scenario
|
|
7
|
+
re-implements the rule. Previously the SCA object was built two divergent ways (the 1.3
|
|
8
|
+
generators via ``dumps_object``; ``scenarios.py`` by string concatenation) — both now route
|
|
9
|
+
through :func:`allocated_method_details`.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from collections.abc import Mapping, Sequence
|
|
15
|
+
|
|
16
|
+
from focus_data_toolkit.focus_json import dumps_object
|
|
17
|
+
|
|
18
|
+
# AllocatedMethodDetails.Elements[*] numeric properties (FOCUS 1.3 Split Cost Allocation).
|
|
19
|
+
SCA_NUMERIC_KEYS: frozenset[str] = frozenset({"AllocatedRatio", "UsageQuantity"})
|
|
20
|
+
# ContractApplied.Elements[*] numeric properties (FOCUS 1.3).
|
|
21
|
+
CONTRACT_APPLIED_NUMERIC_KEYS: frozenset[str] = frozenset(
|
|
22
|
+
{"ContractCommitmentAppliedCost", "ContractCommitmentAppliedQuantity"}
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def allocated_method_details(
|
|
27
|
+
elements: Sequence[Mapping[str, object]],
|
|
28
|
+
*,
|
|
29
|
+
numeric_keys: frozenset[str] = SCA_NUMERIC_KEYS,
|
|
30
|
+
) -> str:
|
|
31
|
+
"""Serialise a FOCUS ``AllocatedMethodDetails`` object from its ``Elements``.
|
|
32
|
+
|
|
33
|
+
Each element's numeric properties (``AllocatedRatio`` / ``UsageQuantity`` by default)
|
|
34
|
+
are emitted as JSON numbers; their string values must be JSON number literals or
|
|
35
|
+
``dumps_object`` raises (a stricter, safer contract than string concatenation).
|
|
36
|
+
"""
|
|
37
|
+
return dumps_object({"Elements": list(elements)}, numeric_keys=numeric_keys)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def contract_applied(
|
|
41
|
+
commit_id: str,
|
|
42
|
+
contract_id: str,
|
|
43
|
+
applied_cost: str,
|
|
44
|
+
applied_qty: str,
|
|
45
|
+
applied_unit: str,
|
|
46
|
+
) -> str:
|
|
47
|
+
"""FOCUS 1.3 ``ContractApplied`` JSON: a top-level ``Elements`` array linking a Cost
|
|
48
|
+
and Usage row to the Contract Commitment dataset via ``ContractCommitmentID``. The
|
|
49
|
+
applied cost/quantity are emitted as JSON numbers."""
|
|
50
|
+
return dumps_object(
|
|
51
|
+
{
|
|
52
|
+
"Elements": [
|
|
53
|
+
{
|
|
54
|
+
"ContractID": contract_id,
|
|
55
|
+
"ContractCommitmentID": commit_id,
|
|
56
|
+
"ContractCommitmentAppliedCost": applied_cost,
|
|
57
|
+
"ContractCommitmentAppliedQuantity": applied_qty,
|
|
58
|
+
"ContractCommitmentAppliedUnit": applied_unit,
|
|
59
|
+
}
|
|
60
|
+
]
|
|
61
|
+
},
|
|
62
|
+
numeric_keys=CONTRACT_APPLIED_NUMERIC_KEYS,
|
|
63
|
+
)
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""The scenario-selection loop shared by every generator.
|
|
2
|
+
|
|
3
|
+
Draws exactly one ``rng.random()`` per output position (as the historical generators did) and
|
|
4
|
+
dispatches to a scenario builder via the version adapter's ladder, preserving the original
|
|
5
|
+
``if/elif`` semantics precisely.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import random
|
|
11
|
+
from collections.abc import Callable
|
|
12
|
+
|
|
13
|
+
from focus_data_toolkit.generators.engine import scenarios_core
|
|
14
|
+
|
|
15
|
+
DEFAULT_ROWS = 1000
|
|
16
|
+
|
|
17
|
+
# A builder returns either one row or a whole group of rows; the branch's ``group`` flag
|
|
18
|
+
# (mirrored in the isinstance checks below) says which, so the union is narrowed per call.
|
|
19
|
+
_Builder = Callable[..., "dict[str, str] | list[dict[str, str]]"]
|
|
20
|
+
|
|
21
|
+
_BUILDERS: dict[str, _Builder] = {
|
|
22
|
+
"credit": scenarios_core.credit_row,
|
|
23
|
+
"tax": scenarios_core.tax_row,
|
|
24
|
+
"purchase": scenarios_core.standalone_purchase_row,
|
|
25
|
+
"split": scenarios_core.split_allocation_row,
|
|
26
|
+
"commitment": scenarios_core.commitment_group,
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def generate_rows(
|
|
31
|
+
rows: int = DEFAULT_ROWS,
|
|
32
|
+
seed: int | None = None,
|
|
33
|
+
*,
|
|
34
|
+
include_credits: bool = False,
|
|
35
|
+
profile,
|
|
36
|
+
adapter,
|
|
37
|
+
) -> list[dict[str, str]]:
|
|
38
|
+
"""Return ``rows`` synthetic records for ``profile``/``adapter`` as ordered string dicts.
|
|
39
|
+
|
|
40
|
+
``rows``/``seed`` default to the historical per-module values (1000 rows; the adapter's
|
|
41
|
+
default seed) so the shim ``generate_rows()`` / ``generate_rows(rows=N)`` calls keep working.
|
|
42
|
+
"""
|
|
43
|
+
if seed is None:
|
|
44
|
+
seed = adapter.default_seed
|
|
45
|
+
if rows < 1:
|
|
46
|
+
raise ValueError("rows must be >= 1")
|
|
47
|
+
rng = random.Random(seed)
|
|
48
|
+
out: list[dict[str, str]] = []
|
|
49
|
+
while len(out) < rows:
|
|
50
|
+
i = len(out)
|
|
51
|
+
remaining = rows - i
|
|
52
|
+
roll = rng.random()
|
|
53
|
+
chosen = None
|
|
54
|
+
for branch in adapter.ladder:
|
|
55
|
+
if branch.requires_credits and not include_credits:
|
|
56
|
+
continue
|
|
57
|
+
if roll < branch.threshold:
|
|
58
|
+
if branch.min_remaining is None or remaining >= branch.min_remaining:
|
|
59
|
+
chosen = branch
|
|
60
|
+
break # first threshold match wins (elif semantics); guard failure -> Usage
|
|
61
|
+
if chosen is None:
|
|
62
|
+
out.append(scenarios_core.usage_row(rng, i, remaining, profile, adapter))
|
|
63
|
+
continue
|
|
64
|
+
built = _BUILDERS[chosen.kind](rng, i, remaining, profile, adapter)
|
|
65
|
+
if chosen.group:
|
|
66
|
+
assert isinstance(built, list), chosen.kind
|
|
67
|
+
out.extend(built)
|
|
68
|
+
else:
|
|
69
|
+
assert isinstance(built, dict), chosen.kind
|
|
70
|
+
out.append(built)
|
|
71
|
+
return out[:rows]
|