focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,145 @@
1
+ """Structured, client-actionable diagnostics.
2
+
3
+ A :class:`Diagnostic` carries everything a user needs to locate and fix a problem in real
4
+ client data: the stable rule code, severity, a human message, the business key of the
5
+ offending record(s), and — where known — the file, dataset, physical line, column, value,
6
+ expected/actual, a fix suggestion, provenance, and the group/join context. It renders to
7
+ both JSON (for machine consumption) and a readable console block, and never degrades to
8
+ ``"invalid input"``.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from dataclasses import dataclass, field
14
+ from enum import StrEnum
15
+
16
+
17
+ class Severity(StrEnum):
18
+ ERROR = "ERROR" # a real conformance/integrity violation
19
+ WARNING = "WARNING" # suspicious, not necessarily invalid
20
+ INFO = "INFO" # informational
21
+ NOT_EXECUTABLE = "NOT_EXECUTABLE" # the check could not run (required data absent)
22
+ NOT_APPLICABLE = "NOT_APPLICABLE" # the check does not apply to this bundle
23
+
24
+
25
+ # Severities that represent an actual failure (used to compute a report's ``ok``).
26
+ FAILING_SEVERITIES: frozenset[Severity] = frozenset({Severity.ERROR})
27
+
28
+
29
+ @dataclass(frozen=True)
30
+ class Diagnostic:
31
+ """One finding, addressable to a specific record / column / rule."""
32
+
33
+ code: str
34
+ severity: Severity
35
+ message: str
36
+ datasets: tuple[str, ...] = ()
37
+ file: str | None = None
38
+ dataset: str | None = None
39
+ line_number: int | None = None
40
+ record_keys: dict[str, str] = field(default_factory=dict)
41
+ column: str | None = None
42
+ value: str | None = None
43
+ expected: str | None = None
44
+ actual: str | None = None
45
+ rule: str | None = None
46
+ suggestion: str | None = None
47
+ provenance: str | None = None
48
+ source: str | None = None
49
+ context: dict[str, str] = field(default_factory=dict)
50
+
51
+ @property
52
+ def is_failure(self) -> bool:
53
+ return self.severity in FAILING_SEVERITIES
54
+
55
+ def as_dict(self) -> dict:
56
+ """JSON-serialisable view (omitting empty fields)."""
57
+ out: dict = {
58
+ "rule_id": self.code,
59
+ "severity": self.severity.value,
60
+ "message": self.message,
61
+ }
62
+ if self.datasets:
63
+ out["datasets"] = list(self.datasets)
64
+ if self.dataset:
65
+ out["dataset"] = self.dataset
66
+ if self.file:
67
+ out["file"] = self.file
68
+ if self.line_number is not None:
69
+ out["line_number"] = self.line_number
70
+ if self.record_keys:
71
+ out["record_keys"] = dict(self.record_keys)
72
+ if self.column:
73
+ out["column"] = self.column
74
+ if self.value is not None:
75
+ out["value"] = self.value
76
+ if self.expected is not None:
77
+ out["expected"] = self.expected
78
+ if self.actual is not None:
79
+ out["actual"] = self.actual
80
+ if self.rule:
81
+ out["rule"] = self.rule
82
+ if self.suggestion:
83
+ out["suggestion"] = self.suggestion
84
+ if self.provenance:
85
+ out["provenance"] = self.provenance
86
+ if self.source:
87
+ out["source"] = self.source
88
+ if self.context:
89
+ out["context"] = dict(self.context)
90
+ return out
91
+
92
+ def as_row(self) -> dict[str, str]:
93
+ """Flat string row for CSV export of large violation sets."""
94
+ return {
95
+ "rule_id": self.code,
96
+ "severity": self.severity.value,
97
+ "datasets": ",".join(self.datasets),
98
+ "dataset": self.dataset or "",
99
+ "file": self.file or "",
100
+ "line_number": "" if self.line_number is None else str(self.line_number),
101
+ "record_keys": ";".join(f"{k}={v}" for k, v in self.record_keys.items()),
102
+ "column": self.column or "",
103
+ "value": "" if self.value is None else self.value,
104
+ "expected": "" if self.expected is None else self.expected,
105
+ "actual": "" if self.actual is None else self.actual,
106
+ "message": self.message,
107
+ }
108
+
109
+ def format(self) -> str:
110
+ """Readable multi-line console rendering (see module docstring example)."""
111
+ lines = [f"{self.severity.value} {self.code}: {self.message}"]
112
+ loc = []
113
+ if self.file:
114
+ loc.append(f"file={self.file}")
115
+ if self.dataset:
116
+ loc.append(f"dataset={self.dataset}")
117
+ if self.line_number is not None:
118
+ loc.append(f"row {self.line_number}")
119
+ if self.column:
120
+ loc.append(f"column={self.column}")
121
+ if loc:
122
+ lines.append(" " + " ".join(loc))
123
+ for key, val in self.record_keys.items():
124
+ lines.append(f' {key}="{val}"')
125
+ if self.expected is not None or self.actual is not None:
126
+ lines.append(f" expected={self.expected!r} actual={self.actual!r}")
127
+ if self.suggestion:
128
+ lines.append(f" suggestion: {self.suggestion}")
129
+ return "\n".join(lines)
130
+
131
+
132
+ CSV_FIELDNAMES: tuple[str, ...] = (
133
+ "rule_id",
134
+ "severity",
135
+ "datasets",
136
+ "dataset",
137
+ "file",
138
+ "line_number",
139
+ "record_keys",
140
+ "column",
141
+ "value",
142
+ "expected",
143
+ "actual",
144
+ "message",
145
+ )
@@ -0,0 +1,68 @@
1
+ """Deterministic JSON serialization for FOCUS JSON-typed columns.
2
+
3
+ FOCUS types several JSON object properties as ``Numeric`` / ``Decimal`` (e.g.
4
+ ``AllocatedMethodDetails.AllocatedRatio``, ``ContractApplied.*AppliedCost``).
5
+ Those MUST be serialized as JSON **numbers**, not quoted strings. To keep exact
6
+ decimals (no float rounding) and byte-reproducible output, numeric properties are
7
+ emitted by inserting their exact decimal text as a raw JSON number token rather
8
+ than round-tripping through ``float``.
9
+
10
+ :class:`JsonNumber` tags text that originated from a real JSON number token (use it
11
+ as ``json.loads``'s ``parse_float``/``parse_int`` hook). It lets a parser tell a
12
+ JSON number apart from a quoted numeric string after parsing, and lets ``_encode``
13
+ re-emit such values as raw number literals (preserving custom numeric properties
14
+ across a parse/serialize round-trip).
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import json
20
+ import re
21
+
22
+ # JSON number grammar restricted to the FOCUS NumericFormat shape: no leading zeros,
23
+ # E-notation with a negative-only exponent sign, no leading '+'.
24
+ _JSON_NUMBER_RE = re.compile(r"-?(?:0|[1-9]\d*)(?:\.\d+)?(?:E-?\d+)?")
25
+
26
+
27
+ class JsonNumber(str):
28
+ """Exact text of a value parsed from a JSON **number** token (not a string)."""
29
+
30
+ __slots__ = ()
31
+
32
+
33
+ def is_json_number_literal(text: str) -> bool:
34
+ """True if ``text`` is a JSON number literal (FOCUS numeric shape)."""
35
+ return bool(_JSON_NUMBER_RE.fullmatch(text))
36
+
37
+
38
+ def _raw_number(text: str, key: str | None = None) -> str:
39
+ if not is_json_number_literal(text):
40
+ where = f"property {key!r}" if key else "value"
41
+ raise ValueError(f"numeric {where} is not a JSON number: {text!r}")
42
+ return text
43
+
44
+
45
+ def _encode(value: object, numeric_keys: frozenset[str]) -> str:
46
+ if isinstance(value, dict):
47
+ parts = []
48
+ for key, val in value.items():
49
+ if key in numeric_keys and isinstance(val, str):
50
+ parts.append(f"{json.dumps(key)}:{_raw_number(val, key)}")
51
+ else:
52
+ parts.append(f"{json.dumps(key)}:{_encode(val, numeric_keys)}")
53
+ return "{" + ",".join(parts) + "}"
54
+ if isinstance(value, list):
55
+ return "[" + ",".join(_encode(v, numeric_keys) for v in value) + "]"
56
+ if isinstance(value, JsonNumber):
57
+ return _raw_number(value)
58
+ return json.dumps(value)
59
+
60
+
61
+ def dumps_object(obj: dict, *, numeric_keys: frozenset[str] = frozenset()) -> str:
62
+ """Serialize ``obj`` to compact JSON, emitting ``numeric_keys`` as JSON numbers.
63
+
64
+ ``numeric_keys`` applies by key name at any nesting depth, so numeric
65
+ properties inside an ``Elements`` array are emitted unquoted too. Values
66
+ tagged :class:`JsonNumber` are emitted as raw number literals regardless of key.
67
+ """
68
+ return _encode(obj, numeric_keys)
@@ -0,0 +1,61 @@
1
+ """Deterministic, provider-realistic FOCUS 1.2/1.3 source generators.
2
+
3
+ Each ``generate_<provider>_focus_<version>`` module is a thin shim binding a provider profile
4
+ to a version adapter (see :mod:`focus_data_toolkit.generators.engine`); a given ``(rows, seed)``
5
+ pair always produces the same CSV. All modules expose ``generate_csv_bytes(rows, seed)``; the
6
+ 1.3 modules additionally expose ``generate_contract_commitment_csv_bytes(rows, seed)``.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from importlib import import_module
12
+ from types import ModuleType, SimpleNamespace
13
+
14
+ PROVIDERS: tuple[str, ...] = ("aws", "azure", "gcp")
15
+ FOCUS_VERSIONS: tuple[str, ...] = ("1.2", "1.3")
16
+
17
+ # In-process registry consulted first by get_generator. Lets tests register an extra provider
18
+ # profile (a "fake" cloud) without adding a module file, proving the engine is provider-agnostic.
19
+ _REGISTRY: dict[tuple[str, str], ModuleType | SimpleNamespace] = {}
20
+
21
+
22
+ def register_generator(provider: str, focus_version: str, api: ModuleType | SimpleNamespace) -> None:
23
+ """Register an in-process generator (e.g. from ``engine._shim.build_module_api``).
24
+
25
+ ``api`` must expose ``generate_csv_bytes`` (and, for a 1.3-style adapter,
26
+ ``generate_contract_commitment_csv_bytes``). Test-oriented seam; production code uses the
27
+ six shipped modules via :func:`get_generator`.
28
+ """
29
+ _REGISTRY[(provider, focus_version)] = api
30
+
31
+
32
+ def unregister_generator(provider: str, focus_version: str) -> None:
33
+ """Remove a previously registered in-process generator (no-op if absent)."""
34
+ _REGISTRY.pop((provider, focus_version), None)
35
+
36
+
37
+ def get_generator(provider: str, focus_version: str) -> ModuleType | SimpleNamespace:
38
+ """Return the generator for ``provider`` and ``focus_version``.
39
+
40
+ Registered in-process generators win over the shipped modules; otherwise the matching
41
+ ``generate_<provider>_focus_<version>`` shim module is imported.
42
+ """
43
+ if (provider, focus_version) in _REGISTRY:
44
+ return _REGISTRY[(provider, focus_version)]
45
+ if provider not in PROVIDERS:
46
+ raise ValueError(f"unknown provider {provider!r}; expected one of {PROVIDERS}")
47
+ if focus_version not in FOCUS_VERSIONS:
48
+ raise ValueError(
49
+ f"unsupported FOCUS version {focus_version!r}; expected one of {FOCUS_VERSIONS}"
50
+ )
51
+ suffix = focus_version.replace(".", "_")
52
+ return import_module(f"focus_data_toolkit.generators.generate_{provider}_focus_{suffix}")
53
+
54
+
55
+ __all__ = [
56
+ "FOCUS_VERSIONS",
57
+ "PROVIDERS",
58
+ "get_generator",
59
+ "register_generator",
60
+ "unregister_generator",
61
+ ]
@@ -0,0 +1,43 @@
1
+ """Bind a (ProviderProfile, VersionAdapter) pair to the historical per-module public API.
2
+
3
+ Each ``generate_<provider>_focus_<version>.py`` module is a ~10-line shim that calls
4
+ :func:`build_module_api` and publishes the result, so ``COLUMNS`` / ``generate_csv_bytes`` /
5
+ ``generate_rows`` / ``main`` / the ``python -m`` entry point keep working unchanged.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from functools import partial
11
+ from typing import Any
12
+
13
+ from focus_data_toolkit.generators.engine import serialize
14
+ from focus_data_toolkit.generators.engine.ladder import generate_rows
15
+ from focus_data_toolkit.generators.providers.profile import ProviderProfile
16
+ from focus_data_toolkit.generators.versions.adapter import VersionAdapter
17
+
18
+
19
+ def build_module_api(profile: ProviderProfile, adapter: VersionAdapter) -> dict[str, Any]:
20
+ """Return the names a generator shim module must expose, bound to (profile, adapter).
21
+
22
+ ``Any``-valued on purpose: the entries are heterogeneous module attributes (tuples,
23
+ callables, profile objects) that shim modules publish verbatim via ``globals().update``.
24
+ """
25
+ api: dict[str, Any] = {
26
+ "COLUMNS": adapter.columns,
27
+ "DEFAULT_ROWS": serialize.DEFAULT_ROWS,
28
+ "DEFAULT_SEED": adapter.default_seed,
29
+ "PROFILE": profile,
30
+ "ADAPTER": adapter,
31
+ "generate_rows": partial(generate_rows, profile=profile, adapter=adapter),
32
+ "generate_csv_bytes": partial(serialize.generate_csv_bytes, profile=profile, adapter=adapter),
33
+ "main": partial(serialize.main, profile=profile, adapter=adapter),
34
+ }
35
+ if adapter.contract_commitment_columns is not None:
36
+ api["CONTRACT_COMMITMENT_COLUMNS"] = adapter.contract_commitment_columns
37
+ api["generate_contract_commitment_rows"] = partial(
38
+ serialize.generate_contract_commitment_rows, profile=profile, adapter=adapter
39
+ )
40
+ api["generate_contract_commitment_csv_bytes"] = partial(
41
+ serialize.generate_contract_commitment_csv_bytes, profile=profile, adapter=adapter
42
+ )
43
+ return api
@@ -0,0 +1,14 @@
1
+ """Provider- and version-agnostic generation engine.
2
+
3
+ The six ``generate_<provider>_focus_<version>`` modules are thin shims that bind a
4
+ :class:`~focus_data_toolkit.generators.providers.profile.ProviderProfile` to a
5
+ :class:`~focus_data_toolkit.generators.versions.adapter.VersionAdapter` and expose the
6
+ public ``generate_csv_bytes`` / ``generate_rows`` / ``main`` API. All shared logic —
7
+ determinism helpers, the FOCUS JSON builders, the row scenarios, the scenario ladder and
8
+ CSV serialization — lives here, defined exactly once.
9
+
10
+ Determinism contract: output is a pure function of the ordered sequence of RNG method
11
+ calls. The engine preserves the historical call order per scenario, and each provider
12
+ callable owns its own draw count, so a given ``(provider, focus_version, rows, seed)``
13
+ reproduces the same bytes as before the refactor.
14
+ """
@@ -0,0 +1,12 @@
1
+ """Small value objects threaded through the engine row builders.
2
+
3
+ ``ResourceRef`` / ``RowContext`` live in :mod:`focus_data_toolkit.generators.providers.profile`
4
+ next to the :class:`ServiceSpec` they reference, so the import graph stays a one-way street
5
+ (engine -> providers) with no cycle. Re-exported here for the engine-side importers.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from focus_data_toolkit.generators.providers.profile import ResourceRef, RowContext
11
+
12
+ __all__ = ["ResourceRef", "RowContext"]
@@ -0,0 +1,117 @@
1
+ """Shared deterministic helpers and constants for the generators.
2
+
3
+ Every value here is identical across all providers and FOCUS versions (verified against
4
+ the six historical generator modules). Rounding rules (``ROUND_HALF_UP`` + the quanta) and
5
+ the billing window live in exactly one place so they can never drift between providers.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import random
12
+ from datetime import UTC, datetime, timedelta
13
+ from decimal import ROUND_HALF_UP, Decimal
14
+
15
+ # --------------------------------------------------------------------------- #
16
+ # Billing window (fixed timestamps -> no clock -> byte-reproducible)
17
+ # --------------------------------------------------------------------------- #
18
+ BILLING_START = datetime(2026, 5, 1, tzinfo=UTC)
19
+ BILLING_END = datetime(2026, 6, 1, tzinfo=UTC)
20
+ PERIOD_DAYS = 28
21
+ PERIOD_HOURS = PERIOD_DAYS * 24
22
+ COMMIT_TERM_DAYS = 365 # 1-year commitment / contract term
23
+
24
+ # --------------------------------------------------------------------------- #
25
+ # Rounding quanta and commercial rates (single source of truth)
26
+ # --------------------------------------------------------------------------- #
27
+ COST_Q = Decimal("0.000001")
28
+ PRICE_Q = Decimal("0.0000000001")
29
+ QTY_Q = Decimal("0.0001")
30
+ EUR_PER_USD = Decimal("0.92")
31
+ COMMIT_RATE = Decimal("0.667") # amortised commitment rate vs on-demand list
32
+ PRIVATE_RATE = Decimal("0.90") # negotiated (contracted) rate vs list for on-demand
33
+ COMMIT_TERM_HOURS = Decimal("8760") # 1-year reservation term
34
+
35
+ # --------------------------------------------------------------------------- #
36
+ # Shared value pools and FOCUS vocabularies
37
+ # --------------------------------------------------------------------------- #
38
+ ENVIRONMENTS = ("prod", "staging", "dev")
39
+ COST_CENTERS = ("cc-1042", "cc-2087", "cc-3110")
40
+ OWNERS = ("team-platform", "team-data", "team-payments")
41
+
42
+ PRICING_CATEGORIES: tuple[str, ...] = ("Standard", "Dynamic", "Committed", "Other")
43
+ # FOCUS-defined SkuPriceDetails property keys (others MUST be x_-prefixed).
44
+ FOCUS_SKU_PRICE_KEYS: frozenset[str] = frozenset(
45
+ {
46
+ "CoreCount",
47
+ "MemorySize",
48
+ "InstanceType",
49
+ "InstanceSeries",
50
+ "OperatingSystem",
51
+ "DiskType",
52
+ "DiskSpace",
53
+ "DiskMaxIops",
54
+ "GpuCount",
55
+ "NetworkMaxIops",
56
+ "NetworkMaxThroughput",
57
+ }
58
+ )
59
+
60
+ HEX_LOWER = "0123456789abcdef"
61
+ HEX_UPPER = "0123456789ABCDEF"
62
+
63
+
64
+ def q(value: Decimal, quant: Decimal) -> Decimal:
65
+ """Quantise ``value`` to ``quant`` using banker-free ROUND_HALF_UP (FOCUS rounding)."""
66
+ return value.quantize(quant, rounding=ROUND_HALF_UP)
67
+
68
+
69
+ def s(value: Decimal) -> str:
70
+ """Render a Decimal as fixed-point text (never scientific notation)."""
71
+ return format(value, "f")
72
+
73
+
74
+ def iso(dt: datetime) -> str:
75
+ return dt.strftime("%Y-%m-%dT%H:%M:%SZ")
76
+
77
+
78
+ def parse_iso(value: str) -> datetime:
79
+ return datetime.strptime(value, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=UTC)
80
+
81
+
82
+ def hexid(rng: random.Random, width: int, alphabet: str = HEX_LOWER) -> str:
83
+ """Draw ``width`` characters from ``alphabet`` (default lowercase hex)."""
84
+ return "".join(rng.choice(alphabet) for _ in range(width))
85
+
86
+
87
+ def period(i: int, granularity: str) -> tuple[str, str]:
88
+ if granularity == "hourly":
89
+ start = BILLING_START + timedelta(hours=i % PERIOD_HOURS)
90
+ return iso(start), iso(start + timedelta(hours=1))
91
+ if granularity == "daily":
92
+ start = BILLING_START + timedelta(days=i % PERIOD_DAYS)
93
+ return iso(start), iso(start + timedelta(days=1))
94
+ return iso(BILLING_START), iso(BILLING_END)
95
+
96
+
97
+ def sku_price_details(spec_sku_details: dict[str, object]) -> str:
98
+ return json.dumps(spec_sku_details, separators=(",", ":"))
99
+
100
+
101
+ def contract_id_for(commit_id: str) -> str:
102
+ """Deterministic parent ContractId for a commitment id (shared by both 1.3 datasets)."""
103
+ return f"CONTRACT-{commit_id.rsplit('/', 1)[-1][:12]}"
104
+
105
+
106
+ def set_currency(
107
+ row: dict[str, str],
108
+ pricing_currency: str,
109
+ list_unit: Decimal,
110
+ contracted_unit: Decimal,
111
+ effective_cost: Decimal,
112
+ ) -> None:
113
+ row["PricingCurrency"] = pricing_currency
114
+ fx = EUR_PER_USD if pricing_currency == "EUR" else Decimal("1")
115
+ row["PricingCurrencyListUnitPrice"] = s(q(list_unit * fx, PRICE_Q))
116
+ row["PricingCurrencyContractedUnitPrice"] = s(q(contracted_unit * fx, PRICE_Q))
117
+ row["PricingCurrencyEffectiveCost"] = s(q(effective_cost * fx, COST_Q))
@@ -0,0 +1,63 @@
1
+ """Single-source FOCUS JSON builders for the generators.
2
+
3
+ Both the Split Cost Allocation ``AllocatedMethodDetails`` object and the
4
+ ``ContractApplied`` object are FOCUS JSON with *Numeric* properties that MUST be emitted
5
+ as JSON numbers (not quoted strings). They are built here, once, on top of
6
+ :func:`focus_data_toolkit.focus_json.dumps_object`, so no generator, provider or scenario
7
+ re-implements the rule. Previously the SCA object was built two divergent ways (the 1.3
8
+ generators via ``dumps_object``; ``scenarios.py`` by string concatenation) — both now route
9
+ through :func:`allocated_method_details`.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from collections.abc import Mapping, Sequence
15
+
16
+ from focus_data_toolkit.focus_json import dumps_object
17
+
18
+ # AllocatedMethodDetails.Elements[*] numeric properties (FOCUS 1.3 Split Cost Allocation).
19
+ SCA_NUMERIC_KEYS: frozenset[str] = frozenset({"AllocatedRatio", "UsageQuantity"})
20
+ # ContractApplied.Elements[*] numeric properties (FOCUS 1.3).
21
+ CONTRACT_APPLIED_NUMERIC_KEYS: frozenset[str] = frozenset(
22
+ {"ContractCommitmentAppliedCost", "ContractCommitmentAppliedQuantity"}
23
+ )
24
+
25
+
26
+ def allocated_method_details(
27
+ elements: Sequence[Mapping[str, object]],
28
+ *,
29
+ numeric_keys: frozenset[str] = SCA_NUMERIC_KEYS,
30
+ ) -> str:
31
+ """Serialise a FOCUS ``AllocatedMethodDetails`` object from its ``Elements``.
32
+
33
+ Each element's numeric properties (``AllocatedRatio`` / ``UsageQuantity`` by default)
34
+ are emitted as JSON numbers; their string values must be JSON number literals or
35
+ ``dumps_object`` raises (a stricter, safer contract than string concatenation).
36
+ """
37
+ return dumps_object({"Elements": list(elements)}, numeric_keys=numeric_keys)
38
+
39
+
40
+ def contract_applied(
41
+ commit_id: str,
42
+ contract_id: str,
43
+ applied_cost: str,
44
+ applied_qty: str,
45
+ applied_unit: str,
46
+ ) -> str:
47
+ """FOCUS 1.3 ``ContractApplied`` JSON: a top-level ``Elements`` array linking a Cost
48
+ and Usage row to the Contract Commitment dataset via ``ContractCommitmentID``. The
49
+ applied cost/quantity are emitted as JSON numbers."""
50
+ return dumps_object(
51
+ {
52
+ "Elements": [
53
+ {
54
+ "ContractID": contract_id,
55
+ "ContractCommitmentID": commit_id,
56
+ "ContractCommitmentAppliedCost": applied_cost,
57
+ "ContractCommitmentAppliedQuantity": applied_qty,
58
+ "ContractCommitmentAppliedUnit": applied_unit,
59
+ }
60
+ ]
61
+ },
62
+ numeric_keys=CONTRACT_APPLIED_NUMERIC_KEYS,
63
+ )
@@ -0,0 +1,71 @@
1
+ """The scenario-selection loop shared by every generator.
2
+
3
+ Draws exactly one ``rng.random()`` per output position (as the historical generators did) and
4
+ dispatches to a scenario builder via the version adapter's ladder, preserving the original
5
+ ``if/elif`` semantics precisely.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import random
11
+ from collections.abc import Callable
12
+
13
+ from focus_data_toolkit.generators.engine import scenarios_core
14
+
15
+ DEFAULT_ROWS = 1000
16
+
17
+ # A builder returns either one row or a whole group of rows; the branch's ``group`` flag
18
+ # (mirrored in the isinstance checks below) says which, so the union is narrowed per call.
19
+ _Builder = Callable[..., "dict[str, str] | list[dict[str, str]]"]
20
+
21
+ _BUILDERS: dict[str, _Builder] = {
22
+ "credit": scenarios_core.credit_row,
23
+ "tax": scenarios_core.tax_row,
24
+ "purchase": scenarios_core.standalone_purchase_row,
25
+ "split": scenarios_core.split_allocation_row,
26
+ "commitment": scenarios_core.commitment_group,
27
+ }
28
+
29
+
30
+ def generate_rows(
31
+ rows: int = DEFAULT_ROWS,
32
+ seed: int | None = None,
33
+ *,
34
+ include_credits: bool = False,
35
+ profile,
36
+ adapter,
37
+ ) -> list[dict[str, str]]:
38
+ """Return ``rows`` synthetic records for ``profile``/``adapter`` as ordered string dicts.
39
+
40
+ ``rows``/``seed`` default to the historical per-module values (1000 rows; the adapter's
41
+ default seed) so the shim ``generate_rows()`` / ``generate_rows(rows=N)`` calls keep working.
42
+ """
43
+ if seed is None:
44
+ seed = adapter.default_seed
45
+ if rows < 1:
46
+ raise ValueError("rows must be >= 1")
47
+ rng = random.Random(seed)
48
+ out: list[dict[str, str]] = []
49
+ while len(out) < rows:
50
+ i = len(out)
51
+ remaining = rows - i
52
+ roll = rng.random()
53
+ chosen = None
54
+ for branch in adapter.ladder:
55
+ if branch.requires_credits and not include_credits:
56
+ continue
57
+ if roll < branch.threshold:
58
+ if branch.min_remaining is None or remaining >= branch.min_remaining:
59
+ chosen = branch
60
+ break # first threshold match wins (elif semantics); guard failure -> Usage
61
+ if chosen is None:
62
+ out.append(scenarios_core.usage_row(rng, i, remaining, profile, adapter))
63
+ continue
64
+ built = _BUILDERS[chosen.kind](rng, i, remaining, profile, adapter)
65
+ if chosen.group:
66
+ assert isinstance(built, list), chosen.kind
67
+ out.extend(built)
68
+ else:
69
+ assert isinstance(built, dict), chosen.kind
70
+ out.append(built)
71
+ return out[:rows]