focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,380 @@
1
+ """Provider- and version-agnostic FOCUS row builders.
2
+
3
+ One implementation per scenario (usage / purchase / tax / credit / split-allocation /
4
+ commitment), parameterised by a ``ProviderProfile`` and a ``VersionAdapter``. The historical
5
+ per-provider RNG draw order is preserved exactly: the shared skeleton draws in the same order,
6
+ and each provider callable owns its own draw count/alphabet. Provider constants live in the
7
+ profile, version deltas in the adapter — no FOCUS rule is implemented here more than once.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import random
14
+ from datetime import timedelta
15
+ from decimal import Decimal
16
+
17
+ from focus_data_toolkit.generators.engine.context import ResourceRef, RowContext
18
+ from focus_data_toolkit.generators.engine.determinism import (
19
+ BILLING_END,
20
+ BILLING_START,
21
+ COMMIT_RATE,
22
+ COMMIT_TERM_HOURS,
23
+ COST_CENTERS,
24
+ COST_Q,
25
+ ENVIRONMENTS,
26
+ OWNERS,
27
+ PRICE_Q,
28
+ PRIVATE_RATE,
29
+ QTY_Q,
30
+ contract_id_for,
31
+ hexid,
32
+ iso,
33
+ period,
34
+ q,
35
+ s,
36
+ set_currency,
37
+ sku_price_details,
38
+ )
39
+ from focus_data_toolkit.generators.engine.json_focus import allocated_method_details
40
+
41
+ # Split Cost Allocation vocabularies (FOCUS 1.3; identical across providers).
42
+ ALLOCATION_METHODS: tuple[tuple[str, dict[str, object]], ...] = (
43
+ ("split-proportional", {"x_Strategy": "Proportional", "x_Basis": "vCPUSeconds"}),
44
+ ("split-even", {"x_Strategy": "Even", "x_Basis": "Workloads"}),
45
+ ("split-weighted", {"x_Strategy": "Weighted", "x_Basis": "MemoryBytes"}),
46
+ )
47
+ ALLOCATION_WORKLOADS = ("checkout", "search", "billing", "analytics", "ingestion")
48
+
49
+
50
+ def base_row(rng: random.Random, profile, adapter) -> tuple[dict[str, str], RowContext]:
51
+ """Return (row, ctx) with identity/account/period-independent fields filled."""
52
+ billing_id, billing_name = rng.choice(profile.billing_accounts)
53
+ sub_id, sub_name = rng.choice(profile.sub_accounts)
54
+ row = {name: "" for name in adapter.columns}
55
+ row["ProviderName"] = profile.provider_name
56
+ row["PublisherName"] = profile.publisher_name
57
+ row["InvoiceIssuerName"] = profile.invoice_issuer_name
58
+ row["InvoiceId"] = profile.invoice_id(billing_id)
59
+ row["BillingAccountId"] = billing_id
60
+ row["BillingAccountName"] = billing_name
61
+ row["BillingAccountType"] = profile.billing_account_type
62
+ row["SubAccountId"] = sub_id
63
+ row["SubAccountName"] = sub_name
64
+ row["SubAccountType"] = profile.sub_account_type
65
+ row["BillingPeriodStart"] = iso(BILLING_START)
66
+ row["BillingPeriodEnd"] = iso(BILLING_END)
67
+ row["BillingCurrency"] = "USD"
68
+ adapter.fill_version_identity(row, profile)
69
+ env_key, cost_center_key, owner_key = profile.tag_keys
70
+ row["Tags"] = json.dumps(
71
+ {
72
+ env_key: rng.choice(ENVIRONMENTS),
73
+ cost_center_key: rng.choice(COST_CENTERS),
74
+ owner_key: rng.choice(OWNERS),
75
+ },
76
+ separators=(",", ":"),
77
+ )
78
+ return row, RowContext(billing_id=billing_id, sub_id=sub_id, sub_name=sub_name)
79
+
80
+
81
+ def _set_service(row: dict[str, str], spec) -> None:
82
+ row["ServiceName"] = spec.name
83
+ row["ServiceCategory"] = spec.category
84
+ row["ServiceSubcategory"] = spec.subcategory
85
+
86
+
87
+ def _set_resource_sku(
88
+ rng: random.Random, row: dict[str, str], spec, ctx: RowContext,
89
+ region_id: str, region_name: str, resource_name: str, profile,
90
+ ) -> None:
91
+ row["RegionId"] = region_id
92
+ row["RegionName"] = region_name
93
+ ref = ResourceRef(
94
+ spec=spec, region_id=region_id, region_name=region_name,
95
+ billing_id=ctx.billing_id, sub_id=ctx.sub_id, sub_name=ctx.sub_name,
96
+ resource_name=resource_name,
97
+ )
98
+ row["ResourceId"] = profile.resource_id(ref)
99
+ row["ResourceName"] = resource_name
100
+ row["ResourceType"] = spec.resource_type
101
+ row["SkuId"] = profile.sku_id(rng, spec)
102
+ row["SkuMeter"] = spec.sku_meter
103
+ row["SkuPriceId"] = profile.sku_price_id(rng)
104
+ row["SkuPriceDetails"] = sku_price_details(dict(spec.sku_details))
105
+
106
+
107
+ def usage_row(rng: random.Random, i: int, remaining: int, profile, adapter) -> dict[str, str]:
108
+ spec = rng.choice(profile.services)
109
+ region_id, region_name, azs = rng.choice(profile.regions)
110
+ row, ctx = base_row(rng, profile, adapter)
111
+ row["ChargePeriodStart"], row["ChargePeriodEnd"] = period(i, spec.granularity)
112
+ _set_service(row, spec)
113
+ resource_name = profile.resource_name(rng, spec)
114
+ _set_resource_sku(rng, row, spec, ctx, region_id, region_name, resource_name, profile)
115
+ if spec.zonal:
116
+ row["AvailabilityZone"] = rng.choice(azs)
117
+
118
+ quantity = q(Decimal(rng.uniform(float(spec.qty_low), float(spec.qty_high))), QTY_Q)
119
+ jitter = Decimal(rng.uniform(0.97, 1.03))
120
+ list_unit = q(spec.unit_price_usd * jitter, PRICE_Q)
121
+ contracted_unit = q(list_unit * PRIVATE_RATE, PRICE_Q)
122
+ list_cost = q(list_unit * quantity, COST_Q)
123
+ contracted_cost = q(contracted_unit * quantity, COST_Q)
124
+
125
+ row["ChargeCategory"] = "Usage"
126
+ row["ChargeFrequency"] = "Usage-Based"
127
+ row["ChargeDescription"] = spec.description
128
+ row["PricingCategory"] = "Standard"
129
+ row["BilledCost"] = s(contracted_cost)
130
+ row["EffectiveCost"] = s(contracted_cost)
131
+ row["ListCost"] = s(list_cost)
132
+ row["ContractedCost"] = s(contracted_cost)
133
+ row["ListUnitPrice"] = s(list_unit)
134
+ row["ContractedUnitPrice"] = s(contracted_unit)
135
+ row["PricingQuantity"] = s(quantity)
136
+ row["PricingUnit"] = spec.pricing_unit
137
+ row["ConsumedQuantity"] = s(quantity)
138
+ row["ConsumedUnit"] = spec.pricing_unit
139
+ set_currency(
140
+ row, "EUR" if rng.random() < 0.10 else "USD", list_unit, contracted_unit, contracted_cost
141
+ )
142
+ return row
143
+
144
+
145
+ def standalone_purchase_row(rng: random.Random, i: int, remaining: int, profile, adapter) -> dict[str, str]:
146
+ spec = rng.choice(profile.services)
147
+ region_id, region_name, _ = rng.choice(profile.regions)
148
+ row, ctx = base_row(rng, profile, adapter)
149
+ row["ChargePeriodStart"], row["ChargePeriodEnd"] = period(i, "daily")
150
+ _set_service(row, spec)
151
+ resource_name = profile.resource_name(rng, spec)
152
+ _set_resource_sku(rng, row, spec, ctx, region_id, region_name, resource_name, profile)
153
+
154
+ amount = q(Decimal(rng.uniform(20.0, 800.0)), COST_Q)
155
+ row["ChargeCategory"] = "Purchase"
156
+ row["ChargeFrequency"] = "Recurring"
157
+ row["ChargeDescription"] = f"{spec.name} subscription fee"
158
+ row["PricingCategory"] = "Standard"
159
+ row["BilledCost"] = s(amount)
160
+ row["EffectiveCost"] = "0" # purchase covers future eligible charges
161
+ row["ListCost"] = s(amount)
162
+ row["ContractedCost"] = s(amount)
163
+ row["ListUnitPrice"] = s(amount)
164
+ row["ContractedUnitPrice"] = s(amount)
165
+ row["PricingQuantity"] = "1"
166
+ row["PricingUnit"] = "Units"
167
+ set_currency(row, "USD", amount, amount, Decimal("0"))
168
+ return row
169
+
170
+
171
+ def tax_row(rng: random.Random, i: int, remaining: int, profile, adapter) -> dict[str, str]:
172
+ spec = rng.choice(profile.services)
173
+ row, _ = base_row(rng, profile, adapter)
174
+ row["ChargePeriodStart"], row["ChargePeriodEnd"] = period(i, "daily")
175
+ _set_service(row, spec)
176
+ amount = q(Decimal(rng.uniform(0.5, 50.0)), COST_Q)
177
+ amount_str = s(amount)
178
+ row["ChargeCategory"] = "Tax"
179
+ row["ChargeFrequency"] = "One-Time"
180
+ row["ChargeDescription"] = f"Tax for {spec.name}"
181
+ row["BilledCost"] = amount_str
182
+ row["EffectiveCost"] = amount_str
183
+ row["ListCost"] = amount_str
184
+ row["ContractedCost"] = amount_str
185
+ adapter.on_tax_row(row, amount_str)
186
+ return row
187
+
188
+
189
+ def credit_row(rng: random.Random, i: int, remaining: int, profile, adapter) -> dict[str, str]:
190
+ spec = rng.choice(profile.services)
191
+ row, _ = base_row(rng, profile, adapter)
192
+ row["ChargePeriodStart"], row["ChargePeriodEnd"] = period(i, "daily")
193
+ _set_service(row, spec)
194
+ negative = s(-q(Decimal(rng.uniform(1.0, 100.0)), COST_Q))
195
+ row["ChargeCategory"] = "Credit"
196
+ row["ChargeFrequency"] = "One-Time"
197
+ row["ChargeDescription"] = f"Credit for {spec.name}"
198
+ row["BilledCost"] = negative
199
+ row["EffectiveCost"] = negative
200
+ row["ListCost"] = negative
201
+ row["ContractedCost"] = negative
202
+ adapter.on_credit_row(row, negative)
203
+ return row
204
+
205
+
206
+ def split_allocation_row(rng: random.Random, i: int, remaining: int, profile, adapter) -> dict[str, str]:
207
+ """A Split Cost Allocation row (FOCUS 1.3): a shared resource's cost allocated to a
208
+ consuming workload. ``ResourceId`` is the shared resource; the ``Allocated*`` columns name
209
+ the workload that received the split."""
210
+ spec = profile.commitment_service # shared compute host split across workloads
211
+ region_id, region_name, azs = rng.choice(profile.regions)
212
+ row, ctx = base_row(rng, profile, adapter)
213
+ row["ChargePeriodStart"], row["ChargePeriodEnd"] = period(i, "hourly")
214
+ _set_service(row, spec)
215
+ shared_name = f"shared-host-{hexid(rng, 8)}"
216
+ _set_resource_sku(rng, row, spec, ctx, region_id, region_name, shared_name, profile)
217
+ row["AvailabilityZone"] = rng.choice(azs)
218
+
219
+ quantity = q(Decimal(rng.uniform(0.05, 1.0)), QTY_Q)
220
+ jitter = Decimal(rng.uniform(0.97, 1.03))
221
+ list_unit = q(spec.unit_price_usd * jitter, PRICE_Q)
222
+ contracted_unit = q(list_unit * PRIVATE_RATE, PRICE_Q)
223
+ list_cost = q(list_unit * quantity, COST_Q)
224
+ contracted_cost = q(contracted_unit * quantity, COST_Q)
225
+
226
+ row["ChargeCategory"] = "Usage"
227
+ row["ChargeFrequency"] = "Usage-Based"
228
+ row["ChargeDescription"] = profile.split_allocation_description
229
+ row["PricingCategory"] = "Standard"
230
+ row["BilledCost"] = s(contracted_cost)
231
+ row["EffectiveCost"] = s(contracted_cost)
232
+ row["ListCost"] = s(list_cost)
233
+ row["ContractedCost"] = s(contracted_cost)
234
+ row["ListUnitPrice"] = s(list_unit)
235
+ row["ContractedUnitPrice"] = s(contracted_unit)
236
+ row["PricingQuantity"] = s(quantity)
237
+ row["PricingUnit"] = spec.pricing_unit
238
+ row["ConsumedQuantity"] = s(quantity)
239
+ row["ConsumedUnit"] = spec.pricing_unit
240
+
241
+ workload = rng.choice(ALLOCATION_WORKLOADS)
242
+ method_id, method_details = rng.choice(ALLOCATION_METHODS)
243
+ row["AllocatedMethodId"] = method_id
244
+ # FOCUS 1.3 split allocation details: an Elements array exposing the allocated ratio and the
245
+ # usage that drove the split (plus x_ method metadata). AllocatedRatio / UsageQuantity are
246
+ # Numeric -> emitted as JSON numbers (single-source builder).
247
+ element = {
248
+ "AllocatedRatio": s(quantity),
249
+ "UsageUnit": spec.pricing_unit,
250
+ "UsageQuantity": s(quantity),
251
+ **method_details,
252
+ }
253
+ row["AllocatedMethodDetails"] = allocated_method_details([element])
254
+ row["AllocatedResourceId"] = profile.allocated_resource_id(rng, region_id, ctx, workload)
255
+ row["AllocatedResourceName"] = f"workload-{workload}"
256
+ row["AllocatedTags"] = json.dumps(
257
+ {"workload": workload, profile.tag_keys[1]: rng.choice(COST_CENTERS)}, separators=(",", ":")
258
+ )
259
+ set_currency(row, "USD", list_unit, contracted_unit, contracted_cost)
260
+ return row
261
+
262
+
263
+ def commitment_group(rng: random.Random, i0: int, remaining: int, profile, adapter) -> list[dict[str, str]]:
264
+ """A commitment Purchase row + linked committed-usage rows (shared CommitmentDiscountId).
265
+
266
+ The Purchase row carries the full commitment terms, which the Contract Commitment dataset
267
+ re-derives so the two datasets join on ``ContractCommitmentId`` == ``CommitmentDiscountId``.
268
+ """
269
+ spec = profile.commitment_service
270
+ commit = profile.commitment
271
+ region_id, region_name, azs = rng.choice(profile.regions)
272
+ az = rng.choice(azs)
273
+ spend_based = rng.random() < 0.6
274
+
275
+ commit_id = ""
276
+ if commit.commit_id_before_base_row:
277
+ commit_id = commit.commit_id(rng, region_id, "", spend_based)
278
+
279
+ commit_name = commit.commit_name(spend_based)
280
+ commit_type = commit.commit_type(spend_based)
281
+ commit_category = commit.commit_category(spend_based)
282
+ commit_unit = commit.commit_unit(spend_based)
283
+
284
+ list_unit = q(spec.unit_price_usd, PRICE_Q)
285
+ commit_unit_price = q(list_unit * COMMIT_RATE, PRICE_Q)
286
+ upfront = q(commit_unit_price * COMMIT_TERM_HOURS, COST_Q)
287
+ commit_total_qty = s(upfront) if spend_based else s(COMMIT_TERM_HOURS)
288
+
289
+ purchase, ctx = base_row(rng, profile, adapter)
290
+ if not commit.commit_id_before_base_row:
291
+ commit_id = commit.commit_id(rng, region_id, ctx.sub_id, spend_based)
292
+
293
+ purchase["ChargePeriodStart"] = iso(BILLING_START)
294
+ purchase["ChargePeriodEnd"] = iso(BILLING_START + timedelta(hours=1))
295
+ _set_service(purchase, spec)
296
+ commit_resource = commit.commit_resource_name(rng, spend_based)
297
+ purchase["ResourceId"] = commit_id
298
+ purchase["ResourceName"] = commit_resource
299
+ purchase["ResourceType"] = commit_type
300
+ purchase["RegionId"] = region_id
301
+ purchase["RegionName"] = region_name
302
+ purchase["SkuId"] = commit.purchase_sku_id(rng)
303
+ purchase["SkuMeter"] = "Commitment"
304
+ purchase["SkuPriceId"] = profile.sku_price_id(rng)
305
+ purchase["SkuPriceDetails"] = commit.purchase_sku_details(spend_based)
306
+ purchase["ChargeCategory"] = "Purchase"
307
+ purchase["ChargeFrequency"] = "One-Time"
308
+ purchase["ChargeDescription"] = commit.purchase_description(commit_type)
309
+ purchase["PricingCategory"] = "Standard"
310
+ purchase["BilledCost"] = s(upfront)
311
+ purchase["EffectiveCost"] = "0.000000"
312
+ purchase["ListCost"] = s(upfront)
313
+ purchase["ContractedCost"] = s(upfront)
314
+ purchase["ListUnitPrice"] = s(upfront)
315
+ purchase["ContractedUnitPrice"] = s(upfront)
316
+ purchase["PricingQuantity"] = "1"
317
+ purchase["PricingUnit"] = "Units"
318
+ purchase["CommitmentDiscountId"] = commit_id
319
+ purchase["CommitmentDiscountName"] = commit_name
320
+ purchase["CommitmentDiscountCategory"] = commit_category
321
+ purchase["CommitmentDiscountType"] = commit_type
322
+ purchase["CommitmentDiscountQuantity"] = commit_total_qty
323
+ purchase["CommitmentDiscountUnit"] = commit_unit
324
+ set_currency(purchase, "USD", upfront, upfront, Decimal("0"))
325
+
326
+ # Full billing identity of the commitment, reused verbatim by every linked usage row so
327
+ # account/invoice grouping and reconciliation stay consistent within the group.
328
+ billing_identity = {key: purchase[key] for key in adapter.commitment_identity_keys}
329
+ contract_id = contract_id_for(commit_id)
330
+
331
+ rows = [purchase]
332
+ n_usage = min(remaining - 1, rng.randint(5, 9))
333
+ for k in range(n_usage):
334
+ usage, _ = base_row(rng, profile, adapter)
335
+ usage.update(billing_identity)
336
+ usage["ChargePeriodStart"], usage["ChargePeriodEnd"] = period(i0 + 1 + k, "hourly")
337
+ _set_service(usage, spec)
338
+ resource_name = profile.committed_resource_name(rng, spec, k)
339
+ usage["RegionId"] = region_id
340
+ usage["RegionName"] = region_name
341
+ ref = ResourceRef(
342
+ spec=spec, region_id=region_id, region_name=region_name,
343
+ billing_id=ctx.billing_id, sub_id=ctx.sub_id, sub_name=ctx.sub_name,
344
+ resource_name=resource_name,
345
+ )
346
+ usage["ResourceId"] = profile.resource_id(ref)
347
+ usage["ResourceName"] = resource_name
348
+ usage["ResourceType"] = spec.resource_type
349
+ usage["AvailabilityZone"] = az
350
+ usage["SkuId"] = profile.sku_id(rng, spec)
351
+ usage["SkuMeter"] = spec.sku_meter
352
+ usage["SkuPriceId"] = profile.sku_price_id(rng)
353
+ usage["SkuPriceDetails"] = sku_price_details(dict(spec.sku_details))
354
+ list_cost = q(list_unit, COST_Q)
355
+ effective = q(commit_unit_price, COST_Q)
356
+ usage["ChargeCategory"] = "Usage"
357
+ usage["ChargeFrequency"] = "Usage-Based"
358
+ usage["ChargeDescription"] = f"{spec.name} committed usage"
359
+ usage["PricingCategory"] = "Committed"
360
+ usage["BilledCost"] = "0.000000" # covered by the upfront purchase
361
+ usage["EffectiveCost"] = s(effective) # amortised, < ListCost
362
+ usage["ListCost"] = s(list_cost)
363
+ usage["ContractedCost"] = s(effective)
364
+ usage["ListUnitPrice"] = s(list_unit)
365
+ usage["ContractedUnitPrice"] = s(commit_unit_price)
366
+ usage["PricingQuantity"] = "1.0000"
367
+ usage["PricingUnit"] = "Hours"
368
+ usage["ConsumedQuantity"] = "1.0000"
369
+ usage["ConsumedUnit"] = "Hours"
370
+ usage["CommitmentDiscountId"] = commit_id
371
+ usage["CommitmentDiscountName"] = commit_name
372
+ usage["CommitmentDiscountCategory"] = commit_category
373
+ usage["CommitmentDiscountType"] = commit_type
374
+ usage["CommitmentDiscountStatus"] = "Used"
375
+ usage["CommitmentDiscountQuantity"] = s(effective) if spend_based else "1.0000"
376
+ usage["CommitmentDiscountUnit"] = commit_unit
377
+ adapter.on_commit_usage(usage, commit_id, contract_id, s(effective))
378
+ set_currency(usage, "USD", list_unit, commit_unit_price, effective)
379
+ rows.append(usage)
380
+ return rows
@@ -0,0 +1,151 @@
1
+ """CSV serialization and the ``python -m`` CLI shared by every generator module.
2
+
3
+ Serialization is kept separate from generation: the row builders never touch CSV, and these
4
+ functions never invent data.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import argparse
10
+ import csv
11
+ import io
12
+ from datetime import timedelta
13
+ from pathlib import Path
14
+
15
+ from focus_data_toolkit.generators.engine.determinism import (
16
+ COMMIT_TERM_DAYS,
17
+ contract_id_for,
18
+ iso,
19
+ parse_iso,
20
+ )
21
+ from focus_data_toolkit.generators.engine.ladder import generate_rows
22
+
23
+ DEFAULT_ROWS = 1000
24
+
25
+
26
+ def generate_csv_bytes(
27
+ rows: int = DEFAULT_ROWS,
28
+ seed: int | None = None,
29
+ *,
30
+ include_credits: bool = False,
31
+ profile,
32
+ adapter,
33
+ ) -> bytes:
34
+ """Serialise the Cost and Usage rows to deterministic UTF-8 CSV bytes (LF line endings)."""
35
+ if seed is None:
36
+ seed = adapter.default_seed
37
+ buffer = io.StringIO()
38
+ writer = csv.DictWriter(buffer, fieldnames=list(adapter.columns), lineterminator="\n")
39
+ writer.writeheader()
40
+ for record in generate_rows(rows, seed, include_credits=include_credits, profile=profile, adapter=adapter):
41
+ writer.writerow(record)
42
+ return buffer.getvalue().encode("utf-8")
43
+
44
+
45
+ def generate_contract_commitment_rows(
46
+ rows: int = DEFAULT_ROWS,
47
+ seed: int | None = None,
48
+ *,
49
+ profile,
50
+ adapter,
51
+ ) -> list[dict[str, str]]:
52
+ """Return the Contract Commitment dataset for the same (rows, seed).
53
+
54
+ Each commitment Purchase row yields exactly one Contract Commitment row, so
55
+ ``ContractCommitmentId`` == ``CommitmentDiscountId`` is a verifiable foreign key.
56
+ """
57
+ if adapter.contract_commitment_columns is None:
58
+ raise ValueError(f"FOCUS {adapter.version} has no Contract Commitment dataset")
59
+ if seed is None:
60
+ seed = adapter.default_seed
61
+ out: list[dict[str, str]] = []
62
+ seen: set[str] = set()
63
+ for cu in generate_rows(rows, seed, include_credits=False, profile=profile, adapter=adapter):
64
+ if cu["ChargeCategory"] != "Purchase" or not cu["CommitmentDiscountId"]:
65
+ continue
66
+ commit_id = cu["CommitmentDiscountId"]
67
+ if commit_id in seen:
68
+ continue
69
+ seen.add(commit_id)
70
+ period_start = parse_iso(cu["ChargePeriodStart"])
71
+ period_end = period_start + timedelta(days=COMMIT_TERM_DAYS)
72
+ contract_id = contract_id_for(commit_id)
73
+ row = {name: "" for name in adapter.contract_commitment_columns}
74
+ row["ContractCommitmentId"] = commit_id
75
+ row["ContractCommitmentType"] = cu["CommitmentDiscountType"]
76
+ row["ContractCommitmentCategory"] = cu["CommitmentDiscountCategory"]
77
+ row["ContractCommitmentCost"] = cu["BilledCost"] # the upfront commitment cost
78
+ row["ContractCommitmentQuantity"] = cu["CommitmentDiscountQuantity"]
79
+ row["ContractCommitmentUnit"] = cu["CommitmentDiscountUnit"]
80
+ row["ContractCommitmentDescription"] = cu["CommitmentDiscountName"]
81
+ row["ContractCommitmentPeriodStart"] = iso(period_start)
82
+ row["ContractCommitmentPeriodEnd"] = iso(period_end)
83
+ row["ContractId"] = contract_id
84
+ row["ContractPeriodStart"] = iso(period_start)
85
+ row["ContractPeriodEnd"] = iso(period_end)
86
+ row["BillingCurrency"] = "USD"
87
+ out.append(row)
88
+ return out
89
+
90
+
91
+ def generate_contract_commitment_csv_bytes(
92
+ rows: int = DEFAULT_ROWS,
93
+ seed: int | None = None,
94
+ *,
95
+ profile,
96
+ adapter,
97
+ ) -> bytes:
98
+ """Serialise the Contract Commitment dataset to deterministic UTF-8 CSV bytes (LF)."""
99
+ if seed is None:
100
+ seed = adapter.default_seed
101
+ buffer = io.StringIO()
102
+ writer = csv.DictWriter(
103
+ buffer, fieldnames=list(adapter.contract_commitment_columns), lineterminator="\n"
104
+ )
105
+ writer.writeheader()
106
+ for record in generate_contract_commitment_rows(rows, seed, profile=profile, adapter=adapter):
107
+ writer.writerow(record)
108
+ return buffer.getvalue().encode("utf-8")
109
+
110
+
111
+ def main(argv: list[str] | None = None, *, profile, adapter) -> int:
112
+ """``python -m focus_data_toolkit.generators.generate_<provider>_focus_<version>`` entry point."""
113
+ label = f"{profile.provider_name} FOCUS {adapter.version}"
114
+ has_cc = adapter.contract_commitment_columns is not None
115
+ parser = argparse.ArgumentParser(description=f"Generate synthetic {label} CSV data.")
116
+ if has_cc:
117
+ parser.add_argument(
118
+ "--dataset",
119
+ choices=("cost_and_usage", "contract_commitment"),
120
+ default="cost_and_usage",
121
+ help="FOCUS dataset to emit (default: cost_and_usage)",
122
+ )
123
+ parser.add_argument("--rows", type=int, default=DEFAULT_ROWS, help="number of data rows")
124
+ parser.add_argument("--seed", type=int, default=adapter.default_seed, help="deterministic RNG seed")
125
+ parser.add_argument("--out", type=Path, default=None, help="output CSV path")
126
+ parser.add_argument(
127
+ "--include-credits",
128
+ action="store_true",
129
+ help="emit some Credit rows with negative BilledCost (excluded from the default fixture)",
130
+ )
131
+ args = parser.parse_args(argv)
132
+
133
+ dataset = getattr(args, "dataset", "cost_and_usage")
134
+ default_stem = f"focus_sample_{{}}_{profile.key}"
135
+ if dataset == "contract_commitment":
136
+ payload = generate_contract_commitment_csv_bytes(
137
+ args.rows, args.seed, profile=profile, adapter=adapter
138
+ )
139
+ out = args.out or Path(f"{default_stem.format('contractcommitment')}.csv")
140
+ columns = adapter.contract_commitment_columns
141
+ else:
142
+ payload = generate_csv_bytes(
143
+ args.rows, args.seed, include_credits=args.include_credits, profile=profile, adapter=adapter
144
+ )
145
+ out = args.out or Path(f"{default_stem.format('costandusage')}_{args.rows}.csv")
146
+ columns = adapter.columns
147
+
148
+ out.parent.mkdir(parents=True, exist_ok=True)
149
+ out.write_bytes(payload)
150
+ print(f"Wrote {dataset} ({len(columns)} {label} columns) -> {out}")
151
+ return 0
@@ -0,0 +1,19 @@
1
+ """Deterministic generator for synthetic AWS data in the FOCUS 1.2 format.
2
+
3
+ Thin shim: all logic lives in :mod:`focus_data_toolkit.generators.engine`, bound to the AWS
4
+ provider profile and the FOCUS 1.2 version adapter. Exposes the historical module API
5
+ (``COLUMNS``, ``generate_rows``, ``generate_csv_bytes``, ``main``) and the
6
+ ``python -m focus_data_toolkit.generators.generate_aws_focus_1_2`` entry point unchanged.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from focus_data_toolkit.generators._shim import build_module_api
12
+ from focus_data_toolkit.generators.providers.aws import AWS
13
+ from focus_data_toolkit.generators.versions.v1_2 import V12
14
+
15
+ _api = build_module_api(AWS, V12)
16
+ globals().update(_api)
17
+
18
+ if __name__ == "__main__":
19
+ raise SystemExit(_api["main"]())
@@ -0,0 +1,20 @@
1
+ """Deterministic generator for synthetic AWS data in the FOCUS 1.3 format.
2
+
3
+ Thin shim: all logic lives in :mod:`focus_data_toolkit.generators.engine`, bound to the AWS
4
+ provider profile and the FOCUS 1.3 version adapter. Exposes the historical module API
5
+ (``COLUMNS``, ``CONTRACT_COMMITMENT_COLUMNS``, ``generate_rows``, ``generate_csv_bytes``,
6
+ ``generate_contract_commitment_csv_bytes``, ``main``) and the
7
+ ``python -m focus_data_toolkit.generators.generate_aws_focus_1_3`` entry point unchanged.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from focus_data_toolkit.generators._shim import build_module_api
13
+ from focus_data_toolkit.generators.providers.aws import AWS
14
+ from focus_data_toolkit.generators.versions.v1_3 import V13
15
+
16
+ _api = build_module_api(AWS, V13)
17
+ globals().update(_api)
18
+
19
+ if __name__ == "__main__":
20
+ raise SystemExit(_api["main"]())
@@ -0,0 +1,17 @@
1
+ """Deterministic generator for synthetic Microsoft Azure data in the FOCUS 1.2 format.
2
+
3
+ Thin shim: logic lives in :mod:`focus_data_toolkit.generators.engine`, bound to the Azure
4
+ provider profile and the FOCUS 1.2 version adapter.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from focus_data_toolkit.generators._shim import build_module_api
10
+ from focus_data_toolkit.generators.providers.azure import AZURE
11
+ from focus_data_toolkit.generators.versions.v1_2 import V12
12
+
13
+ _api = build_module_api(AZURE, V12)
14
+ globals().update(_api)
15
+
16
+ if __name__ == "__main__":
17
+ raise SystemExit(_api["main"]())
@@ -0,0 +1,17 @@
1
+ """Deterministic generator for synthetic Microsoft Azure data in the FOCUS 1.3 format.
2
+
3
+ Thin shim: logic lives in :mod:`focus_data_toolkit.generators.engine`, bound to the Azure
4
+ provider profile and the FOCUS 1.3 version adapter.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from focus_data_toolkit.generators._shim import build_module_api
10
+ from focus_data_toolkit.generators.providers.azure import AZURE
11
+ from focus_data_toolkit.generators.versions.v1_3 import V13
12
+
13
+ _api = build_module_api(AZURE, V13)
14
+ globals().update(_api)
15
+
16
+ if __name__ == "__main__":
17
+ raise SystemExit(_api["main"]())
@@ -0,0 +1,17 @@
1
+ """Deterministic generator for synthetic Google Cloud data in the FOCUS 1.2 format.
2
+
3
+ Thin shim: logic lives in :mod:`focus_data_toolkit.generators.engine`, bound to the GCP
4
+ provider profile and the FOCUS 1.2 version adapter.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from focus_data_toolkit.generators._shim import build_module_api
10
+ from focus_data_toolkit.generators.providers.gcp import GCP
11
+ from focus_data_toolkit.generators.versions.v1_2 import V12
12
+
13
+ _api = build_module_api(GCP, V12)
14
+ globals().update(_api)
15
+
16
+ if __name__ == "__main__":
17
+ raise SystemExit(_api["main"]())
@@ -0,0 +1,17 @@
1
+ """Deterministic generator for synthetic Google Cloud data in the FOCUS 1.3 format.
2
+
3
+ Thin shim: logic lives in :mod:`focus_data_toolkit.generators.engine`, bound to the GCP
4
+ provider profile and the FOCUS 1.3 version adapter.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from focus_data_toolkit.generators._shim import build_module_api
10
+ from focus_data_toolkit.generators.providers.gcp import GCP
11
+ from focus_data_toolkit.generators.versions.v1_3 import V13
12
+
13
+ _api = build_module_api(GCP, V13)
14
+ globals().update(_api)
15
+
16
+ if __name__ == "__main__":
17
+ raise SystemExit(_api["main"]())
@@ -0,0 +1,29 @@
1
+ """Provider profiles: only what is genuinely specific to AWS, Azure and GCP.
2
+
3
+ A provider is *data* (names, service/region/account tables) plus a few pure callables
4
+ (resource-id / sku-id formats, commitment terms). All generation logic lives in
5
+ ``generators.engine``; adding a provider is one profile module, no engine change.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from focus_data_toolkit.generators.providers.aws import AWS
11
+ from focus_data_toolkit.generators.providers.azure import AZURE
12
+ from focus_data_toolkit.generators.providers.gcp import GCP
13
+ from focus_data_toolkit.generators.providers.profile import (
14
+ CommitmentModel,
15
+ ProviderProfile,
16
+ ServiceSpec,
17
+ )
18
+
19
+ PROFILES: dict[str, ProviderProfile] = {AWS.key: AWS, AZURE.key: AZURE, GCP.key: GCP}
20
+
21
+ __all__ = [
22
+ "AWS",
23
+ "AZURE",
24
+ "GCP",
25
+ "PROFILES",
26
+ "CommitmentModel",
27
+ "ProviderProfile",
28
+ "ServiceSpec",
29
+ ]