focus-data-toolkit 0.11.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (226) hide show
  1. focus_data_toolkit-0.11.0/CHANGELOG.md +558 -0
  2. focus_data_toolkit-0.11.0/CONTRIBUTING.md +111 -0
  3. focus_data_toolkit-0.11.0/LICENSE +21 -0
  4. focus_data_toolkit-0.11.0/LICENSES/CC-BY-4.0.txt +156 -0
  5. focus_data_toolkit-0.11.0/MANIFEST.in +25 -0
  6. focus_data_toolkit-0.11.0/NOTICE +60 -0
  7. focus_data_toolkit-0.11.0/PKG-INFO +519 -0
  8. focus_data_toolkit-0.11.0/README.md +466 -0
  9. focus_data_toolkit-0.11.0/SECURITY.md +94 -0
  10. focus_data_toolkit-0.11.0/docs/audit/2026-07-improvement-plan.md +375 -0
  11. focus_data_toolkit-0.11.0/docs/audit/2026-07-repo-audit.md +267 -0
  12. focus_data_toolkit-0.11.0/docs/compatibility.md +88 -0
  13. focus_data_toolkit-0.11.0/docs/model-provenance.md +118 -0
  14. focus_data_toolkit-0.11.0/docs/releasing.md +188 -0
  15. focus_data_toolkit-0.11.0/docs/runner.md +129 -0
  16. focus_data_toolkit-0.11.0/docs/security-model.md +102 -0
  17. focus_data_toolkit-0.11.0/docs/studio.md +70 -0
  18. focus_data_toolkit-0.11.0/docs/supplements.md +142 -0
  19. focus_data_toolkit-0.11.0/docs/versioning.md +78 -0
  20. focus_data_toolkit-0.11.0/pyproject.toml +142 -0
  21. focus_data_toolkit-0.11.0/schema/model_provenance.schema.json +124 -0
  22. focus_data_toolkit-0.11.0/scripts/check_pinned_actions.py +62 -0
  23. focus_data_toolkit-0.11.0/scripts/generate_resolved_sbom.py +267 -0
  24. focus_data_toolkit-0.11.0/scripts/generate_sbom.py +177 -0
  25. focus_data_toolkit-0.11.0/scripts/verify_model_provenance.py +290 -0
  26. focus_data_toolkit-0.11.0/scripts/verify_release.py +196 -0
  27. focus_data_toolkit-0.11.0/setup.cfg +4 -0
  28. focus_data_toolkit-0.11.0/src/focus_data_toolkit/__init__.py +69 -0
  29. focus_data_toolkit-0.11.0/src/focus_data_toolkit/__main__.py +6 -0
  30. focus_data_toolkit-0.11.0/src/focus_data_toolkit/_version.py +8 -0
  31. focus_data_toolkit-0.11.0/src/focus_data_toolkit/cli.py +968 -0
  32. focus_data_toolkit-0.11.0/src/focus_data_toolkit/context/__init__.py +88 -0
  33. focus_data_toolkit-0.11.0/src/focus_data_toolkit/context/billing.py +54 -0
  34. focus_data_toolkit-0.11.0/src/focus_data_toolkit/context/provider.py +90 -0
  35. focus_data_toolkit-0.11.0/src/focus_data_toolkit/convert/__init__.py +708 -0
  36. focus_data_toolkit-0.11.0/src/focus_data_toolkit/convert/billing_period.py +65 -0
  37. focus_data_toolkit-0.11.0/src/focus_data_toolkit/convert/contract_applied.py +235 -0
  38. focus_data_toolkit-0.11.0/src/focus_data_toolkit/convert/contract_commitment.py +182 -0
  39. focus_data_toolkit-0.11.0/src/focus_data_toolkit/convert/cost_and_usage.py +179 -0
  40. focus_data_toolkit-0.11.0/src/focus_data_toolkit/convert/detect.py +39 -0
  41. focus_data_toolkit-0.11.0/src/focus_data_toolkit/convert/invoice_detail.py +199 -0
  42. focus_data_toolkit-0.11.0/src/focus_data_toolkit/convert/streaming.py +1030 -0
  43. focus_data_toolkit-0.11.0/src/focus_data_toolkit/errors.py +145 -0
  44. focus_data_toolkit-0.11.0/src/focus_data_toolkit/focus_json.py +68 -0
  45. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/__init__.py +61 -0
  46. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/_shim.py +43 -0
  47. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/engine/__init__.py +14 -0
  48. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/engine/context.py +12 -0
  49. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/engine/determinism.py +117 -0
  50. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/engine/json_focus.py +63 -0
  51. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/engine/ladder.py +71 -0
  52. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  53. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/engine/serialize.py +151 -0
  54. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  55. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  56. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  57. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  58. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  59. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  60. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/providers/__init__.py +29 -0
  61. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/providers/aws.py +186 -0
  62. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/providers/azure.py +191 -0
  63. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/providers/gcp.py +194 -0
  64. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/providers/profile.py +123 -0
  65. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/scenarios.py +178 -0
  66. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/versions/__init__.py +17 -0
  67. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/versions/adapter.py +41 -0
  68. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/versions/v1_2.py +111 -0
  69. focus_data_toolkit-0.11.0/src/focus_data_toolkit/generators/versions/v1_3.py +154 -0
  70. focus_data_toolkit-0.11.0/src/focus_data_toolkit/io/__init__.py +1 -0
  71. focus_data_toolkit-0.11.0/src/focus_data_toolkit/io/atomic_writer.py +462 -0
  72. focus_data_toolkit-0.11.0/src/focus_data_toolkit/io/csv_io.py +128 -0
  73. focus_data_toolkit-0.11.0/src/focus_data_toolkit/io/parquet_io.py +528 -0
  74. focus_data_toolkit-0.11.0/src/focus_data_toolkit/io/records.py +92 -0
  75. focus_data_toolkit-0.11.0/src/focus_data_toolkit/io/row_source.py +117 -0
  76. focus_data_toolkit-0.11.0/src/focus_data_toolkit/lifecycle.py +342 -0
  77. focus_data_toolkit-0.11.0/src/focus_data_toolkit/manifest.py +114 -0
  78. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/__init__.py +43 -0
  79. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/capabilities.py +66 -0
  80. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  81. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  82. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  83. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/focus_json_keys.py +112 -0
  84. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  85. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/json_schema_check.py +205 -0
  86. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  87. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  88. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  89. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  90. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  91. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/model_provenance.json +58 -0
  92. focus_data_toolkit-0.11.0/src/focus_data_toolkit/model/validator.py +498 -0
  93. focus_data_toolkit-0.11.0/src/focus_data_toolkit/modes.py +18 -0
  94. focus_data_toolkit-0.11.0/src/focus_data_toolkit/official_validator.py +61 -0
  95. focus_data_toolkit-0.11.0/src/focus_data_toolkit/progress.py +89 -0
  96. focus_data_toolkit-0.11.0/src/focus_data_toolkit/provenance.py +106 -0
  97. focus_data_toolkit-0.11.0/src/focus_data_toolkit/py.typed +1 -0
  98. focus_data_toolkit-0.11.0/src/focus_data_toolkit/runtime.py +243 -0
  99. focus_data_toolkit-0.11.0/src/focus_data_toolkit/schema/__init__.py +17 -0
  100. focus_data_toolkit-0.11.0/src/focus_data_toolkit/schema/detection.py +274 -0
  101. focus_data_toolkit-0.11.0/src/focus_data_toolkit/schema/registry.py +127 -0
  102. focus_data_toolkit-0.11.0/src/focus_data_toolkit/storage/__init__.py +1 -0
  103. focus_data_toolkit-0.11.0/src/focus_data_toolkit/storage/external_index.py +99 -0
  104. focus_data_toolkit-0.11.0/src/focus_data_toolkit/storage/spill.py +150 -0
  105. focus_data_toolkit-0.11.0/src/focus_data_toolkit/studio/__init__.py +19 -0
  106. focus_data_toolkit-0.11.0/src/focus_data_toolkit/studio/app.py +467 -0
  107. focus_data_toolkit-0.11.0/src/focus_data_toolkit/studio/config.py +42 -0
  108. focus_data_toolkit-0.11.0/src/focus_data_toolkit/studio/frontend/app.js +214 -0
  109. focus_data_toolkit-0.11.0/src/focus_data_toolkit/studio/frontend/index.html +101 -0
  110. focus_data_toolkit-0.11.0/src/focus_data_toolkit/studio/frontend/style.css +60 -0
  111. focus_data_toolkit-0.11.0/src/focus_data_toolkit/studio/jobs.py +142 -0
  112. focus_data_toolkit-0.11.0/src/focus_data_toolkit/studio/preview.py +32 -0
  113. focus_data_toolkit-0.11.0/src/focus_data_toolkit/studio/security.py +125 -0
  114. focus_data_toolkit-0.11.0/src/focus_data_toolkit/studio/server.py +71 -0
  115. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/__init__.py +50 -0
  116. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  117. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  118. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  119. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  120. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  121. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  122. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/adapters/registry.py +215 -0
  123. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/apply.py +318 -0
  124. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/gaps.py +219 -0
  125. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/kinds.py +118 -0
  126. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/loader.py +409 -0
  127. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/spec.py +74 -0
  128. focus_data_toolkit-0.11.0/src/focus_data_toolkit/supplement/validate.py +215 -0
  129. focus_data_toolkit-0.11.0/src/focus_data_toolkit/validate/__init__.py +15 -0
  130. focus_data_toolkit-0.11.0/src/focus_data_toolkit/validate/allocation.py +333 -0
  131. focus_data_toolkit-0.11.0/src/focus_data_toolkit/validate/bundle.py +254 -0
  132. focus_data_toolkit-0.11.0/src/focus_data_toolkit/validate/codes.py +93 -0
  133. focus_data_toolkit-0.11.0/src/focus_data_toolkit/validate/corrections.py +245 -0
  134. focus_data_toolkit-0.11.0/src/focus_data_toolkit/validate/reconciliation.py +98 -0
  135. focus_data_toolkit-0.11.0/src/focus_data_toolkit/validate/referential.py +289 -0
  136. focus_data_toolkit-0.11.0/src/focus_data_toolkit.egg-info/PKG-INFO +519 -0
  137. focus_data_toolkit-0.11.0/src/focus_data_toolkit.egg-info/SOURCES.txt +224 -0
  138. focus_data_toolkit-0.11.0/src/focus_data_toolkit.egg-info/dependency_links.txt +1 -0
  139. focus_data_toolkit-0.11.0/src/focus_data_toolkit.egg-info/entry_points.txt +2 -0
  140. focus_data_toolkit-0.11.0/src/focus_data_toolkit.egg-info/requires.txt +38 -0
  141. focus_data_toolkit-0.11.0/src/focus_data_toolkit.egg-info/top_level.txt +1 -0
  142. focus_data_toolkit-0.11.0/tests/conftest.py +76 -0
  143. focus_data_toolkit-0.11.0/tests/fixtures/client_like/SOURCES.md +9 -0
  144. focus_data_toolkit-0.11.0/tests/fixtures/client_like/consolidated_multi_provider_1_3.csv +5 -0
  145. focus_data_toolkit-0.11.0/tests/fixtures/golden/README.md +34 -0
  146. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/aws_1_2_cost_and_usage_rows100_seed42.csv +101 -0
  147. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/aws_1_2_cost_and_usage_rows25_seed7.csv +26 -0
  148. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/aws_1_2_cost_and_usage_rows25_seed7_credits.csv +26 -0
  149. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/aws_1_3_contract_commitment_rows100_seed42.csv +7 -0
  150. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/aws_1_3_cost_and_usage_rows100_seed42.csv +101 -0
  151. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/aws_1_3_cost_and_usage_rows25_seed7.csv +26 -0
  152. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/aws_1_3_cost_and_usage_rows25_seed7_credits.csv +26 -0
  153. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/azure_1_2_cost_and_usage_rows100_seed42.csv +101 -0
  154. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/azure_1_2_cost_and_usage_rows25_seed7.csv +26 -0
  155. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/azure_1_2_cost_and_usage_rows25_seed7_credits.csv +26 -0
  156. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/azure_1_3_contract_commitment_rows100_seed42.csv +9 -0
  157. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/azure_1_3_cost_and_usage_rows100_seed42.csv +101 -0
  158. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/azure_1_3_cost_and_usage_rows25_seed7.csv +26 -0
  159. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/azure_1_3_cost_and_usage_rows25_seed7_credits.csv +26 -0
  160. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/gcp_1_2_cost_and_usage_rows100_seed42.csv +101 -0
  161. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/gcp_1_2_cost_and_usage_rows25_seed7.csv +26 -0
  162. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/gcp_1_2_cost_and_usage_rows25_seed7_credits.csv +26 -0
  163. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/gcp_1_3_contract_commitment_rows100_seed42.csv +9 -0
  164. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/gcp_1_3_cost_and_usage_rows100_seed42.csv +101 -0
  165. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/gcp_1_3_cost_and_usage_rows25_seed7.csv +26 -0
  166. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/gcp_1_3_cost_and_usage_rows25_seed7_credits.csv +26 -0
  167. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/scenarios_correction_set.csv +4 -0
  168. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/scenarios_sca_equal.csv +4 -0
  169. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/scenarios_sca_negative.csv +3 -0
  170. focus_data_toolkit-0.11.0/tests/fixtures/golden/compatibility_golden/scenarios_sca_weighted.csv +4 -0
  171. focus_data_toolkit-0.11.0/tests/fixtures/official/SOURCES.md +12 -0
  172. focus_data_toolkit-0.11.0/tests/fixtures/official/invoice_detail_grain_example.json +5 -0
  173. focus_data_toolkit-0.11.0/tests/fixtures/official/numeric_format_examples.json +6 -0
  174. focus_data_toolkit-0.11.0/tests/test_adapters_aws.py +179 -0
  175. focus_data_toolkit-0.11.0/tests/test_adapters_azure_gcp.py +142 -0
  176. focus_data_toolkit-0.11.0/tests/test_atomic_write.py +192 -0
  177. focus_data_toolkit-0.11.0/tests/test_billing_lifecycle.py +101 -0
  178. focus_data_toolkit-0.11.0/tests/test_bundle_gate.py +226 -0
  179. focus_data_toolkit-0.11.0/tests/test_bundle_validation.py +288 -0
  180. focus_data_toolkit-0.11.0/tests/test_capabilities.py +88 -0
  181. focus_data_toolkit-0.11.0/tests/test_cli.py +374 -0
  182. focus_data_toolkit-0.11.0/tests/test_client_like.py +52 -0
  183. focus_data_toolkit-0.11.0/tests/test_container.py +58 -0
  184. focus_data_toolkit-0.11.0/tests/test_contract_applied.py +169 -0
  185. focus_data_toolkit-0.11.0/tests/test_contract_commitment_semantics.py +98 -0
  186. focus_data_toolkit-0.11.0/tests/test_convert_roundtrip.py +130 -0
  187. focus_data_toolkit-0.11.0/tests/test_corrections.py +82 -0
  188. focus_data_toolkit-0.11.0/tests/test_cross_provider.py +94 -0
  189. focus_data_toolkit-0.11.0/tests/test_detect.py +21 -0
  190. focus_data_toolkit-0.11.0/tests/test_fake_provider.py +107 -0
  191. focus_data_toolkit-0.11.0/tests/test_fixtures_are_synthetic.py +100 -0
  192. focus_data_toolkit-0.11.0/tests/test_focus_json.py +35 -0
  193. focus_data_toolkit-0.11.0/tests/test_gaps.py +129 -0
  194. focus_data_toolkit-0.11.0/tests/test_generator_golden.py +80 -0
  195. focus_data_toolkit-0.11.0/tests/test_generators.py +72 -0
  196. focus_data_toolkit-0.11.0/tests/test_grouping_keys.py +60 -0
  197. focus_data_toolkit-0.11.0/tests/test_lifecycle_chains.py +141 -0
  198. focus_data_toolkit-0.11.0/tests/test_lineage_counters.py +76 -0
  199. focus_data_toolkit-0.11.0/tests/test_lint_focus.py +183 -0
  200. focus_data_toolkit-0.11.0/tests/test_manifest.py +109 -0
  201. focus_data_toolkit-0.11.0/tests/test_model_provenance.py +163 -0
  202. focus_data_toolkit-0.11.0/tests/test_multi_provider.py +142 -0
  203. focus_data_toolkit-0.11.0/tests/test_official_json_schemas.py +210 -0
  204. focus_data_toolkit-0.11.0/tests/test_packaging.py +139 -0
  205. focus_data_toolkit-0.11.0/tests/test_parquet.py +207 -0
  206. focus_data_toolkit-0.11.0/tests/test_parquet_input.py +166 -0
  207. focus_data_toolkit-0.11.0/tests/test_participant_entities.py +76 -0
  208. focus_data_toolkit-0.11.0/tests/test_partitioning.py +267 -0
  209. focus_data_toolkit-0.11.0/tests/test_progress_cancel.py +166 -0
  210. focus_data_toolkit-0.11.0/tests/test_release_tooling.py +182 -0
  211. focus_data_toolkit-0.11.0/tests/test_replace_recovery.py +224 -0
  212. focus_data_toolkit-0.11.0/tests/test_runtime.py +145 -0
  213. focus_data_toolkit-0.11.0/tests/test_schema_detection.py +201 -0
  214. focus_data_toolkit-0.11.0/tests/test_single_source_focus_rules.py +59 -0
  215. focus_data_toolkit-0.11.0/tests/test_spill.py +74 -0
  216. focus_data_toolkit-0.11.0/tests/test_split_allocation.py +171 -0
  217. focus_data_toolkit-0.11.0/tests/test_streaming.py +151 -0
  218. focus_data_toolkit-0.11.0/tests/test_strict_conversion.py +57 -0
  219. focus_data_toolkit-0.11.0/tests/test_studio.py +391 -0
  220. focus_data_toolkit-0.11.0/tests/test_supplement_apply.py +260 -0
  221. focus_data_toolkit-0.11.0/tests/test_supplement_loader.py +306 -0
  222. focus_data_toolkit-0.11.0/tests/test_supplement_streaming.py +151 -0
  223. focus_data_toolkit-0.11.0/tests/test_synthetic_conversion.py +75 -0
  224. focus_data_toolkit-0.11.0/tests/test_validator.py +40 -0
  225. focus_data_toolkit-0.11.0/tests/test_workflow_pins.py +51 -0
  226. focus_data_toolkit-0.11.0/tools/extract_focus_1_4_model.py +133 -0
@@ -0,0 +1,558 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented here. The format is based on
4
+ [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project
5
+ adheres to [Semantic Versioning](https://semver.org/) with
6
+ [PEP 440](https://peps.python.org/pep-0440/) version strings. See
7
+ [docs/versioning.md](docs/versioning.md) for the versioning and reproducibility
8
+ policy.
9
+
10
+ ## [Unreleased]
11
+
12
+ ## [0.11.0] — 2026-07-18
13
+
14
+ First **stable** release. Same feature set as `0.11.0rc1`, promoted to a final release now that the
15
+ embedded FOCUS 1.4 model's **provenance is `complete`** — the FinOps Foundation source workbook is
16
+ hashed (`source.artifact_sha256`) and the committed model was reproduced from it **byte-for-byte** by
17
+ the pinned extractor — so it installs with a plain `pip install focus-data-toolkit` (no `--pre`).
18
+
19
+ Highlights (full details in the `0.11.0rc1` entry below):
20
+
21
+ - **Studio** — a local web UI (`focus-toolkit ui`, extra `[studio]`) over the same Core.
22
+ - **Runner** — a containerised batch image (GHCR) whose entrypoint is the `focus-toolkit` CLI.
23
+ - **Core (Lot A)** — progress/cancellation, separate work/output disk budgets, `--exit-policy`, and
24
+ the `detect` / `validate-bundle` / `version` commands.
25
+
26
+ ### Changed
27
+
28
+ - **Model provenance**: `partial` → `complete` (source workbook hashed and end-to-end reproduction
29
+ verified; see [docs/model-provenance.md](docs/model-provenance.md)).
30
+ - **Release workflow (`release.yml`)**: the GitHub Release step is idempotent — if a release for the
31
+ tag already exists (e.g. created via the "Draft a new release" UI), it updates that release in
32
+ place and attaches the attested assets instead of failing.
33
+
34
+ ## [0.11.0rc1] — 2026-07-18
35
+
36
+ First release candidate published to PyPI — a **pre-release** (marked as such per the honesty gate,
37
+ because the embedded FOCUS model's provenance is `partial`; see
38
+ [docs/model-provenance.md](docs/model-provenance.md)). It bundles the three deployment access
39
+ methods on the single Core: the CLI/SDK, the containerised **Runner** (Lot B) and the local
40
+ **Studio** web UI (Lot C), plus the Core progress/cancellation, disk-budget and CLI additions
41
+ (Lot A). Install with `pip install --pre focus-data-toolkit` (pre-releases are not selected by a
42
+ plain `pip install`).
43
+
44
+ ### Added — Studio: local web UI (deployment Lot C)
45
+
46
+ - **`focus-toolkit ui`** launches a local web app (FastAPI, behind the optional `[studio]` extra;
47
+ the command imports it lazily so a core install is unaffected) over the **same Core** — every
48
+ operation drives the same SDK the CLI/Runner use, so its manifests, diagnostics and checksums are
49
+ identical. Detect a source, pick a file under `--root` / upload (capped) / generate synthetic
50
+ data, convert (strict|synthetic, CSV|Parquet) with **live per-phase progress and cancel**,
51
+ preview a **sampled** page (the full file is never loaded), and download datasets, manifest,
52
+ diagnostics (JSON/CSV), `SHA256SUMS` and an HTML summary.
53
+ - **Security:** binds `127.0.0.1` by default (a non-loopback `--host` is refused without
54
+ `--allow-remote`); a fresh per-start token is required on every API call; `Host`/`Origin` are
55
+ validated (anti DNS-rebinding / CSRF); file access is confined to the allowlisted `--root`;
56
+ uploads stream to disk and are size-capped. No telemetry, no external upload.
57
+ - **Path confinement:** `resolve_within_root` walks real directory entries (matching each component
58
+ by name, never concatenating the user string into a path) **and** canonicalises every matched
59
+ entry with `Path.resolve` — a symlink, Windows junction or reparse point whose real target
60
+ escapes `--root` is refused, while a link that stays inside is followed; absolute, drive-relative
61
+ and UNC paths and `..` traversal are rejected.
62
+ - **Bounded by design:** one conversion at a time by default (extra submissions queue); per-job
63
+ scratch under a work dir with TTL + startup cleanup; generation is row-capped in the UI (use the
64
+ CLI/Runner for very large synthetic sets). New extras `studio` and `studio-all`; see
65
+ [docs/studio.md](docs/studio.md).
66
+
67
+ ### Added — Runner: containerised batch image (deployment Lot B)
68
+
69
+ - **OCI image** (`Dockerfile`) whose entrypoint **is** the `focus-toolkit` CLI — a container run
70
+ equals a CLI run (same manifests, diagnostics, checksums, exit codes; no FOCUS logic
71
+ duplicated). Batch-only (no HTTP server). Multi-stage build on a **digest-pinned**
72
+ `python:3.12-slim-bookworm`, bundling the `[parquet]` extra; **non-root** (uid 65532),
73
+ **read-only-rootfs compatible** (only `/work` and `/output` written), `FOCUS_TOOLKIT_WORK_DIR=/work`.
74
+ Exec-form entrypoint so `docker stop` (SIGTERM) cancels cleanly (exit 130, nothing partial
75
+ published). Volumes: `/input` (ro), `/output`, `/work`. See [docs/runner.md](docs/runner.md).
76
+ - **Container CI** (`.github/workflows/container.yml`): builds the image on every PR / push to
77
+ `main` (no publish) and runs `docker run` smoke tests — non-root uid, read-only-rootfs streaming
78
+ Parquet convert, exit codes, SIGTERM handling — plus a trivy scan (fails on HIGH/CRITICAL). A
79
+ fast static test (`tests/test_container.py`) enforces the base-image digest pin, non-root user
80
+ and exec-form entrypoint.
81
+ - **Container release** (`.github/workflows/release-container.yml`): on a `v*` tag, runs the same
82
+ release gates as the PyPI flow (tag matches `__version__`; provenance-honesty gate), builds a
83
+ candidate image, **scans it (trivy) before any public tag is assigned**, then publishes to
84
+ `ghcr.io/guymano/focus-data-toolkit` — **immutable** `<version>` and `sha-<full-commit>` tags
85
+ plus a **rolling** `<major>.<minor>` alias (PEP 440 tag parsing; no `latest`) — generates a
86
+ CycloneDX SBOM, attests build provenance and **signs with cosign** (keyless OIDC), in a
87
+ reviewer-gated `ghcr` environment. All actions pinned by commit SHA.
88
+
89
+ ### Added — progress, cancellation, disk budgets & pipeline ergonomics (deployment Lot A)
90
+
91
+ - **Progress reporting**: the streaming engine (`convert_files`) accepts an optional
92
+ `progress` callback receiving throttled `ProgressEvent`s per phase (`READING`,
93
+ `TRANSFORMING`, `AGGREGATING`, `WRITING`, `VALIDATING`, `PUBLISHING`) with a completed
94
+ count, an optional total, a unit (`rows`/`bytes`) and a message — derived without
95
+ materialising data (CSV byte cursor / Parquet footer row count). `focus-toolkit convert
96
+ --progress` renders a single throttled status line on stderr. All hooks are opt-in and
97
+ keyword-only; output is byte-identical with or without them.
98
+ - **Cooperative cancellation**: `convert_files(..., cancel=...)` checks a predicate between
99
+ rows and validation passes and raises `ConversionCancelled` — the atomic staging directory
100
+ is removed, so nothing partial is ever published. The CLI maps SIGINT/SIGTERM to a clean
101
+ cancel (exit code **130**), so `Ctrl-C` and `docker stop` unwind cleanly instead of dying
102
+ mid-write.
103
+ - **Separate disk budgets** (`focus_data_toolkit.runtime`): the scratch filesystem and the
104
+ output filesystem are budgeted independently via `FOCUS_TOOLKIT_WORK_DIR`,
105
+ `FOCUS_TOOLKIT_MAX_WORK_BYTES`, `FOCUS_TOOLKIT_MIN_WORK_FREE_BYTES` and
106
+ `FOCUS_TOOLKIT_MIN_OUTPUT_FREE_BYTES` (`FOCUS_TOOLKIT_LOG_LEVEL` too). A best-effort
107
+ pre-flight (estimate with a safety margin) plus periodic in-run checks fail fast with a
108
+ structured `FDT-IO-005` (output) / `FDT-IO-006` (work / temp budget) diagnostic and CLI
109
+ exit code **5**, instead of a raw `OSError` mid-run. `WORK_DIR` relocates the SQLite
110
+ aggregation + bundle-spill scratch off the output disk (business artifacts stay
111
+ byte-identical; scratch is always cleaned up).
112
+ - **Pipeline-friendly exit codes**: `focus-toolkit convert --exit-policy pipeline` maps the
113
+ functional-but-complete outcomes (3 = strict incomplete, 4 = synthetic assumptions) to 0,
114
+ so orchestrators (Kubernetes / Airflow / Jenkins / AWS Batch) don't flag a legitimate run
115
+ failed. The default `detailed` policy keeps the historic codes; full status stays in the
116
+ manifest and `_run.json`.
117
+ - **New CLI commands**: `focus-toolkit detect` (dataset/version of a file header, text/JSON),
118
+ `focus-toolkit validate-bundle` (cross-dataset validation gate over explicit per-dataset
119
+ files or an auto-detected `--directory`; ambiguous combinations refused), and
120
+ `focus-toolkit version`. New SDK exports: `ProgressEvent`, `ConversionCancelled`.
121
+
122
+ ### Added — provider-native supplement adapters
123
+
124
+ - **Adapters** translate documented cloud-provider export formats into the
125
+ canonical supplement kinds automatically, so a client passes native exports
126
+ straight to `convert` / `supplements validate` without renaming anything.
127
+ First adapters (AWS): `aws-invoice-summary` (Invoicing API `InvoiceSummary`,
128
+ incl. nested `Entity.InvoicingEntity`, `DueDate`, `PurchaseOrderNumber`) →
129
+ `invoice`; `aws-savings-plans` (Savings Plans inventory: `paymentOption`,
130
+ `state`, `start`) → `contract_commitment`. Each adapter is a vendored,
131
+ versioned JSON mapping table with official-doc provenance
132
+ (`supplement/adapters/adapters_provenance.json`, sha256-verified); the format
133
+ is auto-detected from the header (force with `FILE:<adapter-name>`).
134
+ Translated rows flow through the unchanged supplement validation and carry
135
+ `ENRICHED` lineage attributed as `supplement:<adapter>@<version>:<file>`. An
136
+ adapter only maps fields its table describes (residual gaps are reported, not
137
+ guessed); an unrecognized export falls back to the generic FOCUS-named path.
138
+ New command: `fdt supplements adapters`. Adapters ship for **AWS**
139
+ (`aws-invoice-summary`, `aws-savings-plans`), **Azure** (`azure-invoice` —
140
+ Billing Invoices REST API; `InvoiceStatus` Due/OverDue/Paid → `Issued`,
141
+ Void → `Voided`) and **GCP** (`gcp-compute-commitments` — Compute Engine
142
+ `regionCommitments`; `status` and CUD payment facts → `contract_commitment`).
143
+
144
+ ### Added — supplemental client data (promise #3)
145
+
146
+ - **Gap analysis** (`fdt gaps`): reports, per FOCUS 1.4 dataset, exactly which
147
+ columns block strict production for a given 1.2/1.3 source — computed from
148
+ the converter's own provenance rules and annotated from the embedded model —
149
+ plus ready-to-fill CSV templates per supplement kind. Missing mandatory
150
+ source columns are reported as source-completeness gaps.
151
+ - **Supplement bundles**: clients supply the missing provider-issued facts as
152
+ sidecar files (CSV/JSON, gzip ok; kinds `billing_period`, `invoice`,
153
+ `invoice_line`, `contract_commitment`; kind auto-detected from the header or
154
+ forced with `FILE:KIND`). Supplements are validated against the source and
155
+ the model before any use (`FDT-SUPP-0xx`: duplicate keys, unknown columns,
156
+ format/allowed-value violations, orphans, `BilledCost` reconciliation
157
+ conflicts, per-column coverage). Pre-flight command:
158
+ `fdt supplements validate`.
159
+ - **ENRICHED conversion**: `convert_to_focus_1_4(..., supplements=...)`
160
+ applies supplied facts with `ENRICHED` lineage and full attribution
161
+ (`supplement:<kind>:<file>` + sha256 in the new manifest `supplements`
162
+ section). At full coverage, **strict mode now produces all four FOCUS 1.4
163
+ datasets** with nothing invented; partial coverage keeps the dataset
164
+ `NOT_PRODUCED` with per-value counters showing how close it is. In strict
165
+ mode uncovered nullable assumed columns are emitted empty (synthetic
166
+ defaults never leak); real issuer-assigned `InvoiceDetailId`s replace the
167
+ locally generated back-links.
168
+
169
+ ### Added — capability profiles
170
+
171
+ - New `CapabilityProfile` (`focus_data_toolkit.model.capabilities`): an
172
+ explicit, validated declaration of the FOCUS applicability conditions a
173
+ source supports (`SupportsUnitPricing`,
174
+ `SupportsMultiplePricingCategories`). The linter enforces
175
+ conditionally-required columns only for declared conditions; the conversion
176
+ pipeline records the active profile in the manifest (`capability_profile`),
177
+ so an unevaluated condition set is visible instead of silent. CLI:
178
+ repeatable `--supports CONDITION` on `convert` and `validate`.
179
+
180
+ ### Added — per-value lineage counters
181
+
182
+ - The manifest's produced-dataset entries gain a `lineage_summary` section
183
+ counting, per column, how many values actually took each lineage. Today it
184
+ covers the pricing-currency backfill pair (`PricingCurrency` /
185
+ `PricingCurrencyEffectiveCost`): the headline column lineage stays the
186
+ conservative `DERIVED`, and the summary shows the real observed/backfilled
187
+ mix (e.g. `{"OBSERVED": 99800, "DERIVED": 200}`). Identical in the eager and
188
+ streaming paths; bounded memory (columns × lineage categories).
189
+
190
+ ### Added — official FOCUS JSON schemas
191
+
192
+ - The four official FOCUS 1.4 JSON object schemas (`ContractApplied`,
193
+ `AllocatedMethodDetails`, `CommitmentProgramEligibilityDetails`,
194
+ `ContractCommitmentApplicability`) are vendored verbatim from the
195
+ specification repository (tag `v1.4`) under
196
+ `focus_data_toolkit/model/json_schemas/`, with a provenance manifest
197
+ (source paths, sha256, CC-BY-4.0 attribution). The linter now evaluates
198
+ every JSON-object column against its official schema — conditional scope
199
+ rules, metric exclusivity, ranges, PascalCase `x_` custom keys — via a
200
+ small dependency-free interpreter of the schema subset; violations surface
201
+ as `official_schema_violation`. Previously only `ContractApplied` was
202
+ deep-validated and `ContractCommitmentApplicability` was only checked to
203
+ be a JSON object.
204
+
205
+ ### Fixed — FOCUS conformance (may change output bytes)
206
+
207
+ - **Synthetic `ContractCommitmentApplicability`**: the object now declares
208
+ `{"IsComplexScope": true, ...}` — the official object schema requires a scope
209
+ representation (`Inclusions` + `InclusionOperator` become required when no
210
+ scope flag is set), so the previous `x_Source`-only object was normatively
211
+ invalid. The value remains `ASSUMED` and still never passes strict mode.
212
+ - **`ContractCommitmentDurationType`**: an unparseable or inverted commitment
213
+ period no longer yields a fabricated `"12 Months"`. The value stays empty
214
+ (not derivable) and the affected rows are reported as an aggregated
215
+ `FDT-CC-001` WARNING; the mandatory-column lint then flags the dataset
216
+ instead of silently publishing an arbitrary duration.
217
+
218
+ - **1.2 participant-entity migration**: `HostProviderName` is no longer derived
219
+ from the deprecated `PublisherName` (the entity that *produced* the service —
220
+ not the infrastructure host). Per the official FOCUS 1.4 `HostProviderName`
221
+ rules, when the source does not expose the underlying host the value MUST
222
+ match `ServiceProviderName`; a 1.2 source never exposes it, so both columns
223
+ now derive from `ProviderName` and carry `DERIVED` lineage (previously
224
+ `RENAMED`) with the spec rule recorded in the manifest. The per-row provider
225
+ context applies the same rule (`host == service` when the host is not
226
+ exposed; the publisher is never used as a fallback host).
227
+
228
+ ### Added — release pipeline
229
+
230
+ - Secure release workflows (`.github/workflows/`): a reusable **build-once**
231
+ workflow (`release-build.yml`), a **`release-dry-run.yml`** (no publish, no
232
+ privileged scopes), a tag-triggered **`release.yml`** that attests
233
+ wheel/sdist/SBOM/checksums (GitHub Artifact Attestations, keyless OIDC) and
234
+ publishes via **PyPI Trusted Publishing** in a gated environment, and a
235
+ **`reproducibility.yml`** double-build check. Artifacts flow between jobs by
236
+ digest — the publish job never rebuilds.
237
+ - A deterministic **CycloneDX 1.5 SBOM** generator (`scripts/generate_sbom.py`)
238
+ that records the embedded FOCUS 1.4 model as a first-class `data` component
239
+ (CC-BY-4.0 + provenance hash), and an offline **release verifier**
240
+ (`scripts/verify_release.py`) checking `SHA256SUMS`, the SBOM, and version
241
+ consistency. Both are covered by `tests/test_release_tooling.py`.
242
+
243
+ ### Changed — dependencies
244
+
245
+ - Widened the `parquet` extra to `pyarrow>=15,<26` (the lock resolves to 25.x)
246
+ and `pytest-cov` to `>=5,<8` (dev). The Parquet suite passes unchanged.
247
+
248
+ ### Security
249
+
250
+ - Resolved **PYSEC-2026-113** by moving the resolved `pyarrow` to `>= 23.0.1`
251
+ (25.x); the `pip-audit` gate now runs with **no `--ignore-vuln` exception**.
252
+
253
+ ## [0.9.0] — 2026-07-17
254
+
255
+ **First public release.** `0.9.0` is the first version prepared for publication
256
+ to PyPI (a deliberate "near-stable" signal; `1.0.0` is reserved for after
257
+ real-world feedback — see [docs/versioning.md](docs/versioning.md)). It bundles
258
+ all functionality developed across the `0.2.0` (P0) and `0.3.0` (P1) milestones,
259
+ detailed in their sections below, and adds the packaging, CI/supply-chain,
260
+ governance, and provenance work that makes the project publishable. Publication
261
+ itself is performed by the release pipeline (see
262
+ [docs/releasing.md](docs/releasing.md)).
263
+
264
+ ### Added — packaging & distribution
265
+
266
+ - **PyPI-ready packaging**: single-sourced version (`focus_data_toolkit._version`),
267
+ PEP 639 SPDX license metadata (`license = "MIT"` + `license-files`), a PEP 561
268
+ `py.typed` marker with the `Typing :: Typed` classifier, project URLs, and a
269
+ `MANIFEST.in` that ships the sources needed to build and verify from an sdist.
270
+ - **Reproducible installs**: a committed `uv.lock` and hash-pinned
271
+ `constraints/*.txt`; the optional `[validator]` extra now resolves
272
+ `focus-validator` from **PyPI** (Python 3.12+) instead of a git URL, so wheels
273
+ and sdists upload and install cleanly. New `[release]` extra (`build`, `twine`).
274
+ - **Packaging tests** build and install a wheel **and** an sdist in a clean
275
+ environment and smoke-test the result.
276
+
277
+ ### Added — CI & supply-chain hardening
278
+
279
+ - Modular workflows: lint, type-check (mypy), a test matrix across
280
+ ubuntu/windows × Python 3.11–3.13, coverage floor, and a packaging job.
281
+ - Least-privilege `permissions: contents: read` with per-job elevation; every
282
+ GitHub Action pinned to a full commit SHA (Docker image actions to an
283
+ `@sha256` digest), enforced by `scripts/check_pinned_actions.py` and a test.
284
+ - Security scanning: `pip-audit` (pinned), `gitleaks`, `actionlint`, and
285
+ `zizmor`; `Dependabot` for actions and Python dependencies; CodeQL and OpenSSF
286
+ Scorecard workflows (gated behind `workflow_dispatch` until repository code
287
+ scanning is enabled — see [docs/releasing.md](docs/releasing.md)).
288
+
289
+ ### Added — governance & documentation
290
+
291
+ - `SECURITY.md` (GitHub Private Vulnerability Reporting; the security-vulnerability
292
+ vs FOCUS-conformance-bug distinction; a supported-versions matrix; a "no client
293
+ data" rule), `CONTRIBUTING.md`, a `NOTICE` file, `.github/CODEOWNERS`, and
294
+ GitHub issue / pull-request templates.
295
+ - Documentation under `docs/`: `versioning.md`, `compatibility.md` (Python/OS/
296
+ FOCUS matrix, including the Windows streaming limitation), `security-model.md`,
297
+ `releasing.md` (with the operational, owner-only checklist), and
298
+ `model-provenance.md`.
299
+ - A test that scans committed fixtures for secrets/PII and requires each fixture
300
+ directory to document its synthetic provenance.
301
+
302
+ ### Added — FOCUS model provenance
303
+
304
+ - A machine-readable provenance manifest
305
+ (`src/focus_data_toolkit/model/model_provenance.json`) with a JSON Schema
306
+ (`schema/model_provenance.schema.json`) and a verifier
307
+ (`scripts/verify_model_provenance.py`, run in CI). It records the source
308
+ (FinOps FOCUS 1.4 Data Model workbook), the **verified CC-BY-4.0** license, the
309
+ extraction process, and the reproducible output hash. Status is `partial`
310
+ (the source workbook is not redistributed/hashed here); a `partial` → `complete`
311
+ gate blocks a fully-reproducible-provenance claim until the source is hashed.
312
+
313
+ ### Changed
314
+
315
+ - Project version set to **0.9.0** (first public release).
316
+
317
+ ### Fixed
318
+
319
+ - The committed FOCUS 1.4 model JSON `source` field pointed at a non-existent
320
+ `docs/focus/…xlsx` path, diverging from what `tools/extract_focus_1_4_model.py`
321
+ emits. It now matches the extractor's output, so re-running the extractor on
322
+ the same workbook reproduces the committed JSON byte-for-byte.
323
+
324
+ ## [0.3.0] — pre-release development (P1)
325
+
326
+ The 0.3.0 line ("P1") makes the toolkit reliable on **real client data** — consolidated,
327
+ multi-provider, multi-issuer, multi-currency, volumetric exports — without re-implementing the
328
+ 0.2.0 (P0) guardrails. It lands in two phases: **Phase A** (correctness & integrity) and
329
+ **Phase B** (scale & realism), both described below.
330
+
331
+ ### Added — schema detection
332
+
333
+ - **`focus_data_toolkit.schema.detect_focus_schema(headers)`** identifies the FOCUS
334
+ **dataset** (Cost and Usage / Contract Commitment / Billing Period / Invoice Detail) and
335
+ **version** (1.2 / 1.3 / 1.4) with a `confidence`, `exact_match`, and the exact
336
+ `missing` / `additional_focus` / `extension` / `unknown` columns and `ambiguous_candidates`.
337
+ It scores the header against a registry of normative column sets (`schema/registry.py`)
338
+ computed from the committed 1.4 model plus a removed-columns table, so a 1.4 file is never
339
+ taken for 1.3, a 1.3 export missing an optional column is not taken for 1.2, `x_` columns
340
+ never count against a match, and unknown non-`x_` columns are surfaced.
341
+ - CLI `convert --source-version` / `--source-dataset` force detection; **strict mode refuses
342
+ an ambiguous or low-confidence source** (clear error). The detection decision is recorded
343
+ in the manifest. `detect_focus_version` is retained as a compatibility wrapper.
344
+
345
+ ### Added — multi-provider context & correct grouping
346
+
347
+ - **Per-row context** (`focus_data_toolkit.context`): `BillingContext` and `ProviderContext`
348
+ are derived from the whole source, never from the first row. The first-row `_provider_context`
349
+ is gone; billing-period issuer is no longer back-filled from row 0. A bounded per-row
350
+ **context summary** (distinct providers / issuers / accounts / currencies / periods and
351
+ `multi_*` flags) is recorded in the manifest; ambiguous contexts are reported as diagnostics.
352
+ - **Invoice Detail now groups on the full business grain** — `(InvoiceIssuerName, InvoiceId,
353
+ BillingAccountId, BillingCurrency, BillingPeriodStart, BillingPeriodEnd, ChargeCategory)` —
354
+ preventing collisions across issuers/accounts/currencies/periods. `InvoiceDetailId` is a
355
+ clearly-local, versioned, collision-safe id (`x_fdt_idl_v1_<hash>` over a JSON-encoded key),
356
+ **never presented as issuer-assigned**.
357
+
358
+ *Because the `InvoiceDetailId` value scheme changed, synthetic Invoice Detail (and the Cost
359
+ and Usage back-link) have a new byte baseline as of 0.3.0. Conversion remains deterministic.*
360
+
361
+ ### Added — cross-dataset (bundle) validation
362
+
363
+ - **`validate_dataset_bundle(bundle)`** (alias `validate_bundle`) validates a bundle of
364
+ datasets **against each other** — deliberately separate from the per-dataset linter, which
365
+ never asserts cross-dataset validity. It reports referential integrity (uniqueness, FKs,
366
+ orphans), issuer/account/currency/period coherence, Cost-and-Usage ↔ Invoice-Detail
367
+ reconciliation (only when Invoice Detail is authoritative, with an explicit rounding
368
+ tolerance), **Split Cost Allocation** (ratios sum to 1, allocated costs sum to the origin,
369
+ ratio in [0,1], consistent method/unit, unique resources, incomplete groups flagged), and
370
+ commitment lifecycle (period/percentage) checks. Findings are grouped by severity
371
+ (error / warning / info / not-executable / not-applicable) and serialise to JSON.
372
+
373
+ ### Added — structured diagnostics & error catalog
374
+
375
+ - **`focus_data_toolkit.errors.Diagnostic`** carries a stable code, severity, business record
376
+ keys, column, expected/actual, suggestion, provenance and group/join context, and renders to
377
+ JSON, a readable console block, and CSV rows. A stable **`FDT-*` code catalog**
378
+ (`validate/codes.py`) replaces opaque "invalid input" messages.
379
+
380
+ ### Added — atomic writes
381
+
382
+ - **Results appear in the destination only after everything succeeds.** `write_result` stages
383
+ datasets in a temp directory on the same filesystem, enforces the mandatory lint gate (a
384
+ lint-failing result is never published), then writes the deterministic business manifest, an
385
+ operational `_run.json` sidecar (run id / timestamp / per-file SHA-256, kept **out** of the
386
+ business manifest so dataset bytes stay reproducible) and `SHA256SUMS` last, and publishes
387
+ with a single atomic rename. `io/atomic_writer.py` exposes `AtomicOutputDir` and
388
+ `OnExists` (**refuse** default / replace via crash-safe swap-dir / version). CLI:
389
+ `--on-exists`, `--keep-temp`.
390
+
391
+ ### Added — streaming conversion (Phase B)
392
+
393
+ - **`focus_data_toolkit.convert.convert_files(cost_and_usage, out_dir, …)`** streams the Cost
394
+ and Usage file **once** and stages Invoice Detail aggregation / Billing Period dedup in a
395
+ throwaway SQLite database (`storage/external_index.py`), so **peak memory is flat regardless
396
+ of row count** — a constant ~64 MB peak process RSS from 50k to 300k rows (6× the rows,
397
+ ×1.05 the memory; `tools/benchmark_streaming.py`), and ~38 MB of traced Python allocations
398
+ that a `slow` test asserts do not scale. Costs are summed with Python `Decimal` over an
399
+ `ORDER BY` (BINARY) scan — never SQL `SUM()` — so the streamed totals are bit-for-bit the
400
+ eager ones.
401
+ - **Byte-identical to the in-memory path**: both call the same pure per-row / per-group
402
+ functions (`convert_cost_and_usage_row`, `invoice_detail_row`, `billing_period_row`) and the
403
+ same `assemble_manifest`, so equivalence holds by construction (asserted on the datasets,
404
+ the manifest, and `SHA256SUMS`). A new streaming reader/writer layer (`io/records.py`,
405
+ `io/csv_io.py`) auto-detects gzip input and rejects wrong-field-count rows with a
406
+ line-numbered error. CLI: `convert --stream`.
407
+
408
+ ### Added — Parquet output (Phase B)
409
+
410
+ - **`--output-format parquet`** / `convert_files(…, output_format="parquet")` writes the
411
+ datasets as Parquet with **exact decimal128** (`io/parquet_io.py`): decimal columns use
412
+ `decimal128(precision, scale)` from a committed, reviewed scale registry
413
+ (`model/focus_1_4_decimal_scale.json`) — never binary float, and a value needing more scale
414
+ than the column allows raises with its line number instead of rounding silently. Dates are
415
+ UTC timestamps, JSON/strings verbatim, empty string ↔ null. Exactness contract: **CSV is
416
+ byte-exact, Parquet is decimal-value-exact**. PyArrow stays an optional `[parquet]` extra
417
+ with a clear install hint; the core stays standard-library only.
418
+ - **Partitioning & compression**: `--partition-by` writes the Cost and Usage dataset as a
419
+ Hive-partitioned Parquet tree (`COL=value/…/part-N.parquet`) on low-cardinality String /
420
+ Date-Time columns, keeping memory bounded to one open writer per partition; a high-cardinality
421
+ key is warned about (`FDT-IO-004`) and, past a hard cap, refused (nothing partial is
422
+ published). Partition columns live in the paths (standard Hive), so any `pyarrow.dataset`
423
+ reader — and the toolkit's own read-back lint gate — reconstruct full rows, with the values
424
+ round-tripping exactly (reconstructed as strings). `--compression`
425
+ (snappy default / zstd / gzip / none) and `--target-file-size` (approximate part-file roll)
426
+ tune the layout. The atomic writer now checksums files by path relative to the output root, so
427
+ every partition part appears in `SHA256SUMS`.
428
+
429
+ ### Added — synthetic scenarios & lifecycle (Phase B)
430
+
431
+ - **`generators/scenarios.py`**: deterministic, provider-agnostic scenario builders —
432
+ `split_allocation_group` (ratios sum to exactly 1, allocated costs sum to exactly the origin,
433
+ last consumer absorbs the residue), `correction_set` (original charge plus signed
434
+ `ChargeClass="Correction"` lines that reference it and record the running auditable net in
435
+ `x_NetCharge`; the original is never overwritten), and `billing_lifecycle_instances` (a
436
+ T0→T6 dataset-instance sequence with only allowed status transitions).
437
+ - **`focus_data_toolkit.lifecycle`**: a `DatasetInstance` snapshot structure and
438
+ status-transition checks (`FDT-CORR-004`) from the FOCUS `InvoiceIssueStatus`
439
+ (Open→Issued→Voided) and `BillingPeriodStatus` (Open→Closed) value sets — flagging un-void,
440
+ un-issue and silent reopen, scoped per subject.
441
+ - **New correction checks** in `validate_dataset_bundle`: net-sum reconciliation to the
442
+ declared `x_NetCharge` (`FDT-CORR-002`) and duplicate-`x_ChargeKey` overwrite detection
443
+ (`FDT-CORR-003`).
444
+
445
+ ### Fixed — streaming publish integrity (Phase B)
446
+
447
+ - The streaming path wrote datasets through direct file handles (for bounded memory), which
448
+ bypassed the atomic writer's file registration — so `SHA256SUMS` listed only the manifest and
449
+ the scratch SQLite database was published into the output directory. Produced files are now
450
+ fsync'd and enrolled for checksums, and the scratch DB is deleted before commit; `SHA256SUMS`
451
+ is complete and byte-identical to the eager path. (Regression tests added.)
452
+
453
+ ### Changed
454
+
455
+ - Version 0.3.0 (P1). `focus-data-toolkit[parquet]` / `[scale]` / `[all]` optional extras
456
+ (PyArrow, used for Parquet); the runtime core remains standard-library only. `pyarrow` is
457
+ added to the `dev` extra so CI runs the Parquet suite rather than skipping it. A `slow`
458
+ pytest marker is added and excluded from the default run. New public API surfaced at the
459
+ package root: `convert_files`, `DatasetInstance`, `check_status_transitions`. Reproducible
460
+ benchmark at `tools/benchmark_streaming.py`.
461
+
462
+ ## [0.2.0] — pre-release development (P0)
463
+
464
+ The 0.2.0 line makes the toolkit honest about what it produces and fixes real FOCUS
465
+ conformance defects.
466
+
467
+ ### Added — modes, provenance & manifest
468
+
469
+ - **Conversion modes** (`--mode`, `focus_data_toolkit.modes.Mode`):
470
+ - `strict` (**new default**) never invents provider-issued financial facts. A
471
+ canonical FOCUS 1.4 dataset is produced only when every Mandatory non-nullable
472
+ column has a factual lineage. From a Cost-and-Usage source, only Cost and Usage is
473
+ produced; Billing Period, Invoice Detail and the 1.4-expanded Contract Commitment are
474
+ reported `NOT_PRODUCED` with their blocking columns.
475
+ - `synthetic` generates assumed values for demos/tests; affected datasets are written
476
+ with a `synthetic_` filename prefix and marked `PRODUCED_SYNTHETIC`, never fully
477
+ conformant.
478
+ - **Value provenance / lineage** (`focus_data_toolkit.provenance`): every produced column
479
+ is classified `OBSERVED` / `RENAMED` / `DERIVED` / `ENRICHED` / `ASSUMED` /
480
+ `UNAVAILABLE`. Any `ASSUMED` column prevents a full-conformance claim.
481
+ - **Deterministic conversion manifest** (`focus_1_4_manifest.json`,
482
+ `focus_data_toolkit.manifest`): per-dataset status/conformance/reason and per-column
483
+ lineage. Written on every conversion; `--manifest PATH` writes an extra copy.
484
+ - **CLI exit codes**: `0` success without assumptions · `1` lint violation ·
485
+ `2` invalid arguments · `3` incomplete strict result · `4` synthetic result with
486
+ assumptions.
487
+
488
+ ### Changed — behaviour
489
+
490
+ - **Default conversion behaviour changed**: strict mode no longer emits Billing Period /
491
+ Invoice Detail / Contract Commitment reconstructed from Cost and Usage alone. Use
492
+ `--mode synthetic` for the previous all-four output (now correctly labelled).
493
+ - `convert_to_focus_1_4` gained a `mode` parameter; `ConversionResult` gained
494
+ `mode`, `provenance`, `manifest`, `not_produced`, `assumptions_present`.
495
+ - README rewritten to separate schema migration / enrichment / synthetic projection.
496
+ - In synthetic mode the Cost and Usage `InvoiceDetailId` back-link (a locally generated id)
497
+ is lineage `ASSUMED`, so synthetic Cost and Usage is labelled `PRODUCED_SYNTHETIC` and
498
+ `synthetic_`-prefixed; in strict mode it stays null. `PricingCurrency` /
499
+ `PricingCurrencyEffectiveCost` are lineage `DERIVED` (source value; nulls backfilled), not
500
+ `OBSERVED`.
501
+ - Manifest conformance for a factual dataset is set **after** the lint runs:
502
+ `STRUCTURAL_LINT` (passed), `LINT_FAILED` (failed), or `NOT_VALIDATED` (`--no-validate`).
503
+ - A derivable dataset whose source yields no rows (e.g. no `InvoiceId`, or an empty Contract
504
+ Commitment source) is reported `NOT_PRODUCED` instead of writing a headerless empty file.
505
+
506
+ ### Fixed — FOCUS conformance of generated JSON columns
507
+
508
+ - **`ContractApplied`** now uses the spec-correct FOCUS 1.3 object schema: a top-level
509
+ `Elements` array with keys `ContractID`, `ContractCommitmentID`,
510
+ `ContractCommitmentAppliedCost`, `ContractCommitmentAppliedQuantity`,
511
+ `ContractCommitmentAppliedUnit` (previously the wrong `ContractId`/`AppliedCost`… names
512
+ and casing). Applied cost/quantity are emitted as JSON **numbers**.
513
+ - **`AllocatedMethodDetails`** emits `AllocatedRatio` and `UsageQuantity` as JSON numbers
514
+ (were quoted strings).
515
+ - **`SkuPriceDetails`** treats `StorageClass` and `Redundancy` as FOCUS-defined keys
516
+ (unprefixed), completing the 13-key FOCUS-defined set (were incorrectly `x_`-prefixed).
517
+ - **`InvoiceDetailGrain`** (derived Invoice Detail) uses `x_`-prefixed custom keys, as
518
+ required for non-FOCUS-defined Key-Value keys.
519
+
520
+ *Because generated data changed, the byte-for-byte output of a given `(rows, seed)` has a
521
+ new baseline as of 0.2.0. Generation remains deterministic.*
522
+
523
+ ### Fixed — linter
524
+
525
+ - **Scientific notation** is now accepted per FOCUS `NumericFormat` (E-notation `mEn`,
526
+ negative-only exponent sign; `35.2E-7` valid, `35.2E+7` invalid). The old regex rejected
527
+ all scientific notation. Numeric literals reject leading zeros (`01`), which are invalid
528
+ JSON numbers.
529
+ - The **`x_` custom-key rule** is enforced across JSON columns via a FOCUS-defined key
530
+ registry (`SkuPriceDetails`, `InvoiceDetailGrain`, `ContractApplied`,
531
+ `AllocatedMethodDetails`, `CommitmentProgramEligibilityDetails`), covering both element
532
+ keys **and top-level custom keys** of array-of-objects columns, and is **not** applied to
533
+ `Tags`/`AllocatedTags` (arbitrary tag keys are allowed).
534
+ - `ContractApplied` is structurally validated (Elements, keys, types). Quoted numeric
535
+ strings for metric fields are rejected (the schema types them as JSON numbers), and the
536
+ FOCUS 1.4 metric exclusivity (`ContractAppliedObjectSchema` `oneOf`: cost *xor*
537
+ quantity+unit) is enforced.
538
+
539
+ ### Changed
540
+
541
+ - The internal validator is repositioned as a **structural + semantic linter**:
542
+ `validate_focus_1_4` → **`lint_focus_1_4_structure`**, with explicit levels
543
+ (`STRUCTURAL_VALID`, `SEMANTIC_VALID`); it never asserts cross-dataset or official
544
+ conformance. `validate_focus_1_4` is retained as a **deprecated alias**.
545
+ - Conversion of `ContractApplied` from a 1.3 source now **migrates** it to the 1.4 schema:
546
+ re-cased identifier keys, and — since 1.3 permits all metrics but 1.4's `oneOf` does not
547
+ — a 1.3 element carrying both cost and quantity keeps the cost branch, preserving
548
+ quantity/unit losslessly as `x_ContractCommitmentAppliedQuantity`/`…Unit`.
549
+ - New typed API `focus_data_toolkit.convert.contract_applied` (parse / validate / migrate).
550
+ - Fixed a whitespace-mismatch bug in the Cost-and-Usage → Invoice Detail back-link key.
551
+ - README/docs no longer describe the linter as a full conformance validator.
552
+
553
+ <!-- Reference links. 0.2.0/0.3.0 were pre-release development milestones and were never tagged
554
+ or published, so only the first public release (0.9.0) has a tag link. -->
555
+ [Unreleased]: https://github.com/guymano/focus-data-toolkit/compare/v0.11.0...HEAD
556
+ [0.11.0]: https://github.com/guymano/focus-data-toolkit/compare/v0.11.0rc1...v0.11.0
557
+ [0.11.0rc1]: https://github.com/guymano/focus-data-toolkit/compare/v0.9.0...v0.11.0rc1
558
+ [0.9.0]: https://github.com/guymano/focus-data-toolkit/releases/tag/v0.9.0