focus-data-toolkit 0.11.0rc1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (226) hide show
  1. focus_data_toolkit-0.11.0rc1/CHANGELOG.md +542 -0
  2. focus_data_toolkit-0.11.0rc1/CONTRIBUTING.md +111 -0
  3. focus_data_toolkit-0.11.0rc1/LICENSE +21 -0
  4. focus_data_toolkit-0.11.0rc1/LICENSES/CC-BY-4.0.txt +156 -0
  5. focus_data_toolkit-0.11.0rc1/MANIFEST.in +25 -0
  6. focus_data_toolkit-0.11.0rc1/NOTICE +60 -0
  7. focus_data_toolkit-0.11.0rc1/PKG-INFO +519 -0
  8. focus_data_toolkit-0.11.0rc1/README.md +466 -0
  9. focus_data_toolkit-0.11.0rc1/SECURITY.md +94 -0
  10. focus_data_toolkit-0.11.0rc1/docs/audit/2026-07-improvement-plan.md +375 -0
  11. focus_data_toolkit-0.11.0rc1/docs/audit/2026-07-repo-audit.md +267 -0
  12. focus_data_toolkit-0.11.0rc1/docs/compatibility.md +88 -0
  13. focus_data_toolkit-0.11.0rc1/docs/model-provenance.md +114 -0
  14. focus_data_toolkit-0.11.0rc1/docs/releasing.md +183 -0
  15. focus_data_toolkit-0.11.0rc1/docs/runner.md +129 -0
  16. focus_data_toolkit-0.11.0rc1/docs/security-model.md +102 -0
  17. focus_data_toolkit-0.11.0rc1/docs/studio.md +70 -0
  18. focus_data_toolkit-0.11.0rc1/docs/supplements.md +142 -0
  19. focus_data_toolkit-0.11.0rc1/docs/versioning.md +78 -0
  20. focus_data_toolkit-0.11.0rc1/pyproject.toml +142 -0
  21. focus_data_toolkit-0.11.0rc1/schema/model_provenance.schema.json +124 -0
  22. focus_data_toolkit-0.11.0rc1/scripts/check_pinned_actions.py +62 -0
  23. focus_data_toolkit-0.11.0rc1/scripts/generate_resolved_sbom.py +267 -0
  24. focus_data_toolkit-0.11.0rc1/scripts/generate_sbom.py +177 -0
  25. focus_data_toolkit-0.11.0rc1/scripts/verify_model_provenance.py +290 -0
  26. focus_data_toolkit-0.11.0rc1/scripts/verify_release.py +196 -0
  27. focus_data_toolkit-0.11.0rc1/setup.cfg +4 -0
  28. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/__init__.py +69 -0
  29. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/__main__.py +6 -0
  30. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/_version.py +8 -0
  31. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/cli.py +968 -0
  32. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/context/__init__.py +88 -0
  33. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/context/billing.py +54 -0
  34. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/context/provider.py +90 -0
  35. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/convert/__init__.py +708 -0
  36. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/convert/billing_period.py +65 -0
  37. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/convert/contract_applied.py +235 -0
  38. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/convert/contract_commitment.py +182 -0
  39. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/convert/cost_and_usage.py +179 -0
  40. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/convert/detect.py +39 -0
  41. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/convert/invoice_detail.py +199 -0
  42. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/convert/streaming.py +1030 -0
  43. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/errors.py +145 -0
  44. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/focus_json.py +68 -0
  45. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/__init__.py +61 -0
  46. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/_shim.py +43 -0
  47. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/engine/__init__.py +14 -0
  48. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/engine/context.py +12 -0
  49. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/engine/determinism.py +117 -0
  50. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/engine/json_focus.py +63 -0
  51. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/engine/ladder.py +71 -0
  52. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  53. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/engine/serialize.py +151 -0
  54. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  55. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  56. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  57. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  58. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  59. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  60. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/providers/__init__.py +29 -0
  61. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/providers/aws.py +186 -0
  62. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/providers/azure.py +191 -0
  63. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/providers/gcp.py +194 -0
  64. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/providers/profile.py +123 -0
  65. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/scenarios.py +178 -0
  66. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/versions/__init__.py +17 -0
  67. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/versions/adapter.py +41 -0
  68. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/versions/v1_2.py +111 -0
  69. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/generators/versions/v1_3.py +154 -0
  70. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/io/__init__.py +1 -0
  71. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/io/atomic_writer.py +462 -0
  72. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/io/csv_io.py +128 -0
  73. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/io/parquet_io.py +528 -0
  74. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/io/records.py +92 -0
  75. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/io/row_source.py +117 -0
  76. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/lifecycle.py +342 -0
  77. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/manifest.py +114 -0
  78. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/__init__.py +43 -0
  79. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/capabilities.py +66 -0
  80. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  81. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  82. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  83. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/focus_json_keys.py +112 -0
  84. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  85. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/json_schema_check.py +205 -0
  86. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  87. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  88. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  89. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  90. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  91. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/model_provenance.json +58 -0
  92. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/model/validator.py +498 -0
  93. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/modes.py +18 -0
  94. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/official_validator.py +61 -0
  95. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/progress.py +89 -0
  96. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/provenance.py +106 -0
  97. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/py.typed +1 -0
  98. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/runtime.py +243 -0
  99. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/schema/__init__.py +17 -0
  100. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/schema/detection.py +274 -0
  101. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/schema/registry.py +127 -0
  102. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/storage/__init__.py +1 -0
  103. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/storage/external_index.py +99 -0
  104. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/storage/spill.py +150 -0
  105. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/studio/__init__.py +19 -0
  106. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/studio/app.py +467 -0
  107. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/studio/config.py +42 -0
  108. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/studio/frontend/app.js +214 -0
  109. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/studio/frontend/index.html +101 -0
  110. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/studio/frontend/style.css +60 -0
  111. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/studio/jobs.py +142 -0
  112. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/studio/preview.py +32 -0
  113. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/studio/security.py +125 -0
  114. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/studio/server.py +71 -0
  115. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/__init__.py +50 -0
  116. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  117. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  118. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  119. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  120. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  121. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  122. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/adapters/registry.py +215 -0
  123. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/apply.py +318 -0
  124. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/gaps.py +219 -0
  125. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/kinds.py +118 -0
  126. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/loader.py +409 -0
  127. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/spec.py +74 -0
  128. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/supplement/validate.py +215 -0
  129. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/validate/__init__.py +15 -0
  130. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/validate/allocation.py +333 -0
  131. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/validate/bundle.py +254 -0
  132. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/validate/codes.py +93 -0
  133. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/validate/corrections.py +245 -0
  134. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/validate/reconciliation.py +98 -0
  135. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit/validate/referential.py +289 -0
  136. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit.egg-info/PKG-INFO +519 -0
  137. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit.egg-info/SOURCES.txt +224 -0
  138. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit.egg-info/dependency_links.txt +1 -0
  139. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit.egg-info/entry_points.txt +2 -0
  140. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit.egg-info/requires.txt +38 -0
  141. focus_data_toolkit-0.11.0rc1/src/focus_data_toolkit.egg-info/top_level.txt +1 -0
  142. focus_data_toolkit-0.11.0rc1/tests/conftest.py +76 -0
  143. focus_data_toolkit-0.11.0rc1/tests/fixtures/client_like/SOURCES.md +9 -0
  144. focus_data_toolkit-0.11.0rc1/tests/fixtures/client_like/consolidated_multi_provider_1_3.csv +5 -0
  145. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/README.md +34 -0
  146. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/aws_1_2_cost_and_usage_rows100_seed42.csv +101 -0
  147. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/aws_1_2_cost_and_usage_rows25_seed7.csv +26 -0
  148. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/aws_1_2_cost_and_usage_rows25_seed7_credits.csv +26 -0
  149. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/aws_1_3_contract_commitment_rows100_seed42.csv +7 -0
  150. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/aws_1_3_cost_and_usage_rows100_seed42.csv +101 -0
  151. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/aws_1_3_cost_and_usage_rows25_seed7.csv +26 -0
  152. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/aws_1_3_cost_and_usage_rows25_seed7_credits.csv +26 -0
  153. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/azure_1_2_cost_and_usage_rows100_seed42.csv +101 -0
  154. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/azure_1_2_cost_and_usage_rows25_seed7.csv +26 -0
  155. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/azure_1_2_cost_and_usage_rows25_seed7_credits.csv +26 -0
  156. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/azure_1_3_contract_commitment_rows100_seed42.csv +9 -0
  157. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/azure_1_3_cost_and_usage_rows100_seed42.csv +101 -0
  158. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/azure_1_3_cost_and_usage_rows25_seed7.csv +26 -0
  159. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/azure_1_3_cost_and_usage_rows25_seed7_credits.csv +26 -0
  160. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/gcp_1_2_cost_and_usage_rows100_seed42.csv +101 -0
  161. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/gcp_1_2_cost_and_usage_rows25_seed7.csv +26 -0
  162. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/gcp_1_2_cost_and_usage_rows25_seed7_credits.csv +26 -0
  163. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/gcp_1_3_contract_commitment_rows100_seed42.csv +9 -0
  164. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/gcp_1_3_cost_and_usage_rows100_seed42.csv +101 -0
  165. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/gcp_1_3_cost_and_usage_rows25_seed7.csv +26 -0
  166. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/gcp_1_3_cost_and_usage_rows25_seed7_credits.csv +26 -0
  167. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/scenarios_correction_set.csv +4 -0
  168. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/scenarios_sca_equal.csv +4 -0
  169. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/scenarios_sca_negative.csv +3 -0
  170. focus_data_toolkit-0.11.0rc1/tests/fixtures/golden/compatibility_golden/scenarios_sca_weighted.csv +4 -0
  171. focus_data_toolkit-0.11.0rc1/tests/fixtures/official/SOURCES.md +12 -0
  172. focus_data_toolkit-0.11.0rc1/tests/fixtures/official/invoice_detail_grain_example.json +5 -0
  173. focus_data_toolkit-0.11.0rc1/tests/fixtures/official/numeric_format_examples.json +6 -0
  174. focus_data_toolkit-0.11.0rc1/tests/test_adapters_aws.py +179 -0
  175. focus_data_toolkit-0.11.0rc1/tests/test_adapters_azure_gcp.py +142 -0
  176. focus_data_toolkit-0.11.0rc1/tests/test_atomic_write.py +192 -0
  177. focus_data_toolkit-0.11.0rc1/tests/test_billing_lifecycle.py +101 -0
  178. focus_data_toolkit-0.11.0rc1/tests/test_bundle_gate.py +226 -0
  179. focus_data_toolkit-0.11.0rc1/tests/test_bundle_validation.py +288 -0
  180. focus_data_toolkit-0.11.0rc1/tests/test_capabilities.py +88 -0
  181. focus_data_toolkit-0.11.0rc1/tests/test_cli.py +374 -0
  182. focus_data_toolkit-0.11.0rc1/tests/test_client_like.py +52 -0
  183. focus_data_toolkit-0.11.0rc1/tests/test_container.py +58 -0
  184. focus_data_toolkit-0.11.0rc1/tests/test_contract_applied.py +169 -0
  185. focus_data_toolkit-0.11.0rc1/tests/test_contract_commitment_semantics.py +98 -0
  186. focus_data_toolkit-0.11.0rc1/tests/test_convert_roundtrip.py +130 -0
  187. focus_data_toolkit-0.11.0rc1/tests/test_corrections.py +82 -0
  188. focus_data_toolkit-0.11.0rc1/tests/test_cross_provider.py +94 -0
  189. focus_data_toolkit-0.11.0rc1/tests/test_detect.py +21 -0
  190. focus_data_toolkit-0.11.0rc1/tests/test_fake_provider.py +107 -0
  191. focus_data_toolkit-0.11.0rc1/tests/test_fixtures_are_synthetic.py +100 -0
  192. focus_data_toolkit-0.11.0rc1/tests/test_focus_json.py +35 -0
  193. focus_data_toolkit-0.11.0rc1/tests/test_gaps.py +129 -0
  194. focus_data_toolkit-0.11.0rc1/tests/test_generator_golden.py +80 -0
  195. focus_data_toolkit-0.11.0rc1/tests/test_generators.py +72 -0
  196. focus_data_toolkit-0.11.0rc1/tests/test_grouping_keys.py +60 -0
  197. focus_data_toolkit-0.11.0rc1/tests/test_lifecycle_chains.py +141 -0
  198. focus_data_toolkit-0.11.0rc1/tests/test_lineage_counters.py +76 -0
  199. focus_data_toolkit-0.11.0rc1/tests/test_lint_focus.py +183 -0
  200. focus_data_toolkit-0.11.0rc1/tests/test_manifest.py +109 -0
  201. focus_data_toolkit-0.11.0rc1/tests/test_model_provenance.py +157 -0
  202. focus_data_toolkit-0.11.0rc1/tests/test_multi_provider.py +142 -0
  203. focus_data_toolkit-0.11.0rc1/tests/test_official_json_schemas.py +210 -0
  204. focus_data_toolkit-0.11.0rc1/tests/test_packaging.py +139 -0
  205. focus_data_toolkit-0.11.0rc1/tests/test_parquet.py +207 -0
  206. focus_data_toolkit-0.11.0rc1/tests/test_parquet_input.py +166 -0
  207. focus_data_toolkit-0.11.0rc1/tests/test_participant_entities.py +76 -0
  208. focus_data_toolkit-0.11.0rc1/tests/test_partitioning.py +267 -0
  209. focus_data_toolkit-0.11.0rc1/tests/test_progress_cancel.py +166 -0
  210. focus_data_toolkit-0.11.0rc1/tests/test_release_tooling.py +182 -0
  211. focus_data_toolkit-0.11.0rc1/tests/test_replace_recovery.py +224 -0
  212. focus_data_toolkit-0.11.0rc1/tests/test_runtime.py +145 -0
  213. focus_data_toolkit-0.11.0rc1/tests/test_schema_detection.py +201 -0
  214. focus_data_toolkit-0.11.0rc1/tests/test_single_source_focus_rules.py +59 -0
  215. focus_data_toolkit-0.11.0rc1/tests/test_spill.py +74 -0
  216. focus_data_toolkit-0.11.0rc1/tests/test_split_allocation.py +171 -0
  217. focus_data_toolkit-0.11.0rc1/tests/test_streaming.py +151 -0
  218. focus_data_toolkit-0.11.0rc1/tests/test_strict_conversion.py +57 -0
  219. focus_data_toolkit-0.11.0rc1/tests/test_studio.py +391 -0
  220. focus_data_toolkit-0.11.0rc1/tests/test_supplement_apply.py +260 -0
  221. focus_data_toolkit-0.11.0rc1/tests/test_supplement_loader.py +306 -0
  222. focus_data_toolkit-0.11.0rc1/tests/test_supplement_streaming.py +151 -0
  223. focus_data_toolkit-0.11.0rc1/tests/test_synthetic_conversion.py +75 -0
  224. focus_data_toolkit-0.11.0rc1/tests/test_validator.py +40 -0
  225. focus_data_toolkit-0.11.0rc1/tests/test_workflow_pins.py +51 -0
  226. focus_data_toolkit-0.11.0rc1/tools/extract_focus_1_4_model.py +133 -0
@@ -0,0 +1,542 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented here. The format is based on
4
+ [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project
5
+ adheres to [Semantic Versioning](https://semver.org/) with
6
+ [PEP 440](https://peps.python.org/pep-0440/) version strings. See
7
+ [docs/versioning.md](docs/versioning.md) for the versioning and reproducibility
8
+ policy.
9
+
10
+ ## [Unreleased]
11
+
12
+ ### Changed
13
+
14
+ - **Release workflow (`release.yml`)**: the GitHub Release step is now idempotent — if a release for
15
+ the tag already exists (e.g. the tag was created via the "Draft a new release" UI, which also
16
+ creates the release), it updates that release in place and attaches the attested assets instead of
17
+ failing. A plain `git push` of the tag (no pre-existing release) still creates it fresh.
18
+
19
+ ## [0.11.0rc1] — 2026-07-18
20
+
21
+ First release candidate published to PyPI — a **pre-release** (marked as such per the honesty gate,
22
+ because the embedded FOCUS model's provenance is `partial`; see
23
+ [docs/model-provenance.md](docs/model-provenance.md)). It bundles the three deployment access
24
+ methods on the single Core: the CLI/SDK, the containerised **Runner** (Lot B) and the local
25
+ **Studio** web UI (Lot C), plus the Core progress/cancellation, disk-budget and CLI additions
26
+ (Lot A). Install with `pip install --pre focus-data-toolkit` (pre-releases are not selected by a
27
+ plain `pip install`).
28
+
29
+ ### Added — Studio: local web UI (deployment Lot C)
30
+
31
+ - **`focus-toolkit ui`** launches a local web app (FastAPI, behind the optional `[studio]` extra;
32
+ the command imports it lazily so a core install is unaffected) over the **same Core** — every
33
+ operation drives the same SDK the CLI/Runner use, so its manifests, diagnostics and checksums are
34
+ identical. Detect a source, pick a file under `--root` / upload (capped) / generate synthetic
35
+ data, convert (strict|synthetic, CSV|Parquet) with **live per-phase progress and cancel**,
36
+ preview a **sampled** page (the full file is never loaded), and download datasets, manifest,
37
+ diagnostics (JSON/CSV), `SHA256SUMS` and an HTML summary.
38
+ - **Security:** binds `127.0.0.1` by default (a non-loopback `--host` is refused without
39
+ `--allow-remote`); a fresh per-start token is required on every API call; `Host`/`Origin` are
40
+ validated (anti DNS-rebinding / CSRF); file access is confined to the allowlisted `--root`;
41
+ uploads stream to disk and are size-capped. No telemetry, no external upload.
42
+ - **Path confinement:** `resolve_within_root` walks real directory entries (matching each component
43
+ by name, never concatenating the user string into a path) **and** canonicalises every matched
44
+ entry with `Path.resolve` — a symlink, Windows junction or reparse point whose real target
45
+ escapes `--root` is refused, while a link that stays inside is followed; absolute, drive-relative
46
+ and UNC paths and `..` traversal are rejected.
47
+ - **Bounded by design:** one conversion at a time by default (extra submissions queue); per-job
48
+ scratch under a work dir with TTL + startup cleanup; generation is row-capped in the UI (use the
49
+ CLI/Runner for very large synthetic sets). New extras `studio` and `studio-all`; see
50
+ [docs/studio.md](docs/studio.md).
51
+
52
+ ### Added — Runner: containerised batch image (deployment Lot B)
53
+
54
+ - **OCI image** (`Dockerfile`) whose entrypoint **is** the `focus-toolkit` CLI — a container run
55
+ equals a CLI run (same manifests, diagnostics, checksums, exit codes; no FOCUS logic
56
+ duplicated). Batch-only (no HTTP server). Multi-stage build on a **digest-pinned**
57
+ `python:3.12-slim-bookworm`, bundling the `[parquet]` extra; **non-root** (uid 65532),
58
+ **read-only-rootfs compatible** (only `/work` and `/output` written), `FOCUS_TOOLKIT_WORK_DIR=/work`.
59
+ Exec-form entrypoint so `docker stop` (SIGTERM) cancels cleanly (exit 130, nothing partial
60
+ published). Volumes: `/input` (ro), `/output`, `/work`. See [docs/runner.md](docs/runner.md).
61
+ - **Container CI** (`.github/workflows/container.yml`): builds the image on every PR / push to
62
+ `main` (no publish) and runs `docker run` smoke tests — non-root uid, read-only-rootfs streaming
63
+ Parquet convert, exit codes, SIGTERM handling — plus a trivy scan (fails on HIGH/CRITICAL). A
64
+ fast static test (`tests/test_container.py`) enforces the base-image digest pin, non-root user
65
+ and exec-form entrypoint.
66
+ - **Container release** (`.github/workflows/release-container.yml`): on a `v*` tag, runs the same
67
+ release gates as the PyPI flow (tag matches `__version__`; provenance-honesty gate), builds a
68
+ candidate image, **scans it (trivy) before any public tag is assigned**, then publishes to
69
+ `ghcr.io/guymano/focus-data-toolkit` — **immutable** `<version>` and `sha-<full-commit>` tags
70
+ plus a **rolling** `<major>.<minor>` alias (PEP 440 tag parsing; no `latest`) — generates a
71
+ CycloneDX SBOM, attests build provenance and **signs with cosign** (keyless OIDC), in a
72
+ reviewer-gated `ghcr` environment. All actions pinned by commit SHA.
73
+
74
+ ### Added — progress, cancellation, disk budgets & pipeline ergonomics (deployment Lot A)
75
+
76
+ - **Progress reporting**: the streaming engine (`convert_files`) accepts an optional
77
+ `progress` callback receiving throttled `ProgressEvent`s per phase (`READING`,
78
+ `TRANSFORMING`, `AGGREGATING`, `WRITING`, `VALIDATING`, `PUBLISHING`) with a completed
79
+ count, an optional total, a unit (`rows`/`bytes`) and a message — derived without
80
+ materialising data (CSV byte cursor / Parquet footer row count). `focus-toolkit convert
81
+ --progress` renders a single throttled status line on stderr. All hooks are opt-in and
82
+ keyword-only; output is byte-identical with or without them.
83
+ - **Cooperative cancellation**: `convert_files(..., cancel=...)` checks a predicate between
84
+ rows and validation passes and raises `ConversionCancelled` — the atomic staging directory
85
+ is removed, so nothing partial is ever published. The CLI maps SIGINT/SIGTERM to a clean
86
+ cancel (exit code **130**), so `Ctrl-C` and `docker stop` unwind cleanly instead of dying
87
+ mid-write.
88
+ - **Separate disk budgets** (`focus_data_toolkit.runtime`): the scratch filesystem and the
89
+ output filesystem are budgeted independently via `FOCUS_TOOLKIT_WORK_DIR`,
90
+ `FOCUS_TOOLKIT_MAX_WORK_BYTES`, `FOCUS_TOOLKIT_MIN_WORK_FREE_BYTES` and
91
+ `FOCUS_TOOLKIT_MIN_OUTPUT_FREE_BYTES` (`FOCUS_TOOLKIT_LOG_LEVEL` too). A best-effort
92
+ pre-flight (estimate with a safety margin) plus periodic in-run checks fail fast with a
93
+ structured `FDT-IO-005` (output) / `FDT-IO-006` (work / temp budget) diagnostic and CLI
94
+ exit code **5**, instead of a raw `OSError` mid-run. `WORK_DIR` relocates the SQLite
95
+ aggregation + bundle-spill scratch off the output disk (business artifacts stay
96
+ byte-identical; scratch is always cleaned up).
97
+ - **Pipeline-friendly exit codes**: `focus-toolkit convert --exit-policy pipeline` maps the
98
+ functional-but-complete outcomes (3 = strict incomplete, 4 = synthetic assumptions) to 0,
99
+ so orchestrators (Kubernetes / Airflow / Jenkins / AWS Batch) don't flag a legitimate run
100
+ failed. The default `detailed` policy keeps the historic codes; full status stays in the
101
+ manifest and `_run.json`.
102
+ - **New CLI commands**: `focus-toolkit detect` (dataset/version of a file header, text/JSON),
103
+ `focus-toolkit validate-bundle` (cross-dataset validation gate over explicit per-dataset
104
+ files or an auto-detected `--directory`; ambiguous combinations refused), and
105
+ `focus-toolkit version`. New SDK exports: `ProgressEvent`, `ConversionCancelled`.
106
+
107
+ ### Added — provider-native supplement adapters
108
+
109
+ - **Adapters** translate documented cloud-provider export formats into the
110
+ canonical supplement kinds automatically, so a client passes native exports
111
+ straight to `convert` / `supplements validate` without renaming anything.
112
+ First adapters (AWS): `aws-invoice-summary` (Invoicing API `InvoiceSummary`,
113
+ incl. nested `Entity.InvoicingEntity`, `DueDate`, `PurchaseOrderNumber`) →
114
+ `invoice`; `aws-savings-plans` (Savings Plans inventory: `paymentOption`,
115
+ `state`, `start`) → `contract_commitment`. Each adapter is a vendored,
116
+ versioned JSON mapping table with official-doc provenance
117
+ (`supplement/adapters/adapters_provenance.json`, sha256-verified); the format
118
+ is auto-detected from the header (force with `FILE:<adapter-name>`).
119
+ Translated rows flow through the unchanged supplement validation and carry
120
+ `ENRICHED` lineage attributed as `supplement:<adapter>@<version>:<file>`. An
121
+ adapter only maps fields its table describes (residual gaps are reported, not
122
+ guessed); an unrecognized export falls back to the generic FOCUS-named path.
123
+ New command: `fdt supplements adapters`. Adapters ship for **AWS**
124
+ (`aws-invoice-summary`, `aws-savings-plans`), **Azure** (`azure-invoice` —
125
+ Billing Invoices REST API; `InvoiceStatus` Due/OverDue/Paid → `Issued`,
126
+ Void → `Voided`) and **GCP** (`gcp-compute-commitments` — Compute Engine
127
+ `regionCommitments`; `status` and CUD payment facts → `contract_commitment`).
128
+
129
+ ### Added — supplemental client data (promise #3)
130
+
131
+ - **Gap analysis** (`fdt gaps`): reports, per FOCUS 1.4 dataset, exactly which
132
+ columns block strict production for a given 1.2/1.3 source — computed from
133
+ the converter's own provenance rules and annotated from the embedded model —
134
+ plus ready-to-fill CSV templates per supplement kind. Missing mandatory
135
+ source columns are reported as source-completeness gaps.
136
+ - **Supplement bundles**: clients supply the missing provider-issued facts as
137
+ sidecar files (CSV/JSON, gzip ok; kinds `billing_period`, `invoice`,
138
+ `invoice_line`, `contract_commitment`; kind auto-detected from the header or
139
+ forced with `FILE:KIND`). Supplements are validated against the source and
140
+ the model before any use (`FDT-SUPP-0xx`: duplicate keys, unknown columns,
141
+ format/allowed-value violations, orphans, `BilledCost` reconciliation
142
+ conflicts, per-column coverage). Pre-flight command:
143
+ `fdt supplements validate`.
144
+ - **ENRICHED conversion**: `convert_to_focus_1_4(..., supplements=...)`
145
+ applies supplied facts with `ENRICHED` lineage and full attribution
146
+ (`supplement:<kind>:<file>` + sha256 in the new manifest `supplements`
147
+ section). At full coverage, **strict mode now produces all four FOCUS 1.4
148
+ datasets** with nothing invented; partial coverage keeps the dataset
149
+ `NOT_PRODUCED` with per-value counters showing how close it is. In strict
150
+ mode uncovered nullable assumed columns are emitted empty (synthetic
151
+ defaults never leak); real issuer-assigned `InvoiceDetailId`s replace the
152
+ locally generated back-links.
153
+
154
+ ### Added — capability profiles
155
+
156
+ - New `CapabilityProfile` (`focus_data_toolkit.model.capabilities`): an
157
+ explicit, validated declaration of the FOCUS applicability conditions a
158
+ source supports (`SupportsUnitPricing`,
159
+ `SupportsMultiplePricingCategories`). The linter enforces
160
+ conditionally-required columns only for declared conditions; the conversion
161
+ pipeline records the active profile in the manifest (`capability_profile`),
162
+ so an unevaluated condition set is visible instead of silent. CLI:
163
+ repeatable `--supports CONDITION` on `convert` and `validate`.
164
+
165
+ ### Added — per-value lineage counters
166
+
167
+ - The manifest's produced-dataset entries gain a `lineage_summary` section
168
+ counting, per column, how many values actually took each lineage. Today it
169
+ covers the pricing-currency backfill pair (`PricingCurrency` /
170
+ `PricingCurrencyEffectiveCost`): the headline column lineage stays the
171
+ conservative `DERIVED`, and the summary shows the real observed/backfilled
172
+ mix (e.g. `{"OBSERVED": 99800, "DERIVED": 200}`). Identical in the eager and
173
+ streaming paths; bounded memory (columns × lineage categories).
174
+
175
+ ### Added — official FOCUS JSON schemas
176
+
177
+ - The four official FOCUS 1.4 JSON object schemas (`ContractApplied`,
178
+ `AllocatedMethodDetails`, `CommitmentProgramEligibilityDetails`,
179
+ `ContractCommitmentApplicability`) are vendored verbatim from the
180
+ specification repository (tag `v1.4`) under
181
+ `focus_data_toolkit/model/json_schemas/`, with a provenance manifest
182
+ (source paths, sha256, CC-BY-4.0 attribution). The linter now evaluates
183
+ every JSON-object column against its official schema — conditional scope
184
+ rules, metric exclusivity, ranges, PascalCase `x_` custom keys — via a
185
+ small dependency-free interpreter of the schema subset; violations surface
186
+ as `official_schema_violation`. Previously only `ContractApplied` was
187
+ deep-validated and `ContractCommitmentApplicability` was only checked to
188
+ be a JSON object.
189
+
190
+ ### Fixed — FOCUS conformance (may change output bytes)
191
+
192
+ - **Synthetic `ContractCommitmentApplicability`**: the object now declares
193
+ `{"IsComplexScope": true, ...}` — the official object schema requires a scope
194
+ representation (`Inclusions` + `InclusionOperator` become required when no
195
+ scope flag is set), so the previous `x_Source`-only object was normatively
196
+ invalid. The value remains `ASSUMED` and still never passes strict mode.
197
+ - **`ContractCommitmentDurationType`**: an unparseable or inverted commitment
198
+ period no longer yields a fabricated `"12 Months"`. The value stays empty
199
+ (not derivable) and the affected rows are reported as an aggregated
200
+ `FDT-CC-001` WARNING; the mandatory-column lint then flags the dataset
201
+ instead of silently publishing an arbitrary duration.
202
+
203
+ - **1.2 participant-entity migration**: `HostProviderName` is no longer derived
204
+ from the deprecated `PublisherName` (the entity that *produced* the service —
205
+ not the infrastructure host). Per the official FOCUS 1.4 `HostProviderName`
206
+ rules, when the source does not expose the underlying host the value MUST
207
+ match `ServiceProviderName`; a 1.2 source never exposes it, so both columns
208
+ now derive from `ProviderName` and carry `DERIVED` lineage (previously
209
+ `RENAMED`) with the spec rule recorded in the manifest. The per-row provider
210
+ context applies the same rule (`host == service` when the host is not
211
+ exposed; the publisher is never used as a fallback host).
212
+
213
+ ### Added — release pipeline
214
+
215
+ - Secure release workflows (`.github/workflows/`): a reusable **build-once**
216
+ workflow (`release-build.yml`), a **`release-dry-run.yml`** (no publish, no
217
+ privileged scopes), a tag-triggered **`release.yml`** that attests
218
+ wheel/sdist/SBOM/checksums (GitHub Artifact Attestations, keyless OIDC) and
219
+ publishes via **PyPI Trusted Publishing** in a gated environment, and a
220
+ **`reproducibility.yml`** double-build check. Artifacts flow between jobs by
221
+ digest — the publish job never rebuilds.
222
+ - A deterministic **CycloneDX 1.5 SBOM** generator (`scripts/generate_sbom.py`)
223
+ that records the embedded FOCUS 1.4 model as a first-class `data` component
224
+ (CC-BY-4.0 + provenance hash), and an offline **release verifier**
225
+ (`scripts/verify_release.py`) checking `SHA256SUMS`, the SBOM, and version
226
+ consistency. Both are covered by `tests/test_release_tooling.py`.
227
+
228
+ ### Changed — dependencies
229
+
230
+ - Widened the `parquet` extra to `pyarrow>=15,<26` (the lock resolves to 25.x)
231
+ and `pytest-cov` to `>=5,<8` (dev). The Parquet suite passes unchanged.
232
+
233
+ ### Security
234
+
235
+ - Resolved **PYSEC-2026-113** by moving the resolved `pyarrow` to `>= 23.0.1`
236
+ (25.x); the `pip-audit` gate now runs with **no `--ignore-vuln` exception**.
237
+
238
+ ## [0.9.0] — 2026-07-17
239
+
240
+ **First public release.** `0.9.0` is the first version prepared for publication
241
+ to PyPI (a deliberate "near-stable" signal; `1.0.0` is reserved for after
242
+ real-world feedback — see [docs/versioning.md](docs/versioning.md)). It bundles
243
+ all functionality developed across the `0.2.0` (P0) and `0.3.0` (P1) milestones,
244
+ detailed in their sections below, and adds the packaging, CI/supply-chain,
245
+ governance, and provenance work that makes the project publishable. Publication
246
+ itself is performed by the release pipeline (see
247
+ [docs/releasing.md](docs/releasing.md)).
248
+
249
+ ### Added — packaging & distribution
250
+
251
+ - **PyPI-ready packaging**: single-sourced version (`focus_data_toolkit._version`),
252
+ PEP 639 SPDX license metadata (`license = "MIT"` + `license-files`), a PEP 561
253
+ `py.typed` marker with the `Typing :: Typed` classifier, project URLs, and a
254
+ `MANIFEST.in` that ships the sources needed to build and verify from an sdist.
255
+ - **Reproducible installs**: a committed `uv.lock` and hash-pinned
256
+ `constraints/*.txt`; the optional `[validator]` extra now resolves
257
+ `focus-validator` from **PyPI** (Python 3.12+) instead of a git URL, so wheels
258
+ and sdists upload and install cleanly. New `[release]` extra (`build`, `twine`).
259
+ - **Packaging tests** build and install a wheel **and** an sdist in a clean
260
+ environment and smoke-test the result.
261
+
262
+ ### Added — CI & supply-chain hardening
263
+
264
+ - Modular workflows: lint, type-check (mypy), a test matrix across
265
+ ubuntu/windows × Python 3.11–3.13, coverage floor, and a packaging job.
266
+ - Least-privilege `permissions: contents: read` with per-job elevation; every
267
+ GitHub Action pinned to a full commit SHA (Docker image actions to an
268
+ `@sha256` digest), enforced by `scripts/check_pinned_actions.py` and a test.
269
+ - Security scanning: `pip-audit` (pinned), `gitleaks`, `actionlint`, and
270
+ `zizmor`; `Dependabot` for actions and Python dependencies; CodeQL and OpenSSF
271
+ Scorecard workflows (gated behind `workflow_dispatch` until repository code
272
+ scanning is enabled — see [docs/releasing.md](docs/releasing.md)).
273
+
274
+ ### Added — governance & documentation
275
+
276
+ - `SECURITY.md` (GitHub Private Vulnerability Reporting; the security-vulnerability
277
+ vs FOCUS-conformance-bug distinction; a supported-versions matrix; a "no client
278
+ data" rule), `CONTRIBUTING.md`, a `NOTICE` file, `.github/CODEOWNERS`, and
279
+ GitHub issue / pull-request templates.
280
+ - Documentation under `docs/`: `versioning.md`, `compatibility.md` (Python/OS/
281
+ FOCUS matrix, including the Windows streaming limitation), `security-model.md`,
282
+ `releasing.md` (with the operational, owner-only checklist), and
283
+ `model-provenance.md`.
284
+ - A test that scans committed fixtures for secrets/PII and requires each fixture
285
+ directory to document its synthetic provenance.
286
+
287
+ ### Added — FOCUS model provenance
288
+
289
+ - A machine-readable provenance manifest
290
+ (`src/focus_data_toolkit/model/model_provenance.json`) with a JSON Schema
291
+ (`schema/model_provenance.schema.json`) and a verifier
292
+ (`scripts/verify_model_provenance.py`, run in CI). It records the source
293
+ (FinOps FOCUS 1.4 Data Model workbook), the **verified CC-BY-4.0** license, the
294
+ extraction process, and the reproducible output hash. Status is `partial`
295
+ (the source workbook is not redistributed/hashed here); a `partial` → `complete`
296
+ gate blocks a fully-reproducible-provenance claim until the source is hashed.
297
+
298
+ ### Changed
299
+
300
+ - Project version set to **0.9.0** (first public release).
301
+
302
+ ### Fixed
303
+
304
+ - The committed FOCUS 1.4 model JSON `source` field pointed at a non-existent
305
+ `docs/focus/…xlsx` path, diverging from what `tools/extract_focus_1_4_model.py`
306
+ emits. It now matches the extractor's output, so re-running the extractor on
307
+ the same workbook reproduces the committed JSON byte-for-byte.
308
+
309
+ ## [0.3.0] — pre-release development (P1)
310
+
311
+ The 0.3.0 line ("P1") makes the toolkit reliable on **real client data** — consolidated,
312
+ multi-provider, multi-issuer, multi-currency, volumetric exports — without re-implementing the
313
+ 0.2.0 (P0) guardrails. It lands in two phases: **Phase A** (correctness & integrity) and
314
+ **Phase B** (scale & realism), both described below.
315
+
316
+ ### Added — schema detection
317
+
318
+ - **`focus_data_toolkit.schema.detect_focus_schema(headers)`** identifies the FOCUS
319
+ **dataset** (Cost and Usage / Contract Commitment / Billing Period / Invoice Detail) and
320
+ **version** (1.2 / 1.3 / 1.4) with a `confidence`, `exact_match`, and the exact
321
+ `missing` / `additional_focus` / `extension` / `unknown` columns and `ambiguous_candidates`.
322
+ It scores the header against a registry of normative column sets (`schema/registry.py`)
323
+ computed from the committed 1.4 model plus a removed-columns table, so a 1.4 file is never
324
+ taken for 1.3, a 1.3 export missing an optional column is not taken for 1.2, `x_` columns
325
+ never count against a match, and unknown non-`x_` columns are surfaced.
326
+ - CLI `convert --source-version` / `--source-dataset` force detection; **strict mode refuses
327
+ an ambiguous or low-confidence source** (clear error). The detection decision is recorded
328
+ in the manifest. `detect_focus_version` is retained as a compatibility wrapper.
329
+
330
+ ### Added — multi-provider context & correct grouping
331
+
332
+ - **Per-row context** (`focus_data_toolkit.context`): `BillingContext` and `ProviderContext`
333
+ are derived from the whole source, never from the first row. The first-row `_provider_context`
334
+ is gone; billing-period issuer is no longer back-filled from row 0. A bounded per-row
335
+ **context summary** (distinct providers / issuers / accounts / currencies / periods and
336
+ `multi_*` flags) is recorded in the manifest; ambiguous contexts are reported as diagnostics.
337
+ - **Invoice Detail now groups on the full business grain** — `(InvoiceIssuerName, InvoiceId,
338
+ BillingAccountId, BillingCurrency, BillingPeriodStart, BillingPeriodEnd, ChargeCategory)` —
339
+ preventing collisions across issuers/accounts/currencies/periods. `InvoiceDetailId` is a
340
+ clearly-local, versioned, collision-safe id (`x_fdt_idl_v1_<hash>` over a JSON-encoded key),
341
+ **never presented as issuer-assigned**.
342
+
343
+ *Because the `InvoiceDetailId` value scheme changed, synthetic Invoice Detail (and the Cost
344
+ and Usage back-link) have a new byte baseline as of 0.3.0. Conversion remains deterministic.*
345
+
346
+ ### Added — cross-dataset (bundle) validation
347
+
348
+ - **`validate_dataset_bundle(bundle)`** (alias `validate_bundle`) validates a bundle of
349
+ datasets **against each other** — deliberately separate from the per-dataset linter, which
350
+ never asserts cross-dataset validity. It reports referential integrity (uniqueness, FKs,
351
+ orphans), issuer/account/currency/period coherence, Cost-and-Usage ↔ Invoice-Detail
352
+ reconciliation (only when Invoice Detail is authoritative, with an explicit rounding
353
+ tolerance), **Split Cost Allocation** (ratios sum to 1, allocated costs sum to the origin,
354
+ ratio in [0,1], consistent method/unit, unique resources, incomplete groups flagged), and
355
+ commitment lifecycle (period/percentage) checks. Findings are grouped by severity
356
+ (error / warning / info / not-executable / not-applicable) and serialise to JSON.
357
+
358
+ ### Added — structured diagnostics & error catalog
359
+
360
+ - **`focus_data_toolkit.errors.Diagnostic`** carries a stable code, severity, business record
361
+ keys, column, expected/actual, suggestion, provenance and group/join context, and renders to
362
+ JSON, a readable console block, and CSV rows. A stable **`FDT-*` code catalog**
363
+ (`validate/codes.py`) replaces opaque "invalid input" messages.
364
+
365
+ ### Added — atomic writes
366
+
367
+ - **Results appear in the destination only after everything succeeds.** `write_result` stages
368
+ datasets in a temp directory on the same filesystem, enforces the mandatory lint gate (a
369
+ lint-failing result is never published), then writes the deterministic business manifest, an
370
+ operational `_run.json` sidecar (run id / timestamp / per-file SHA-256, kept **out** of the
371
+ business manifest so dataset bytes stay reproducible) and `SHA256SUMS` last, and publishes
372
+ with a single atomic rename. `io/atomic_writer.py` exposes `AtomicOutputDir` and
373
+ `OnExists` (**refuse** default / replace via crash-safe swap-dir / version). CLI:
374
+ `--on-exists`, `--keep-temp`.
375
+
376
+ ### Added — streaming conversion (Phase B)
377
+
378
+ - **`focus_data_toolkit.convert.convert_files(cost_and_usage, out_dir, …)`** streams the Cost
379
+ and Usage file **once** and stages Invoice Detail aggregation / Billing Period dedup in a
380
+ throwaway SQLite database (`storage/external_index.py`), so **peak memory is flat regardless
381
+ of row count** — a constant ~64 MB peak process RSS from 50k to 300k rows (6× the rows,
382
+ ×1.05 the memory; `tools/benchmark_streaming.py`), and ~38 MB of traced Python allocations
383
+ that a `slow` test asserts do not scale. Costs are summed with Python `Decimal` over an
384
+ `ORDER BY` (BINARY) scan — never SQL `SUM()` — so the streamed totals are bit-for-bit the
385
+ eager ones.
386
+ - **Byte-identical to the in-memory path**: both call the same pure per-row / per-group
387
+ functions (`convert_cost_and_usage_row`, `invoice_detail_row`, `billing_period_row`) and the
388
+ same `assemble_manifest`, so equivalence holds by construction (asserted on the datasets,
389
+ the manifest, and `SHA256SUMS`). A new streaming reader/writer layer (`io/records.py`,
390
+ `io/csv_io.py`) auto-detects gzip input and rejects wrong-field-count rows with a
391
+ line-numbered error. CLI: `convert --stream`.
392
+
393
+ ### Added — Parquet output (Phase B)
394
+
395
+ - **`--output-format parquet`** / `convert_files(…, output_format="parquet")` writes the
396
+ datasets as Parquet with **exact decimal128** (`io/parquet_io.py`): decimal columns use
397
+ `decimal128(precision, scale)` from a committed, reviewed scale registry
398
+ (`model/focus_1_4_decimal_scale.json`) — never binary float, and a value needing more scale
399
+ than the column allows raises with its line number instead of rounding silently. Dates are
400
+ UTC timestamps, JSON/strings verbatim, empty string ↔ null. Exactness contract: **CSV is
401
+ byte-exact, Parquet is decimal-value-exact**. PyArrow stays an optional `[parquet]` extra
402
+ with a clear install hint; the core stays standard-library only.
403
+ - **Partitioning & compression**: `--partition-by` writes the Cost and Usage dataset as a
404
+ Hive-partitioned Parquet tree (`COL=value/…/part-N.parquet`) on low-cardinality String /
405
+ Date-Time columns, keeping memory bounded to one open writer per partition; a high-cardinality
406
+ key is warned about (`FDT-IO-004`) and, past a hard cap, refused (nothing partial is
407
+ published). Partition columns live in the paths (standard Hive), so any `pyarrow.dataset`
408
+ reader — and the toolkit's own read-back lint gate — reconstruct full rows, with the values
409
+ round-tripping exactly (reconstructed as strings). `--compression`
410
+ (snappy default / zstd / gzip / none) and `--target-file-size` (approximate part-file roll)
411
+ tune the layout. The atomic writer now checksums files by path relative to the output root, so
412
+ every partition part appears in `SHA256SUMS`.
413
+
414
+ ### Added — synthetic scenarios & lifecycle (Phase B)
415
+
416
+ - **`generators/scenarios.py`**: deterministic, provider-agnostic scenario builders —
417
+ `split_allocation_group` (ratios sum to exactly 1, allocated costs sum to exactly the origin,
418
+ last consumer absorbs the residue), `correction_set` (original charge plus signed
419
+ `ChargeClass="Correction"` lines that reference it and record the running auditable net in
420
+ `x_NetCharge`; the original is never overwritten), and `billing_lifecycle_instances` (a
421
+ T0→T6 dataset-instance sequence with only allowed status transitions).
422
+ - **`focus_data_toolkit.lifecycle`**: a `DatasetInstance` snapshot structure and
423
+ status-transition checks (`FDT-CORR-004`) from the FOCUS `InvoiceIssueStatus`
424
+ (Open→Issued→Voided) and `BillingPeriodStatus` (Open→Closed) value sets — flagging un-void,
425
+ un-issue and silent reopen, scoped per subject.
426
+ - **New correction checks** in `validate_dataset_bundle`: net-sum reconciliation to the
427
+ declared `x_NetCharge` (`FDT-CORR-002`) and duplicate-`x_ChargeKey` overwrite detection
428
+ (`FDT-CORR-003`).
429
+
430
+ ### Fixed — streaming publish integrity (Phase B)
431
+
432
+ - The streaming path wrote datasets through direct file handles (for bounded memory), which
433
+ bypassed the atomic writer's file registration — so `SHA256SUMS` listed only the manifest and
434
+ the scratch SQLite database was published into the output directory. Produced files are now
435
+ fsync'd and enrolled for checksums, and the scratch DB is deleted before commit; `SHA256SUMS`
436
+ is complete and byte-identical to the eager path. (Regression tests added.)
437
+
438
+ ### Changed
439
+
440
+ - Version 0.3.0 (P1). `focus-data-toolkit[parquet]` / `[scale]` / `[all]` optional extras
441
+ (PyArrow, used for Parquet); the runtime core remains standard-library only. `pyarrow` is
442
+ added to the `dev` extra so CI runs the Parquet suite rather than skipping it. A `slow`
443
+ pytest marker is added and excluded from the default run. New public API surfaced at the
444
+ package root: `convert_files`, `DatasetInstance`, `check_status_transitions`. Reproducible
445
+ benchmark at `tools/benchmark_streaming.py`.
446
+
447
+ ## [0.2.0] — pre-release development (P0)
448
+
449
+ The 0.2.0 line makes the toolkit honest about what it produces and fixes real FOCUS
450
+ conformance defects.
451
+
452
+ ### Added — modes, provenance & manifest
453
+
454
+ - **Conversion modes** (`--mode`, `focus_data_toolkit.modes.Mode`):
455
+ - `strict` (**new default**) never invents provider-issued financial facts. A
456
+ canonical FOCUS 1.4 dataset is produced only when every Mandatory non-nullable
457
+ column has a factual lineage. From a Cost-and-Usage source, only Cost and Usage is
458
+ produced; Billing Period, Invoice Detail and the 1.4-expanded Contract Commitment are
459
+ reported `NOT_PRODUCED` with their blocking columns.
460
+ - `synthetic` generates assumed values for demos/tests; affected datasets are written
461
+ with a `synthetic_` filename prefix and marked `PRODUCED_SYNTHETIC`, never fully
462
+ conformant.
463
+ - **Value provenance / lineage** (`focus_data_toolkit.provenance`): every produced column
464
+ is classified `OBSERVED` / `RENAMED` / `DERIVED` / `ENRICHED` / `ASSUMED` /
465
+ `UNAVAILABLE`. Any `ASSUMED` column prevents a full-conformance claim.
466
+ - **Deterministic conversion manifest** (`focus_1_4_manifest.json`,
467
+ `focus_data_toolkit.manifest`): per-dataset status/conformance/reason and per-column
468
+ lineage. Written on every conversion; `--manifest PATH` writes an extra copy.
469
+ - **CLI exit codes**: `0` success without assumptions · `1` lint violation ·
470
+ `2` invalid arguments · `3` incomplete strict result · `4` synthetic result with
471
+ assumptions.
472
+
473
+ ### Changed — behaviour
474
+
475
+ - **Default conversion behaviour changed**: strict mode no longer emits Billing Period /
476
+ Invoice Detail / Contract Commitment reconstructed from Cost and Usage alone. Use
477
+ `--mode synthetic` for the previous all-four output (now correctly labelled).
478
+ - `convert_to_focus_1_4` gained a `mode` parameter; `ConversionResult` gained
479
+ `mode`, `provenance`, `manifest`, `not_produced`, `assumptions_present`.
480
+ - README rewritten to separate schema migration / enrichment / synthetic projection.
481
+ - In synthetic mode the Cost and Usage `InvoiceDetailId` back-link (a locally generated id)
482
+ is lineage `ASSUMED`, so synthetic Cost and Usage is labelled `PRODUCED_SYNTHETIC` and
483
+ `synthetic_`-prefixed; in strict mode it stays null. `PricingCurrency` /
484
+ `PricingCurrencyEffectiveCost` are lineage `DERIVED` (source value; nulls backfilled), not
485
+ `OBSERVED`.
486
+ - Manifest conformance for a factual dataset is set **after** the lint runs:
487
+ `STRUCTURAL_LINT` (passed), `LINT_FAILED` (failed), or `NOT_VALIDATED` (`--no-validate`).
488
+ - A derivable dataset whose source yields no rows (e.g. no `InvoiceId`, or an empty Contract
489
+ Commitment source) is reported `NOT_PRODUCED` instead of writing a headerless empty file.
490
+
491
+ ### Fixed — FOCUS conformance of generated JSON columns
492
+
493
+ - **`ContractApplied`** now uses the spec-correct FOCUS 1.3 object schema: a top-level
494
+ `Elements` array with keys `ContractID`, `ContractCommitmentID`,
495
+ `ContractCommitmentAppliedCost`, `ContractCommitmentAppliedQuantity`,
496
+ `ContractCommitmentAppliedUnit` (previously the wrong `ContractId`/`AppliedCost`… names
497
+ and casing). Applied cost/quantity are emitted as JSON **numbers**.
498
+ - **`AllocatedMethodDetails`** emits `AllocatedRatio` and `UsageQuantity` as JSON numbers
499
+ (were quoted strings).
500
+ - **`SkuPriceDetails`** treats `StorageClass` and `Redundancy` as FOCUS-defined keys
501
+ (unprefixed), completing the 13-key FOCUS-defined set (were incorrectly `x_`-prefixed).
502
+ - **`InvoiceDetailGrain`** (derived Invoice Detail) uses `x_`-prefixed custom keys, as
503
+ required for non-FOCUS-defined Key-Value keys.
504
+
505
+ *Because generated data changed, the byte-for-byte output of a given `(rows, seed)` has a
506
+ new baseline as of 0.2.0. Generation remains deterministic.*
507
+
508
+ ### Fixed — linter
509
+
510
+ - **Scientific notation** is now accepted per FOCUS `NumericFormat` (E-notation `mEn`,
511
+ negative-only exponent sign; `35.2E-7` valid, `35.2E+7` invalid). The old regex rejected
512
+ all scientific notation. Numeric literals reject leading zeros (`01`), which are invalid
513
+ JSON numbers.
514
+ - The **`x_` custom-key rule** is enforced across JSON columns via a FOCUS-defined key
515
+ registry (`SkuPriceDetails`, `InvoiceDetailGrain`, `ContractApplied`,
516
+ `AllocatedMethodDetails`, `CommitmentProgramEligibilityDetails`), covering both element
517
+ keys **and top-level custom keys** of array-of-objects columns, and is **not** applied to
518
+ `Tags`/`AllocatedTags` (arbitrary tag keys are allowed).
519
+ - `ContractApplied` is structurally validated (Elements, keys, types). Quoted numeric
520
+ strings for metric fields are rejected (the schema types them as JSON numbers), and the
521
+ FOCUS 1.4 metric exclusivity (`ContractAppliedObjectSchema` `oneOf`: cost *xor*
522
+ quantity+unit) is enforced.
523
+
524
+ ### Changed
525
+
526
+ - The internal validator is repositioned as a **structural + semantic linter**:
527
+ `validate_focus_1_4` → **`lint_focus_1_4_structure`**, with explicit levels
528
+ (`STRUCTURAL_VALID`, `SEMANTIC_VALID`); it never asserts cross-dataset or official
529
+ conformance. `validate_focus_1_4` is retained as a **deprecated alias**.
530
+ - Conversion of `ContractApplied` from a 1.3 source now **migrates** it to the 1.4 schema:
531
+ re-cased identifier keys, and — since 1.3 permits all metrics but 1.4's `oneOf` does not
532
+ — a 1.3 element carrying both cost and quantity keeps the cost branch, preserving
533
+ quantity/unit losslessly as `x_ContractCommitmentAppliedQuantity`/`…Unit`.
534
+ - New typed API `focus_data_toolkit.convert.contract_applied` (parse / validate / migrate).
535
+ - Fixed a whitespace-mismatch bug in the Cost-and-Usage → Invoice Detail back-link key.
536
+ - README/docs no longer describe the linter as a full conformance validator.
537
+
538
+ <!-- Reference links. 0.2.0/0.3.0 were pre-release development milestones and were never tagged
539
+ or published, so only the first public release (0.9.0) has a tag link. -->
540
+ [Unreleased]: https://github.com/guymano/focus-data-toolkit/compare/v0.11.0rc1...HEAD
541
+ [0.11.0rc1]: https://github.com/guymano/focus-data-toolkit/compare/v0.9.0...v0.11.0rc1
542
+ [0.9.0]: https://github.com/guymano/focus-data-toolkit/releases/tag/v0.9.0
@@ -0,0 +1,111 @@
1
+ # Contributing to focus-data-toolkit
2
+
3
+ Thanks for your interest in improving `focus-data-toolkit`! This is a
4
+ community-maintained toolkit for generating provider-realistic FOCUS sample
5
+ data, converting it toward FOCUS 1.4, and linting/validating it. Contributions —
6
+ bug reports, FOCUS-conformance fixes, docs, and features — are welcome.
7
+
8
+ By contributing, you agree that your contributions are licensed under the
9
+ project's [MIT License](LICENSE).
10
+
11
+ ## Reporting issues
12
+
13
+ - **Security vulnerabilities:** do **not** open a public issue. Follow
14
+ [SECURITY.md](SECURITY.md) (GitHub Private Vulnerability Reporting).
15
+ - **FOCUS conformance / correctness bugs** (wrong column, value, rounding, or
16
+ validation verdict) and **feature requests:** open a normal
17
+ [issue](https://github.com/guymano/focus-data-toolkit/issues) using the
18
+ templates. Please include the toolkit version, Python version, platform, and a
19
+ **synthetic** minimal reproduction.
20
+ - **Never include real client/billing data** in an issue, PR, test, or fixture.
21
+ Use the synthetic generators or hand-made minimal fixtures (see "No client
22
+ data" below).
23
+
24
+ ## Development setup
25
+
26
+ Requires Python 3.11+ (3.12+ to exercise the optional `[validator]` extra).
27
+
28
+ ```bash
29
+ git clone https://github.com/guymano/focus-data-toolkit && cd focus-data-toolkit
30
+
31
+ # Option A — pip
32
+ python -m venv .venv && . .venv/bin/activate
33
+ pip install -e '.[dev]' # dev pulls pyarrow + tzdata + release/type tooling
34
+
35
+ # Option B — uv (matches CI; installs from the committed lock)
36
+ uv sync --extra dev
37
+ ```
38
+
39
+ ## Checks to run before opening a PR
40
+
41
+ CI runs these across an OS/Python matrix; run them locally first:
42
+
43
+ ```bash
44
+ ruff check src tests # lint
45
+ mypy src/focus_data_toolkit # types (gate; see pyproject for scoped excludes)
46
+ pytest -q # default suite (fast, hermetic)
47
+ pytest -m slow # large-scale / bounded-memory tests
48
+ pytest -m packaging # build + install a wheel/sdist (needs [release] extra)
49
+
50
+ python scripts/check_pinned_actions.py # every GitHub Action is SHA/digest pinned
51
+ python scripts/verify_model_provenance.py # the embedded FOCUS model matches its provenance
52
+ ```
53
+
54
+ A PR is expected to be green on lint, types, and the default test suite, and to
55
+ keep test coverage at or above the configured floor.
56
+
57
+ ## Determinism & reproducibility (important)
58
+
59
+ The synthetic generators and the converter are **deterministic**: for an
60
+ identical toolkit version and identical parameters (`provider`, `focus_version`,
61
+ `rows`, `seed`, options), the output bytes are stable. Please preserve this:
62
+
63
+ - Keep generator draws in a fixed order; do not introduce a clock, real RNG, or
64
+ environment-dependent behaviour into generation/conversion.
65
+ - Golden snapshots live in `tests/fixtures/golden/`:
66
+ - `compatibility_golden/` — outputs verified correct against the FOCUS spec.
67
+ These are a **byte-for-byte** intra-version contract.
68
+ - `correctness_migration/` — cases with a *known* prior defect: the old output
69
+ is archived for comparison, and the **new** output is the spec-correct one.
70
+ - If a change **deliberately** alters correct output (a legitimate FOCUS fix),
71
+ regenerate the affected `compatibility_golden` snapshot **in the same PR**, and
72
+ add a `CHANGELOG.md` entry noting the new byte baseline. Byte identity is only
73
+ promised within an exact version — see [docs/versioning.md](docs/versioning.md).
74
+ - If a diff changes a snapshot you did **not** intend to touch, treat it as a
75
+ regression and investigate before updating the fixture.
76
+
77
+ ## Changing the embedded FOCUS model
78
+
79
+ The FOCUS 1.4 model JSON (`src/focus_data_toolkit/model/focus_1_4_model.json`) is
80
+ a generated artifact, not a hand-edited file. To change it:
81
+
82
+ 1. Update the source workbook and/or `tools/extract_focus_1_4_model.py`.
83
+ 2. Re-run the extractor: `python tools/extract_focus_1_4_model.py <workbook.xlsx>`.
84
+ 3. Update `src/focus_data_toolkit/model/model_provenance.json` (the
85
+ `output.sha256`, `output.bytes`, and `generator.script_sha256` fields).
86
+ 4. Run `python scripts/verify_model_provenance.py` until it passes.
87
+
88
+ See [docs/model-provenance.md](docs/model-provenance.md) for the full process
89
+ and the `partial` → `complete` provenance gate.
90
+
91
+ ## No client data
92
+
93
+ This project must never ship, test against, or accept real cloud-billing data.
94
+ All fixtures and generated data are synthetic. A test
95
+ (`tests/test_fixtures_are_synthetic.py`) scans committed fixtures for patterns
96
+ that look like real account identifiers or PII; keep it green.
97
+
98
+ ## Pull requests
99
+
100
+ - Branch from `main`, keep PRs focused, and open them as **draft** until ready.
101
+ - Add or update tests for behaviour changes.
102
+ - Update `CHANGELOG.md` under `## [Unreleased]` (Keep a Changelog format).
103
+ - PRs touching workflows, packaging, the FOCUS model, or security surfaces
104
+ request a review from the code owner (see [CODEOWNERS](.github/CODEOWNERS)).
105
+ - Write clear, imperative commit messages explaining the *why*.
106
+
107
+ ## Code of conduct
108
+
109
+ Please be respectful and constructive. Harassment or discrimination of any kind
110
+ is not tolerated. Maintainers may remove comments, commits, or contributions
111
+ that violate this expectation.