focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,106 @@
1
+ """Value provenance (lineage) for converted FOCUS 1.4 columns.
2
+
3
+ Every produced column is classified by *how* its value was obtained. The headline
4
+ classification is a property of the conversion **rule** for a column, so it is
5
+ deterministic and drives both the manifest and the strict-mode gating: in ``STRICT``
6
+ mode a dataset is produced only when every Mandatory non-nullable column has a
7
+ **factual** lineage (not assumed / unavailable).
8
+
9
+ Some rules act differently per row (e.g. a backfill only touches null source
10
+ values). For those columns the headline rule stays the *weakest* lineage the rule
11
+ can produce (conservative for gating), and a :class:`LineageCounters` accumulator
12
+ records how many values actually took each lineage — surfaced in the manifest as
13
+ ``lineage_summary`` so a column-level label never hides the per-value mix.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from collections import Counter, defaultdict
19
+ from dataclasses import dataclass
20
+ from enum import StrEnum
21
+
22
+
23
+ class Lineage(StrEnum):
24
+ OBSERVED = "OBSERVED" # value present directly in the source
25
+ RENAMED = "RENAMED" # copied from an equivalent column, name change only
26
+ DERIVED = "DERIVED" # computed exactly and verifiably from source values
27
+ ENRICHED = "ENRICHED" # from a complementary authoritative source/context
28
+ ASSUMED = "ASSUMED" # hypothetical value (not a source fact)
29
+ UNAVAILABLE = "UNAVAILABLE" # absent and not derivable (emitted null)
30
+
31
+
32
+ # Lineages that represent real, non-fabricated data. A Mandatory non-nullable column
33
+ # whose lineage is NOT factual blocks strict production of its dataset.
34
+ FACTUAL_LINEAGES: frozenset[Lineage] = frozenset(
35
+ {Lineage.OBSERVED, Lineage.RENAMED, Lineage.DERIVED, Lineage.ENRICHED}
36
+ )
37
+
38
+
39
+ @dataclass(frozen=True)
40
+ class ColumnRule:
41
+ """How one target column's value is obtained during conversion."""
42
+
43
+ lineage: Lineage
44
+ source: str | None = None
45
+ note: str | None = None
46
+
47
+ @property
48
+ def is_factual(self) -> bool:
49
+ return self.lineage in FACTUAL_LINEAGES
50
+
51
+ @property
52
+ def is_assumed(self) -> bool:
53
+ return self.lineage is Lineage.ASSUMED
54
+
55
+ def as_dict(self) -> dict[str, str]:
56
+ out: dict[str, str] = {"lineage": self.lineage.value}
57
+ if self.source:
58
+ out["source"] = self.source
59
+ if self.note:
60
+ out["note"] = self.note
61
+ return out
62
+
63
+
64
+ def strict_blockers(provenance: dict[str, ColumnRule], columns: dict[str, dict]) -> list[str]:
65
+ """Return the Mandatory non-nullable columns that block strict production.
66
+
67
+ ``columns`` is the model's column spec map for the dataset. A column blocks when it
68
+ is Mandatory and non-nullable and its lineage is not factual (assumed, unavailable,
69
+ or missing from ``provenance``).
70
+ """
71
+ blockers: list[str] = []
72
+ for col, spec in columns.items():
73
+ if spec.get("feature_level") != "Mandatory" or spec.get("allows_nulls", True):
74
+ continue
75
+ rule = provenance.get(col)
76
+ if rule is None or not rule.is_factual:
77
+ blockers.append(col)
78
+ return sorted(blockers)
79
+
80
+
81
+ def has_assumptions(provenance: dict[str, ColumnRule]) -> bool:
82
+ return any(rule.is_assumed for rule in provenance.values())
83
+
84
+
85
+ class LineageCounters:
86
+ """Per-column counts of the lineage each emitted value actually took.
87
+
88
+ Bounded (columns x lineage categories) and deterministic, so the eager and
89
+ streaming pipelines produce identical summaries for the same input.
90
+ """
91
+
92
+ def __init__(self) -> None:
93
+ self._counts: defaultdict[str, Counter[str]] = defaultdict(Counter)
94
+
95
+ def record(self, column: str, lineage: Lineage, n: int = 1) -> None:
96
+ self._counts[column][lineage.value] += n
97
+
98
+ def summary(self) -> dict[str, dict[str, int]]:
99
+ """Sorted ``{column: {lineage: count}}`` (deterministic manifest payload)."""
100
+ return {
101
+ column: {lineage: count for lineage, count in sorted(counts.items())}
102
+ for column, counts in sorted(self._counts.items())
103
+ }
104
+
105
+ def __bool__(self) -> bool:
106
+ return bool(self._counts)
@@ -0,0 +1 @@
1
+ # PEP 561 marker: focus-data-toolkit ships inline type annotations.
@@ -0,0 +1,243 @@
1
+ """Runtime configuration: working directory + disk budgets, read once from the environment.
2
+
3
+ A streaming conversion uses **two filesystems**, budgeted separately:
4
+
5
+ * the **work** filesystem — scratch state (the SQLite aggregation index and the bundle-validation
6
+ spill), relocatable via ``FOCUS_TOOLKIT_WORK_DIR`` to a faster/larger disk;
7
+ * the **output** filesystem — the atomic staging directory and the final files, pinned to the
8
+ parent of ``--out`` by the publish rename (same-``st_dev`` requirement), so it cannot be moved.
9
+
10
+ Environment variables:
11
+
12
+ * ``FOCUS_TOOLKIT_WORK_DIR`` — directory for scratch state (default: alongside the output).
13
+ * ``FOCUS_TOOLKIT_MAX_WORK_BYTES`` — cap on scratch bytes; exceeding it fails the run (FDT-IO-006).
14
+ * ``FOCUS_TOOLKIT_MIN_WORK_FREE_BYTES`` — refuse/abort if the work filesystem free space drops below.
15
+ * ``FOCUS_TOOLKIT_MIN_OUTPUT_FREE_BYTES`` — refuse/abort if the output filesystem free space drops below.
16
+ * ``FOCUS_TOOLKIT_LOG_LEVEL`` — level for the ``focus_data_toolkit`` logger (default: WARNING).
17
+
18
+ Sizes accept ``128MB`` / ``512KB`` / ``2GB`` style suffixes or a plain byte count. The pre-flight
19
+ estimate is **best-effort with a safety margin** derived from the input size — never a precise
20
+ prediction of the space a conversion will consume.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import logging
26
+ import os
27
+ import shutil
28
+ from collections.abc import Mapping, Sequence
29
+ from dataclasses import dataclass
30
+ from pathlib import Path
31
+
32
+ from focus_data_toolkit.errors import Diagnostic, Severity
33
+
34
+ _InputPaths = Sequence["str | os.PathLike[str] | None"]
35
+
36
+ _ENV_WORK_DIR = "FOCUS_TOOLKIT_WORK_DIR"
37
+ _ENV_MAX_WORK = "FOCUS_TOOLKIT_MAX_WORK_BYTES"
38
+ _ENV_MIN_WORK_FREE = "FOCUS_TOOLKIT_MIN_WORK_FREE_BYTES"
39
+ _ENV_MIN_OUTPUT_FREE = "FOCUS_TOOLKIT_MIN_OUTPUT_FREE_BYTES"
40
+ _ENV_LOG_LEVEL = "FOCUS_TOOLKIT_LOG_LEVEL"
41
+
42
+ # CSV 1.4 output can be somewhat larger than the source, so the pre-flight uses a margin on the
43
+ # input size. Parquet output is smaller, making the estimate a conservative upper bound there.
44
+ _OUTPUT_ESTIMATE_MARGIN = 1.3
45
+
46
+ _SIZE_SUFFIXES = (("KB", 1000), ("MB", 1000**2), ("GB", 1000**3), ("TB", 1000**4), ("B", 1))
47
+
48
+
49
+ class ResourceLimitError(Exception):
50
+ """Raised when a disk budget / free-space check fails (the CLI maps it to exit code 5).
51
+
52
+ Carries the structured :class:`~focus_data_toolkit.errors.Diagnostic` (``FDT-IO-005`` for the
53
+ output filesystem, ``FDT-IO-006`` for the work filesystem / temp budget) so callers can
54
+ render it uniformly. It is **not** a ``ConversionError``: raising it inside the atomic output
55
+ context still removes the staging directory (cleanup keys on "not committed", not on the
56
+ exception type), so nothing partial is ever published.
57
+ """
58
+
59
+ def __init__(self, diagnostic: Diagnostic) -> None:
60
+ super().__init__(diagnostic.message)
61
+ self.diagnostic = diagnostic
62
+
63
+
64
+ def parse_size(value: str | None) -> int | None:
65
+ """Parse a byte size (``128MB`` / ``512KB`` / ``2GB`` / plain count). ``None``/empty -> ``None``.
66
+
67
+ Raises ``ValueError`` on a malformed value.
68
+ """
69
+ if value is None:
70
+ return None
71
+ text = value.strip().upper()
72
+ if not text:
73
+ return None
74
+ for suffix, mult in _SIZE_SUFFIXES:
75
+ if text.endswith(suffix):
76
+ return int(float(text[: -len(suffix)]) * mult)
77
+ return int(text)
78
+
79
+
80
+ def _size_from(env: Mapping[str, str], name: str) -> int | None:
81
+ """Parse a size env var, ignoring (not crashing on) a malformed value."""
82
+ try:
83
+ return parse_size(env.get(name))
84
+ except ValueError:
85
+ return None
86
+
87
+
88
+ @dataclass(frozen=True)
89
+ class RuntimeConfig:
90
+ """Resolved runtime configuration (see the module docstring for the env vars)."""
91
+
92
+ work_dir: Path | None = None
93
+ max_work_bytes: int | None = None
94
+ min_work_free_bytes: int | None = None
95
+ min_output_free_bytes: int | None = None
96
+ log_level: str | None = None
97
+
98
+ @classmethod
99
+ def from_env(cls, env: Mapping[str, str] | None = None) -> RuntimeConfig:
100
+ env = os.environ if env is None else env
101
+ work_dir = env.get(_ENV_WORK_DIR)
102
+ return cls(
103
+ work_dir=Path(work_dir) if work_dir else None,
104
+ max_work_bytes=_size_from(env, _ENV_MAX_WORK),
105
+ min_work_free_bytes=_size_from(env, _ENV_MIN_WORK_FREE),
106
+ min_output_free_bytes=_size_from(env, _ENV_MIN_OUTPUT_FREE),
107
+ log_level=env.get(_ENV_LOG_LEVEL) or None,
108
+ )
109
+
110
+ def apply_logging(self) -> None:
111
+ """Configure the ``focus_data_toolkit`` logger from ``FOCUS_TOOLKIT_LOG_LEVEL`` (if set)."""
112
+ if not self.log_level:
113
+ return
114
+ level = logging.getLevelName(self.log_level.strip().upper())
115
+ if isinstance(level, int):
116
+ logging.getLogger("focus_data_toolkit").setLevel(level)
117
+
118
+
119
+ def _free(path: Path) -> int | None:
120
+ try:
121
+ return shutil.disk_usage(path).free
122
+ except OSError:
123
+ return None
124
+
125
+
126
+ def _estimate_output_bytes(input_paths: _InputPaths) -> int:
127
+ total = 0
128
+ for raw in input_paths:
129
+ if not raw:
130
+ continue
131
+ try:
132
+ total += Path(raw).stat().st_size
133
+ except OSError:
134
+ continue
135
+ return int(total * _OUTPUT_ESTIMATE_MARGIN)
136
+
137
+
138
+ def _io_error(code: str, message: str, path: Path, **context: str) -> ResourceLimitError:
139
+ ctx = {"path": str(path), **context}
140
+ return ResourceLimitError(
141
+ Diagnostic(code=code, severity=Severity.ERROR, message=message, context=ctx)
142
+ )
143
+
144
+
145
+ def work_run_dir(config: RuntimeConfig, run_id: str) -> Path | None:
146
+ """Per-run scratch subdirectory under ``WORK_DIR`` (created), or ``None`` when unset.
147
+
148
+ Relocated scratch is scoped by the conversion's run id (``WORK_DIR/fdt-<run_id>/``) so that
149
+ concurrent runs sharing a single ``FOCUS_TOOLKIT_WORK_DIR`` never collide on the same SQLite
150
+ files. The directory lives outside the atomic staging dir, so the caller removes it on exit.
151
+ """
152
+ if config.work_dir is None:
153
+ return None
154
+ run_dir = config.work_dir / f"fdt-{run_id}"
155
+ run_dir.mkdir(parents=True, exist_ok=True)
156
+ return run_dir
157
+
158
+
159
+ def preflight(config: RuntimeConfig, out_parent: Path, input_paths: _InputPaths) -> None:
160
+ """Best-effort pre-flight run *before* any staging; raises on a clear shortfall.
161
+
162
+ Estimates the output need from the input size (with a margin) and checks both the output and
163
+ work filesystems against their free space and configured minimums. The estimate is deliberately
164
+ rough — it guards against the obvious "nowhere near enough disk" case, not against every run.
165
+ """
166
+ out_parent = Path(out_parent)
167
+ free_out = _free(out_parent)
168
+ if free_out is not None:
169
+ # Require room for the bytes we expect to write AND the reserve that must remain free
170
+ # afterwards — not the larger of the two (which would let a run drive free space below
171
+ # the configured minimum and only fail later during an in-run check).
172
+ need = _estimate_output_bytes(input_paths) + (config.min_output_free_bytes or 0)
173
+ if need and free_out < need:
174
+ raise _io_error(
175
+ "FDT-IO-005",
176
+ f"insufficient free space on the output filesystem at {out_parent}: "
177
+ f"need ~{need} bytes, {free_out} free",
178
+ out_parent,
179
+ needed=str(need),
180
+ free=str(free_out),
181
+ )
182
+ work_dir = config.work_dir or out_parent
183
+ free_work = _free(Path(work_dir))
184
+ if (
185
+ free_work is not None
186
+ and config.min_work_free_bytes is not None
187
+ and free_work < config.min_work_free_bytes
188
+ ):
189
+ raise _io_error(
190
+ "FDT-IO-006",
191
+ f"insufficient free space on the work filesystem at {work_dir}: "
192
+ f"need {config.min_work_free_bytes} bytes, {free_work} free",
193
+ Path(work_dir),
194
+ needed=str(config.min_work_free_bytes),
195
+ free=str(free_work),
196
+ )
197
+
198
+
199
+ def enforce_limits(
200
+ config: RuntimeConfig, out_parent: Path, work_dir: Path, scratch_bytes: int
201
+ ) -> None:
202
+ """In-run enforcement (called periodically): min free space on both filesystems + work budget."""
203
+ if config.min_output_free_bytes is not None:
204
+ free_out = _free(Path(out_parent))
205
+ if free_out is not None and free_out < config.min_output_free_bytes:
206
+ raise _io_error(
207
+ "FDT-IO-005",
208
+ f"output filesystem free space fell below the configured minimum at {out_parent}: "
209
+ f"{free_out} < {config.min_output_free_bytes} bytes",
210
+ Path(out_parent),
211
+ free=str(free_out),
212
+ minimum=str(config.min_output_free_bytes),
213
+ )
214
+ if config.min_work_free_bytes is not None:
215
+ free_work = _free(Path(work_dir))
216
+ if free_work is not None and free_work < config.min_work_free_bytes:
217
+ raise _io_error(
218
+ "FDT-IO-006",
219
+ f"work filesystem free space fell below the configured minimum at {work_dir}: "
220
+ f"{free_work} < {config.min_work_free_bytes} bytes",
221
+ Path(work_dir),
222
+ free=str(free_work),
223
+ minimum=str(config.min_work_free_bytes),
224
+ )
225
+ if config.max_work_bytes is not None and scratch_bytes > config.max_work_bytes:
226
+ raise _io_error(
227
+ "FDT-IO-006",
228
+ f"temporary work budget exceeded: {scratch_bytes} > {config.max_work_bytes} bytes "
229
+ "(FOCUS_TOOLKIT_MAX_WORK_BYTES)",
230
+ Path(work_dir),
231
+ used=str(scratch_bytes),
232
+ budget=str(config.max_work_bytes),
233
+ )
234
+
235
+
236
+ __all__ = [
237
+ "ResourceLimitError",
238
+ "RuntimeConfig",
239
+ "enforce_limits",
240
+ "parse_size",
241
+ "preflight",
242
+ "work_run_dir",
243
+ ]
@@ -0,0 +1,17 @@
1
+ """FOCUS schema knowledge: per-(dataset, version) column registry and detection.
2
+
3
+ The registry (:mod:`focus_data_toolkit.schema.registry`) derives the normative column
4
+ set of each FOCUS dataset at each supported version from the committed FOCUS 1.4 model
5
+ (every column carries its introduction ``version``) plus a small table of columns removed
6
+ by 1.4. The detector (:mod:`focus_data_toolkit.schema.detection`) uses it to identify the
7
+ dataset and version of an arbitrary header row, with a confidence assessment.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from focus_data_toolkit.schema.detection import (
13
+ SchemaDetectionResult,
14
+ detect_focus_schema,
15
+ )
16
+
17
+ __all__ = ["SchemaDetectionResult", "detect_focus_schema"]
@@ -0,0 +1,274 @@
1
+ """Identify the FOCUS dataset and version of an arbitrary header row.
2
+
3
+ The previous approach tested a header against four marker columns and returned only
4
+ ``"1.2"``/``"1.3"``. That mis-detects a 1.3 export missing an optional 1.3 column as 1.2,
5
+ never recognises 1.4, and cannot tell the four datasets apart.
6
+
7
+ :func:`detect_focus_schema` instead scores the header against every ``(dataset, version)``
8
+ schema in the registry (present columns, absent-but-expected columns, and FOCUS columns
9
+ that belong to the dataset but a *different* version — the hybrid signal), and reports a
10
+ confidence plus the exact discrepancies. ``x_``-prefixed extension columns never count
11
+ against a match; unknown non-``x_`` columns are surfaced separately.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from collections.abc import Iterable
17
+ from dataclasses import dataclass
18
+
19
+ from focus_data_toolkit.schema import registry
20
+
21
+ CONF_HIGH = "HIGH"
22
+ CONF_MEDIUM = "MEDIUM"
23
+ CONF_LOW = "LOW"
24
+
25
+ # A runner-up candidate whose Jaccard similarity is within this of the best is "ambiguous".
26
+ _AMBIGUITY_DELTA = 0.04
27
+ # Below this best-similarity floor the header is treated as "not FOCUS" (no dataset).
28
+ _DATASET_FLOOR = 0.20
29
+ # When a version is forced, the user's choice is respected down to a much lower floor (a
30
+ # sparse-but-real projection is valid); only near-zero overlap (a non-FOCUS file) is rejected.
31
+ _FORCED_OVERLAP_FLOOR = 0.05
32
+
33
+
34
+ @dataclass(frozen=True)
35
+ class SchemaDetectionResult:
36
+ """Outcome of detecting the FOCUS dataset/version of a header row."""
37
+
38
+ dataset: str | None
39
+ detected_version: str | None
40
+ confidence: str
41
+ exact_match: bool
42
+ score: float
43
+ missing_columns: tuple[str, ...]
44
+ additional_focus_columns: tuple[str, ...]
45
+ extension_columns: tuple[str, ...]
46
+ unknown_columns: tuple[str, ...]
47
+ ambiguous_candidates: tuple[tuple[str, str], ...]
48
+ forced: bool = False
49
+ notes: tuple[str, ...] = ()
50
+
51
+ @property
52
+ def ok(self) -> bool:
53
+ """A confident, unambiguous identification of a supported schema."""
54
+ return (
55
+ self.dataset is not None
56
+ and self.detected_version is not None
57
+ and self.confidence == CONF_HIGH
58
+ )
59
+
60
+ def as_dict(self) -> dict:
61
+ """JSON-serialisable view for the manifest."""
62
+ return {
63
+ "dataset": self.dataset,
64
+ "detected_version": self.detected_version,
65
+ "confidence": self.confidence,
66
+ "exact_match": self.exact_match,
67
+ "score": round(self.score, 4),
68
+ "forced": self.forced,
69
+ "missing_columns": list(self.missing_columns),
70
+ "additional_focus_columns": list(self.additional_focus_columns),
71
+ "extension_columns": list(self.extension_columns),
72
+ "unknown_columns": list(self.unknown_columns),
73
+ "ambiguous_candidates": [list(c) for c in self.ambiguous_candidates],
74
+ "notes": list(self.notes),
75
+ }
76
+
77
+
78
+ @dataclass(frozen=True)
79
+ class _Candidate:
80
+ dataset: str
81
+ version: str
82
+ jaccard: float
83
+ mandatory_coverage: float
84
+ missing: tuple[str, ...]
85
+ additional_focus: tuple[str, ...]
86
+
87
+
88
+ def _score(dataset: str, version: str, headers: frozenset[str]) -> _Candidate:
89
+ expected = registry.version_columns(dataset, version)
90
+ dataset_focus = registry.all_dataset_columns(dataset)
91
+ present = expected & headers
92
+ missing = expected - headers
93
+ additional_focus = (headers & dataset_focus) - expected
94
+ union = expected | (headers & dataset_focus)
95
+ jaccard = len(present) / len(union) if union else 0.0
96
+ mandatory = registry.mandatory_columns(dataset, version)
97
+ mandatory_coverage = len(mandatory & headers) / len(mandatory) if mandatory else 1.0
98
+ return _Candidate(
99
+ dataset=dataset,
100
+ version=version,
101
+ jaccard=jaccard,
102
+ mandatory_coverage=mandatory_coverage,
103
+ missing=tuple(sorted(missing)),
104
+ additional_focus=tuple(sorted(additional_focus)),
105
+ )
106
+
107
+
108
+ def detect_focus_schema(
109
+ headers: Iterable[str],
110
+ *,
111
+ dataset: str | None = None,
112
+ version: str | None = None,
113
+ ) -> SchemaDetectionResult:
114
+ """Detect the FOCUS dataset and version of ``headers``.
115
+
116
+ ``dataset`` and/or ``version`` force the corresponding dimension (they still get scored,
117
+ so a bad forced choice yields a low confidence the caller can reject). Unknown forced
118
+ values raise ``ValueError``.
119
+ """
120
+ all_headers = list(headers)
121
+ # A malformed CSV row with surplus fields makes ``csv.DictReader`` emit a ``None`` key;
122
+ # drop non-string header names (and record it) rather than crashing on ``.startswith``.
123
+ header_list = [h for h in all_headers if isinstance(h, str)]
124
+ malformed = len(all_headers) - len(header_list)
125
+ header_set = frozenset(header_list)
126
+ extension = tuple(sorted(h for h in header_set if h.startswith("x_")))
127
+ focus_all = registry.all_focus_columns()
128
+ unknown = tuple(sorted(h for h in header_set if not h.startswith("x_") and h not in focus_all))
129
+
130
+ forced_dataset = registry.resolve_dataset_name(dataset) if dataset is not None else None
131
+ forced_version = registry.normalize_version(version) if version is not None else None
132
+ forced = forced_dataset is not None or forced_version is not None
133
+
134
+ candidates = registry.candidate_schemas()
135
+ if forced_dataset is not None:
136
+ candidates = [c for c in candidates if c[0] == forced_dataset]
137
+ if forced_version is not None:
138
+ candidates = [c for c in candidates if c[1] == forced_version]
139
+
140
+ if not candidates:
141
+ note = "forced (dataset, version) does not exist in FOCUS"
142
+ return SchemaDetectionResult(
143
+ dataset=forced_dataset,
144
+ detected_version=forced_version,
145
+ confidence=CONF_LOW,
146
+ exact_match=False,
147
+ score=0.0,
148
+ missing_columns=(),
149
+ additional_focus_columns=(),
150
+ extension_columns=extension,
151
+ unknown_columns=unknown,
152
+ ambiguous_candidates=(),
153
+ forced=forced,
154
+ notes=(note,),
155
+ )
156
+
157
+ scored = sorted(
158
+ (_score(d, v, header_set) for d, v in candidates),
159
+ key=lambda c: (c.jaccard, c.mandatory_coverage),
160
+ reverse=True,
161
+ )
162
+ best = scored[0]
163
+
164
+ # Ambiguity: any other candidate whose similarity is within delta of the best.
165
+ ambiguous = tuple(
166
+ (c.dataset, c.version)
167
+ for c in scored[1:]
168
+ if c.jaccard > 0 and best.jaccard - c.jaccard < _AMBIGUITY_DELTA
169
+ )
170
+
171
+ notes: list[str] = []
172
+ dataset_name: str | None = best.dataset
173
+ detected_version: str | None = best.version
174
+
175
+ if not forced and best.jaccard < _DATASET_FLOOR:
176
+ # Header does not resemble any FOCUS schema.
177
+ return SchemaDetectionResult(
178
+ dataset=None,
179
+ detected_version=None,
180
+ confidence=CONF_LOW,
181
+ exact_match=False,
182
+ score=round(best.jaccard, 4),
183
+ missing_columns=(),
184
+ additional_focus_columns=(),
185
+ extension_columns=extension,
186
+ unknown_columns=unknown,
187
+ ambiguous_candidates=(),
188
+ forced=forced,
189
+ notes=("header does not match any known FOCUS dataset/version",),
190
+ )
191
+
192
+ # A FOCUS column that belongs to a *different* dataset (e.g. PaymentTerms on a Cost and
193
+ # Usage header) is neither "additional focus of this dataset" nor "unknown"; count it as a
194
+ # mismatch so it breaks exact_match and caps confidence (strict must not silently drop it).
195
+ foreign_focus = tuple(
196
+ sorted((header_set & focus_all) - registry.all_dataset_columns(best.dataset))
197
+ )
198
+ additional_focus = tuple(sorted(set(best.additional_focus) | set(foreign_focus)))
199
+
200
+ exact_match = not best.missing and not additional_focus and not unknown and not malformed
201
+
202
+ # Confidence.
203
+ if forced_version is not None:
204
+ # The version is locked by the user. Only columns that belong to *another* version of
205
+ # this dataset (``additional_focus``) make the forced version genuinely impossible;
206
+ # merely missing columns are a completeness issue the lint reports, not a reason to
207
+ # override an explicit choice. But a header with essentially no overlap (e.g. a
208
+ # non-FOCUS file) must still be rejected even when a version is forced.
209
+ if best.jaccard < _FORCED_OVERLAP_FLOOR:
210
+ confidence = CONF_LOW
211
+ notes.append(
212
+ f"header has no meaningful overlap with the forced schema "
213
+ f"{best.dataset} {forced_version}"
214
+ )
215
+ elif additional_focus:
216
+ confidence = CONF_LOW
217
+ notes.append(
218
+ f"header is incompatible with forced version {forced_version} "
219
+ f"(columns not in this schema: {', '.join(additional_focus)})"
220
+ )
221
+ elif best.missing or best.mandatory_coverage < 0.999:
222
+ confidence = CONF_MEDIUM # compatible but incomplete (lint will flag specifics)
223
+ else:
224
+ confidence = CONF_HIGH
225
+ elif (
226
+ best.jaccard >= 0.9
227
+ and best.mandatory_coverage >= 0.999
228
+ and not ambiguous
229
+ and not unknown
230
+ and not additional_focus
231
+ and not malformed
232
+ ):
233
+ confidence = CONF_HIGH
234
+ elif best.jaccard >= 0.6 and best.mandatory_coverage >= 0.8:
235
+ confidence = CONF_MEDIUM
236
+ else:
237
+ confidence = CONF_LOW
238
+
239
+ if malformed:
240
+ notes.append(f"{malformed} malformed (non-string) header name(s) ignored")
241
+ if foreign_focus:
242
+ notes.append(
243
+ "FOCUS columns from another dataset present: " + ", ".join(foreign_focus)
244
+ )
245
+ if ambiguous and confidence == CONF_HIGH:
246
+ confidence = CONF_MEDIUM
247
+ if unknown and confidence == CONF_HIGH:
248
+ confidence = CONF_MEDIUM
249
+ notes.append(f"{len(unknown)} unknown non-x_ column(s) present")
250
+ elif unknown:
251
+ notes.append(f"{len(unknown)} unknown non-x_ column(s) present")
252
+ if best.additional_focus:
253
+ notes.append(
254
+ "columns from another FOCUS version present: " + ", ".join(best.additional_focus)
255
+ )
256
+ if ambiguous:
257
+ notes.append(
258
+ "close alternative schema(s): " + ", ".join(f"{d} {v}" for d, v in ambiguous)
259
+ )
260
+
261
+ return SchemaDetectionResult(
262
+ dataset=dataset_name,
263
+ detected_version=detected_version,
264
+ confidence=confidence,
265
+ exact_match=exact_match,
266
+ score=round(best.jaccard, 4),
267
+ missing_columns=best.missing,
268
+ additional_focus_columns=additional_focus,
269
+ extension_columns=extension,
270
+ unknown_columns=unknown,
271
+ ambiguous_candidates=ambiguous,
272
+ forced=forced,
273
+ notes=tuple(notes),
274
+ )