focus-data-toolkit 0.11.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- focus_data_toolkit/__init__.py +69 -0
- focus_data_toolkit/__main__.py +6 -0
- focus_data_toolkit/_version.py +8 -0
- focus_data_toolkit/cli.py +968 -0
- focus_data_toolkit/context/__init__.py +88 -0
- focus_data_toolkit/context/billing.py +54 -0
- focus_data_toolkit/context/provider.py +90 -0
- focus_data_toolkit/convert/__init__.py +708 -0
- focus_data_toolkit/convert/billing_period.py +65 -0
- focus_data_toolkit/convert/contract_applied.py +235 -0
- focus_data_toolkit/convert/contract_commitment.py +182 -0
- focus_data_toolkit/convert/cost_and_usage.py +179 -0
- focus_data_toolkit/convert/detect.py +39 -0
- focus_data_toolkit/convert/invoice_detail.py +199 -0
- focus_data_toolkit/convert/streaming.py +1030 -0
- focus_data_toolkit/errors.py +145 -0
- focus_data_toolkit/focus_json.py +68 -0
- focus_data_toolkit/generators/__init__.py +61 -0
- focus_data_toolkit/generators/_shim.py +43 -0
- focus_data_toolkit/generators/engine/__init__.py +14 -0
- focus_data_toolkit/generators/engine/context.py +12 -0
- focus_data_toolkit/generators/engine/determinism.py +117 -0
- focus_data_toolkit/generators/engine/json_focus.py +63 -0
- focus_data_toolkit/generators/engine/ladder.py +71 -0
- focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
- focus_data_toolkit/generators/engine/serialize.py +151 -0
- focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
- focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
- focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
- focus_data_toolkit/generators/providers/__init__.py +29 -0
- focus_data_toolkit/generators/providers/aws.py +186 -0
- focus_data_toolkit/generators/providers/azure.py +191 -0
- focus_data_toolkit/generators/providers/gcp.py +194 -0
- focus_data_toolkit/generators/providers/profile.py +123 -0
- focus_data_toolkit/generators/scenarios.py +178 -0
- focus_data_toolkit/generators/versions/__init__.py +17 -0
- focus_data_toolkit/generators/versions/adapter.py +41 -0
- focus_data_toolkit/generators/versions/v1_2.py +111 -0
- focus_data_toolkit/generators/versions/v1_3.py +154 -0
- focus_data_toolkit/io/__init__.py +1 -0
- focus_data_toolkit/io/atomic_writer.py +462 -0
- focus_data_toolkit/io/csv_io.py +128 -0
- focus_data_toolkit/io/parquet_io.py +528 -0
- focus_data_toolkit/io/records.py +92 -0
- focus_data_toolkit/io/row_source.py +117 -0
- focus_data_toolkit/lifecycle.py +342 -0
- focus_data_toolkit/manifest.py +114 -0
- focus_data_toolkit/model/__init__.py +43 -0
- focus_data_toolkit/model/capabilities.py +66 -0
- focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
- focus_data_toolkit/model/focus_1_4_model.json +1913 -0
- focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
- focus_data_toolkit/model/focus_json_keys.py +112 -0
- focus_data_toolkit/model/iso_4217_currencies.json +23 -0
- focus_data_toolkit/model/json_schema_check.py +205 -0
- focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
- focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
- focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
- focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
- focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
- focus_data_toolkit/model/model_provenance.json +58 -0
- focus_data_toolkit/model/validator.py +498 -0
- focus_data_toolkit/modes.py +18 -0
- focus_data_toolkit/official_validator.py +61 -0
- focus_data_toolkit/progress.py +89 -0
- focus_data_toolkit/provenance.py +106 -0
- focus_data_toolkit/py.typed +1 -0
- focus_data_toolkit/runtime.py +243 -0
- focus_data_toolkit/schema/__init__.py +17 -0
- focus_data_toolkit/schema/detection.py +274 -0
- focus_data_toolkit/schema/registry.py +127 -0
- focus_data_toolkit/storage/__init__.py +1 -0
- focus_data_toolkit/storage/external_index.py +99 -0
- focus_data_toolkit/storage/spill.py +150 -0
- focus_data_toolkit/studio/__init__.py +19 -0
- focus_data_toolkit/studio/app.py +467 -0
- focus_data_toolkit/studio/config.py +42 -0
- focus_data_toolkit/studio/frontend/app.js +214 -0
- focus_data_toolkit/studio/frontend/index.html +101 -0
- focus_data_toolkit/studio/frontend/style.css +60 -0
- focus_data_toolkit/studio/jobs.py +142 -0
- focus_data_toolkit/studio/preview.py +32 -0
- focus_data_toolkit/studio/security.py +125 -0
- focus_data_toolkit/studio/server.py +71 -0
- focus_data_toolkit/supplement/__init__.py +50 -0
- focus_data_toolkit/supplement/adapters/__init__.py +21 -0
- focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
- focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
- focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
- focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
- focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
- focus_data_toolkit/supplement/adapters/registry.py +215 -0
- focus_data_toolkit/supplement/apply.py +318 -0
- focus_data_toolkit/supplement/gaps.py +219 -0
- focus_data_toolkit/supplement/kinds.py +118 -0
- focus_data_toolkit/supplement/loader.py +409 -0
- focus_data_toolkit/supplement/spec.py +74 -0
- focus_data_toolkit/supplement/validate.py +215 -0
- focus_data_toolkit/validate/__init__.py +15 -0
- focus_data_toolkit/validate/allocation.py +333 -0
- focus_data_toolkit/validate/bundle.py +254 -0
- focus_data_toolkit/validate/codes.py +93 -0
- focus_data_toolkit/validate/corrections.py +245 -0
- focus_data_toolkit/validate/reconciliation.py +98 -0
- focus_data_toolkit/validate/referential.py +289 -0
- focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
- focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
- focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
- focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
- focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""Value provenance (lineage) for converted FOCUS 1.4 columns.
|
|
2
|
+
|
|
3
|
+
Every produced column is classified by *how* its value was obtained. The headline
|
|
4
|
+
classification is a property of the conversion **rule** for a column, so it is
|
|
5
|
+
deterministic and drives both the manifest and the strict-mode gating: in ``STRICT``
|
|
6
|
+
mode a dataset is produced only when every Mandatory non-nullable column has a
|
|
7
|
+
**factual** lineage (not assumed / unavailable).
|
|
8
|
+
|
|
9
|
+
Some rules act differently per row (e.g. a backfill only touches null source
|
|
10
|
+
values). For those columns the headline rule stays the *weakest* lineage the rule
|
|
11
|
+
can produce (conservative for gating), and a :class:`LineageCounters` accumulator
|
|
12
|
+
records how many values actually took each lineage — surfaced in the manifest as
|
|
13
|
+
``lineage_summary`` so a column-level label never hides the per-value mix.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from collections import Counter, defaultdict
|
|
19
|
+
from dataclasses import dataclass
|
|
20
|
+
from enum import StrEnum
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class Lineage(StrEnum):
|
|
24
|
+
OBSERVED = "OBSERVED" # value present directly in the source
|
|
25
|
+
RENAMED = "RENAMED" # copied from an equivalent column, name change only
|
|
26
|
+
DERIVED = "DERIVED" # computed exactly and verifiably from source values
|
|
27
|
+
ENRICHED = "ENRICHED" # from a complementary authoritative source/context
|
|
28
|
+
ASSUMED = "ASSUMED" # hypothetical value (not a source fact)
|
|
29
|
+
UNAVAILABLE = "UNAVAILABLE" # absent and not derivable (emitted null)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
# Lineages that represent real, non-fabricated data. A Mandatory non-nullable column
|
|
33
|
+
# whose lineage is NOT factual blocks strict production of its dataset.
|
|
34
|
+
FACTUAL_LINEAGES: frozenset[Lineage] = frozenset(
|
|
35
|
+
{Lineage.OBSERVED, Lineage.RENAMED, Lineage.DERIVED, Lineage.ENRICHED}
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True)
|
|
40
|
+
class ColumnRule:
|
|
41
|
+
"""How one target column's value is obtained during conversion."""
|
|
42
|
+
|
|
43
|
+
lineage: Lineage
|
|
44
|
+
source: str | None = None
|
|
45
|
+
note: str | None = None
|
|
46
|
+
|
|
47
|
+
@property
|
|
48
|
+
def is_factual(self) -> bool:
|
|
49
|
+
return self.lineage in FACTUAL_LINEAGES
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def is_assumed(self) -> bool:
|
|
53
|
+
return self.lineage is Lineage.ASSUMED
|
|
54
|
+
|
|
55
|
+
def as_dict(self) -> dict[str, str]:
|
|
56
|
+
out: dict[str, str] = {"lineage": self.lineage.value}
|
|
57
|
+
if self.source:
|
|
58
|
+
out["source"] = self.source
|
|
59
|
+
if self.note:
|
|
60
|
+
out["note"] = self.note
|
|
61
|
+
return out
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def strict_blockers(provenance: dict[str, ColumnRule], columns: dict[str, dict]) -> list[str]:
|
|
65
|
+
"""Return the Mandatory non-nullable columns that block strict production.
|
|
66
|
+
|
|
67
|
+
``columns`` is the model's column spec map for the dataset. A column blocks when it
|
|
68
|
+
is Mandatory and non-nullable and its lineage is not factual (assumed, unavailable,
|
|
69
|
+
or missing from ``provenance``).
|
|
70
|
+
"""
|
|
71
|
+
blockers: list[str] = []
|
|
72
|
+
for col, spec in columns.items():
|
|
73
|
+
if spec.get("feature_level") != "Mandatory" or spec.get("allows_nulls", True):
|
|
74
|
+
continue
|
|
75
|
+
rule = provenance.get(col)
|
|
76
|
+
if rule is None or not rule.is_factual:
|
|
77
|
+
blockers.append(col)
|
|
78
|
+
return sorted(blockers)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def has_assumptions(provenance: dict[str, ColumnRule]) -> bool:
|
|
82
|
+
return any(rule.is_assumed for rule in provenance.values())
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class LineageCounters:
|
|
86
|
+
"""Per-column counts of the lineage each emitted value actually took.
|
|
87
|
+
|
|
88
|
+
Bounded (columns x lineage categories) and deterministic, so the eager and
|
|
89
|
+
streaming pipelines produce identical summaries for the same input.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
def __init__(self) -> None:
|
|
93
|
+
self._counts: defaultdict[str, Counter[str]] = defaultdict(Counter)
|
|
94
|
+
|
|
95
|
+
def record(self, column: str, lineage: Lineage, n: int = 1) -> None:
|
|
96
|
+
self._counts[column][lineage.value] += n
|
|
97
|
+
|
|
98
|
+
def summary(self) -> dict[str, dict[str, int]]:
|
|
99
|
+
"""Sorted ``{column: {lineage: count}}`` (deterministic manifest payload)."""
|
|
100
|
+
return {
|
|
101
|
+
column: {lineage: count for lineage, count in sorted(counts.items())}
|
|
102
|
+
for column, counts in sorted(self._counts.items())
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
def __bool__(self) -> bool:
|
|
106
|
+
return bool(self._counts)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# PEP 561 marker: focus-data-toolkit ships inline type annotations.
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
"""Runtime configuration: working directory + disk budgets, read once from the environment.
|
|
2
|
+
|
|
3
|
+
A streaming conversion uses **two filesystems**, budgeted separately:
|
|
4
|
+
|
|
5
|
+
* the **work** filesystem — scratch state (the SQLite aggregation index and the bundle-validation
|
|
6
|
+
spill), relocatable via ``FOCUS_TOOLKIT_WORK_DIR`` to a faster/larger disk;
|
|
7
|
+
* the **output** filesystem — the atomic staging directory and the final files, pinned to the
|
|
8
|
+
parent of ``--out`` by the publish rename (same-``st_dev`` requirement), so it cannot be moved.
|
|
9
|
+
|
|
10
|
+
Environment variables:
|
|
11
|
+
|
|
12
|
+
* ``FOCUS_TOOLKIT_WORK_DIR`` — directory for scratch state (default: alongside the output).
|
|
13
|
+
* ``FOCUS_TOOLKIT_MAX_WORK_BYTES`` — cap on scratch bytes; exceeding it fails the run (FDT-IO-006).
|
|
14
|
+
* ``FOCUS_TOOLKIT_MIN_WORK_FREE_BYTES`` — refuse/abort if the work filesystem free space drops below.
|
|
15
|
+
* ``FOCUS_TOOLKIT_MIN_OUTPUT_FREE_BYTES`` — refuse/abort if the output filesystem free space drops below.
|
|
16
|
+
* ``FOCUS_TOOLKIT_LOG_LEVEL`` — level for the ``focus_data_toolkit`` logger (default: WARNING).
|
|
17
|
+
|
|
18
|
+
Sizes accept ``128MB`` / ``512KB`` / ``2GB`` style suffixes or a plain byte count. The pre-flight
|
|
19
|
+
estimate is **best-effort with a safety margin** derived from the input size — never a precise
|
|
20
|
+
prediction of the space a conversion will consume.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import logging
|
|
26
|
+
import os
|
|
27
|
+
import shutil
|
|
28
|
+
from collections.abc import Mapping, Sequence
|
|
29
|
+
from dataclasses import dataclass
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
|
|
32
|
+
from focus_data_toolkit.errors import Diagnostic, Severity
|
|
33
|
+
|
|
34
|
+
_InputPaths = Sequence["str | os.PathLike[str] | None"]
|
|
35
|
+
|
|
36
|
+
_ENV_WORK_DIR = "FOCUS_TOOLKIT_WORK_DIR"
|
|
37
|
+
_ENV_MAX_WORK = "FOCUS_TOOLKIT_MAX_WORK_BYTES"
|
|
38
|
+
_ENV_MIN_WORK_FREE = "FOCUS_TOOLKIT_MIN_WORK_FREE_BYTES"
|
|
39
|
+
_ENV_MIN_OUTPUT_FREE = "FOCUS_TOOLKIT_MIN_OUTPUT_FREE_BYTES"
|
|
40
|
+
_ENV_LOG_LEVEL = "FOCUS_TOOLKIT_LOG_LEVEL"
|
|
41
|
+
|
|
42
|
+
# CSV 1.4 output can be somewhat larger than the source, so the pre-flight uses a margin on the
|
|
43
|
+
# input size. Parquet output is smaller, making the estimate a conservative upper bound there.
|
|
44
|
+
_OUTPUT_ESTIMATE_MARGIN = 1.3
|
|
45
|
+
|
|
46
|
+
_SIZE_SUFFIXES = (("KB", 1000), ("MB", 1000**2), ("GB", 1000**3), ("TB", 1000**4), ("B", 1))
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class ResourceLimitError(Exception):
|
|
50
|
+
"""Raised when a disk budget / free-space check fails (the CLI maps it to exit code 5).
|
|
51
|
+
|
|
52
|
+
Carries the structured :class:`~focus_data_toolkit.errors.Diagnostic` (``FDT-IO-005`` for the
|
|
53
|
+
output filesystem, ``FDT-IO-006`` for the work filesystem / temp budget) so callers can
|
|
54
|
+
render it uniformly. It is **not** a ``ConversionError``: raising it inside the atomic output
|
|
55
|
+
context still removes the staging directory (cleanup keys on "not committed", not on the
|
|
56
|
+
exception type), so nothing partial is ever published.
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
def __init__(self, diagnostic: Diagnostic) -> None:
|
|
60
|
+
super().__init__(diagnostic.message)
|
|
61
|
+
self.diagnostic = diagnostic
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def parse_size(value: str | None) -> int | None:
|
|
65
|
+
"""Parse a byte size (``128MB`` / ``512KB`` / ``2GB`` / plain count). ``None``/empty -> ``None``.
|
|
66
|
+
|
|
67
|
+
Raises ``ValueError`` on a malformed value.
|
|
68
|
+
"""
|
|
69
|
+
if value is None:
|
|
70
|
+
return None
|
|
71
|
+
text = value.strip().upper()
|
|
72
|
+
if not text:
|
|
73
|
+
return None
|
|
74
|
+
for suffix, mult in _SIZE_SUFFIXES:
|
|
75
|
+
if text.endswith(suffix):
|
|
76
|
+
return int(float(text[: -len(suffix)]) * mult)
|
|
77
|
+
return int(text)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _size_from(env: Mapping[str, str], name: str) -> int | None:
|
|
81
|
+
"""Parse a size env var, ignoring (not crashing on) a malformed value."""
|
|
82
|
+
try:
|
|
83
|
+
return parse_size(env.get(name))
|
|
84
|
+
except ValueError:
|
|
85
|
+
return None
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@dataclass(frozen=True)
|
|
89
|
+
class RuntimeConfig:
|
|
90
|
+
"""Resolved runtime configuration (see the module docstring for the env vars)."""
|
|
91
|
+
|
|
92
|
+
work_dir: Path | None = None
|
|
93
|
+
max_work_bytes: int | None = None
|
|
94
|
+
min_work_free_bytes: int | None = None
|
|
95
|
+
min_output_free_bytes: int | None = None
|
|
96
|
+
log_level: str | None = None
|
|
97
|
+
|
|
98
|
+
@classmethod
|
|
99
|
+
def from_env(cls, env: Mapping[str, str] | None = None) -> RuntimeConfig:
|
|
100
|
+
env = os.environ if env is None else env
|
|
101
|
+
work_dir = env.get(_ENV_WORK_DIR)
|
|
102
|
+
return cls(
|
|
103
|
+
work_dir=Path(work_dir) if work_dir else None,
|
|
104
|
+
max_work_bytes=_size_from(env, _ENV_MAX_WORK),
|
|
105
|
+
min_work_free_bytes=_size_from(env, _ENV_MIN_WORK_FREE),
|
|
106
|
+
min_output_free_bytes=_size_from(env, _ENV_MIN_OUTPUT_FREE),
|
|
107
|
+
log_level=env.get(_ENV_LOG_LEVEL) or None,
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
def apply_logging(self) -> None:
|
|
111
|
+
"""Configure the ``focus_data_toolkit`` logger from ``FOCUS_TOOLKIT_LOG_LEVEL`` (if set)."""
|
|
112
|
+
if not self.log_level:
|
|
113
|
+
return
|
|
114
|
+
level = logging.getLevelName(self.log_level.strip().upper())
|
|
115
|
+
if isinstance(level, int):
|
|
116
|
+
logging.getLogger("focus_data_toolkit").setLevel(level)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _free(path: Path) -> int | None:
|
|
120
|
+
try:
|
|
121
|
+
return shutil.disk_usage(path).free
|
|
122
|
+
except OSError:
|
|
123
|
+
return None
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _estimate_output_bytes(input_paths: _InputPaths) -> int:
|
|
127
|
+
total = 0
|
|
128
|
+
for raw in input_paths:
|
|
129
|
+
if not raw:
|
|
130
|
+
continue
|
|
131
|
+
try:
|
|
132
|
+
total += Path(raw).stat().st_size
|
|
133
|
+
except OSError:
|
|
134
|
+
continue
|
|
135
|
+
return int(total * _OUTPUT_ESTIMATE_MARGIN)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _io_error(code: str, message: str, path: Path, **context: str) -> ResourceLimitError:
|
|
139
|
+
ctx = {"path": str(path), **context}
|
|
140
|
+
return ResourceLimitError(
|
|
141
|
+
Diagnostic(code=code, severity=Severity.ERROR, message=message, context=ctx)
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def work_run_dir(config: RuntimeConfig, run_id: str) -> Path | None:
|
|
146
|
+
"""Per-run scratch subdirectory under ``WORK_DIR`` (created), or ``None`` when unset.
|
|
147
|
+
|
|
148
|
+
Relocated scratch is scoped by the conversion's run id (``WORK_DIR/fdt-<run_id>/``) so that
|
|
149
|
+
concurrent runs sharing a single ``FOCUS_TOOLKIT_WORK_DIR`` never collide on the same SQLite
|
|
150
|
+
files. The directory lives outside the atomic staging dir, so the caller removes it on exit.
|
|
151
|
+
"""
|
|
152
|
+
if config.work_dir is None:
|
|
153
|
+
return None
|
|
154
|
+
run_dir = config.work_dir / f"fdt-{run_id}"
|
|
155
|
+
run_dir.mkdir(parents=True, exist_ok=True)
|
|
156
|
+
return run_dir
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def preflight(config: RuntimeConfig, out_parent: Path, input_paths: _InputPaths) -> None:
|
|
160
|
+
"""Best-effort pre-flight run *before* any staging; raises on a clear shortfall.
|
|
161
|
+
|
|
162
|
+
Estimates the output need from the input size (with a margin) and checks both the output and
|
|
163
|
+
work filesystems against their free space and configured minimums. The estimate is deliberately
|
|
164
|
+
rough — it guards against the obvious "nowhere near enough disk" case, not against every run.
|
|
165
|
+
"""
|
|
166
|
+
out_parent = Path(out_parent)
|
|
167
|
+
free_out = _free(out_parent)
|
|
168
|
+
if free_out is not None:
|
|
169
|
+
# Require room for the bytes we expect to write AND the reserve that must remain free
|
|
170
|
+
# afterwards — not the larger of the two (which would let a run drive free space below
|
|
171
|
+
# the configured minimum and only fail later during an in-run check).
|
|
172
|
+
need = _estimate_output_bytes(input_paths) + (config.min_output_free_bytes or 0)
|
|
173
|
+
if need and free_out < need:
|
|
174
|
+
raise _io_error(
|
|
175
|
+
"FDT-IO-005",
|
|
176
|
+
f"insufficient free space on the output filesystem at {out_parent}: "
|
|
177
|
+
f"need ~{need} bytes, {free_out} free",
|
|
178
|
+
out_parent,
|
|
179
|
+
needed=str(need),
|
|
180
|
+
free=str(free_out),
|
|
181
|
+
)
|
|
182
|
+
work_dir = config.work_dir or out_parent
|
|
183
|
+
free_work = _free(Path(work_dir))
|
|
184
|
+
if (
|
|
185
|
+
free_work is not None
|
|
186
|
+
and config.min_work_free_bytes is not None
|
|
187
|
+
and free_work < config.min_work_free_bytes
|
|
188
|
+
):
|
|
189
|
+
raise _io_error(
|
|
190
|
+
"FDT-IO-006",
|
|
191
|
+
f"insufficient free space on the work filesystem at {work_dir}: "
|
|
192
|
+
f"need {config.min_work_free_bytes} bytes, {free_work} free",
|
|
193
|
+
Path(work_dir),
|
|
194
|
+
needed=str(config.min_work_free_bytes),
|
|
195
|
+
free=str(free_work),
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def enforce_limits(
|
|
200
|
+
config: RuntimeConfig, out_parent: Path, work_dir: Path, scratch_bytes: int
|
|
201
|
+
) -> None:
|
|
202
|
+
"""In-run enforcement (called periodically): min free space on both filesystems + work budget."""
|
|
203
|
+
if config.min_output_free_bytes is not None:
|
|
204
|
+
free_out = _free(Path(out_parent))
|
|
205
|
+
if free_out is not None and free_out < config.min_output_free_bytes:
|
|
206
|
+
raise _io_error(
|
|
207
|
+
"FDT-IO-005",
|
|
208
|
+
f"output filesystem free space fell below the configured minimum at {out_parent}: "
|
|
209
|
+
f"{free_out} < {config.min_output_free_bytes} bytes",
|
|
210
|
+
Path(out_parent),
|
|
211
|
+
free=str(free_out),
|
|
212
|
+
minimum=str(config.min_output_free_bytes),
|
|
213
|
+
)
|
|
214
|
+
if config.min_work_free_bytes is not None:
|
|
215
|
+
free_work = _free(Path(work_dir))
|
|
216
|
+
if free_work is not None and free_work < config.min_work_free_bytes:
|
|
217
|
+
raise _io_error(
|
|
218
|
+
"FDT-IO-006",
|
|
219
|
+
f"work filesystem free space fell below the configured minimum at {work_dir}: "
|
|
220
|
+
f"{free_work} < {config.min_work_free_bytes} bytes",
|
|
221
|
+
Path(work_dir),
|
|
222
|
+
free=str(free_work),
|
|
223
|
+
minimum=str(config.min_work_free_bytes),
|
|
224
|
+
)
|
|
225
|
+
if config.max_work_bytes is not None and scratch_bytes > config.max_work_bytes:
|
|
226
|
+
raise _io_error(
|
|
227
|
+
"FDT-IO-006",
|
|
228
|
+
f"temporary work budget exceeded: {scratch_bytes} > {config.max_work_bytes} bytes "
|
|
229
|
+
"(FOCUS_TOOLKIT_MAX_WORK_BYTES)",
|
|
230
|
+
Path(work_dir),
|
|
231
|
+
used=str(scratch_bytes),
|
|
232
|
+
budget=str(config.max_work_bytes),
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
__all__ = [
|
|
237
|
+
"ResourceLimitError",
|
|
238
|
+
"RuntimeConfig",
|
|
239
|
+
"enforce_limits",
|
|
240
|
+
"parse_size",
|
|
241
|
+
"preflight",
|
|
242
|
+
"work_run_dir",
|
|
243
|
+
]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""FOCUS schema knowledge: per-(dataset, version) column registry and detection.
|
|
2
|
+
|
|
3
|
+
The registry (:mod:`focus_data_toolkit.schema.registry`) derives the normative column
|
|
4
|
+
set of each FOCUS dataset at each supported version from the committed FOCUS 1.4 model
|
|
5
|
+
(every column carries its introduction ``version``) plus a small table of columns removed
|
|
6
|
+
by 1.4. The detector (:mod:`focus_data_toolkit.schema.detection`) uses it to identify the
|
|
7
|
+
dataset and version of an arbitrary header row, with a confidence assessment.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from focus_data_toolkit.schema.detection import (
|
|
13
|
+
SchemaDetectionResult,
|
|
14
|
+
detect_focus_schema,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
__all__ = ["SchemaDetectionResult", "detect_focus_schema"]
|
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
"""Identify the FOCUS dataset and version of an arbitrary header row.
|
|
2
|
+
|
|
3
|
+
The previous approach tested a header against four marker columns and returned only
|
|
4
|
+
``"1.2"``/``"1.3"``. That mis-detects a 1.3 export missing an optional 1.3 column as 1.2,
|
|
5
|
+
never recognises 1.4, and cannot tell the four datasets apart.
|
|
6
|
+
|
|
7
|
+
:func:`detect_focus_schema` instead scores the header against every ``(dataset, version)``
|
|
8
|
+
schema in the registry (present columns, absent-but-expected columns, and FOCUS columns
|
|
9
|
+
that belong to the dataset but a *different* version — the hybrid signal), and reports a
|
|
10
|
+
confidence plus the exact discrepancies. ``x_``-prefixed extension columns never count
|
|
11
|
+
against a match; unknown non-``x_`` columns are surfaced separately.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from collections.abc import Iterable
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
|
|
19
|
+
from focus_data_toolkit.schema import registry
|
|
20
|
+
|
|
21
|
+
CONF_HIGH = "HIGH"
|
|
22
|
+
CONF_MEDIUM = "MEDIUM"
|
|
23
|
+
CONF_LOW = "LOW"
|
|
24
|
+
|
|
25
|
+
# A runner-up candidate whose Jaccard similarity is within this of the best is "ambiguous".
|
|
26
|
+
_AMBIGUITY_DELTA = 0.04
|
|
27
|
+
# Below this best-similarity floor the header is treated as "not FOCUS" (no dataset).
|
|
28
|
+
_DATASET_FLOOR = 0.20
|
|
29
|
+
# When a version is forced, the user's choice is respected down to a much lower floor (a
|
|
30
|
+
# sparse-but-real projection is valid); only near-zero overlap (a non-FOCUS file) is rejected.
|
|
31
|
+
_FORCED_OVERLAP_FLOOR = 0.05
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass(frozen=True)
|
|
35
|
+
class SchemaDetectionResult:
|
|
36
|
+
"""Outcome of detecting the FOCUS dataset/version of a header row."""
|
|
37
|
+
|
|
38
|
+
dataset: str | None
|
|
39
|
+
detected_version: str | None
|
|
40
|
+
confidence: str
|
|
41
|
+
exact_match: bool
|
|
42
|
+
score: float
|
|
43
|
+
missing_columns: tuple[str, ...]
|
|
44
|
+
additional_focus_columns: tuple[str, ...]
|
|
45
|
+
extension_columns: tuple[str, ...]
|
|
46
|
+
unknown_columns: tuple[str, ...]
|
|
47
|
+
ambiguous_candidates: tuple[tuple[str, str], ...]
|
|
48
|
+
forced: bool = False
|
|
49
|
+
notes: tuple[str, ...] = ()
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def ok(self) -> bool:
|
|
53
|
+
"""A confident, unambiguous identification of a supported schema."""
|
|
54
|
+
return (
|
|
55
|
+
self.dataset is not None
|
|
56
|
+
and self.detected_version is not None
|
|
57
|
+
and self.confidence == CONF_HIGH
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
def as_dict(self) -> dict:
|
|
61
|
+
"""JSON-serialisable view for the manifest."""
|
|
62
|
+
return {
|
|
63
|
+
"dataset": self.dataset,
|
|
64
|
+
"detected_version": self.detected_version,
|
|
65
|
+
"confidence": self.confidence,
|
|
66
|
+
"exact_match": self.exact_match,
|
|
67
|
+
"score": round(self.score, 4),
|
|
68
|
+
"forced": self.forced,
|
|
69
|
+
"missing_columns": list(self.missing_columns),
|
|
70
|
+
"additional_focus_columns": list(self.additional_focus_columns),
|
|
71
|
+
"extension_columns": list(self.extension_columns),
|
|
72
|
+
"unknown_columns": list(self.unknown_columns),
|
|
73
|
+
"ambiguous_candidates": [list(c) for c in self.ambiguous_candidates],
|
|
74
|
+
"notes": list(self.notes),
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
@dataclass(frozen=True)
|
|
79
|
+
class _Candidate:
|
|
80
|
+
dataset: str
|
|
81
|
+
version: str
|
|
82
|
+
jaccard: float
|
|
83
|
+
mandatory_coverage: float
|
|
84
|
+
missing: tuple[str, ...]
|
|
85
|
+
additional_focus: tuple[str, ...]
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _score(dataset: str, version: str, headers: frozenset[str]) -> _Candidate:
|
|
89
|
+
expected = registry.version_columns(dataset, version)
|
|
90
|
+
dataset_focus = registry.all_dataset_columns(dataset)
|
|
91
|
+
present = expected & headers
|
|
92
|
+
missing = expected - headers
|
|
93
|
+
additional_focus = (headers & dataset_focus) - expected
|
|
94
|
+
union = expected | (headers & dataset_focus)
|
|
95
|
+
jaccard = len(present) / len(union) if union else 0.0
|
|
96
|
+
mandatory = registry.mandatory_columns(dataset, version)
|
|
97
|
+
mandatory_coverage = len(mandatory & headers) / len(mandatory) if mandatory else 1.0
|
|
98
|
+
return _Candidate(
|
|
99
|
+
dataset=dataset,
|
|
100
|
+
version=version,
|
|
101
|
+
jaccard=jaccard,
|
|
102
|
+
mandatory_coverage=mandatory_coverage,
|
|
103
|
+
missing=tuple(sorted(missing)),
|
|
104
|
+
additional_focus=tuple(sorted(additional_focus)),
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def detect_focus_schema(
|
|
109
|
+
headers: Iterable[str],
|
|
110
|
+
*,
|
|
111
|
+
dataset: str | None = None,
|
|
112
|
+
version: str | None = None,
|
|
113
|
+
) -> SchemaDetectionResult:
|
|
114
|
+
"""Detect the FOCUS dataset and version of ``headers``.
|
|
115
|
+
|
|
116
|
+
``dataset`` and/or ``version`` force the corresponding dimension (they still get scored,
|
|
117
|
+
so a bad forced choice yields a low confidence the caller can reject). Unknown forced
|
|
118
|
+
values raise ``ValueError``.
|
|
119
|
+
"""
|
|
120
|
+
all_headers = list(headers)
|
|
121
|
+
# A malformed CSV row with surplus fields makes ``csv.DictReader`` emit a ``None`` key;
|
|
122
|
+
# drop non-string header names (and record it) rather than crashing on ``.startswith``.
|
|
123
|
+
header_list = [h for h in all_headers if isinstance(h, str)]
|
|
124
|
+
malformed = len(all_headers) - len(header_list)
|
|
125
|
+
header_set = frozenset(header_list)
|
|
126
|
+
extension = tuple(sorted(h for h in header_set if h.startswith("x_")))
|
|
127
|
+
focus_all = registry.all_focus_columns()
|
|
128
|
+
unknown = tuple(sorted(h for h in header_set if not h.startswith("x_") and h not in focus_all))
|
|
129
|
+
|
|
130
|
+
forced_dataset = registry.resolve_dataset_name(dataset) if dataset is not None else None
|
|
131
|
+
forced_version = registry.normalize_version(version) if version is not None else None
|
|
132
|
+
forced = forced_dataset is not None or forced_version is not None
|
|
133
|
+
|
|
134
|
+
candidates = registry.candidate_schemas()
|
|
135
|
+
if forced_dataset is not None:
|
|
136
|
+
candidates = [c for c in candidates if c[0] == forced_dataset]
|
|
137
|
+
if forced_version is not None:
|
|
138
|
+
candidates = [c for c in candidates if c[1] == forced_version]
|
|
139
|
+
|
|
140
|
+
if not candidates:
|
|
141
|
+
note = "forced (dataset, version) does not exist in FOCUS"
|
|
142
|
+
return SchemaDetectionResult(
|
|
143
|
+
dataset=forced_dataset,
|
|
144
|
+
detected_version=forced_version,
|
|
145
|
+
confidence=CONF_LOW,
|
|
146
|
+
exact_match=False,
|
|
147
|
+
score=0.0,
|
|
148
|
+
missing_columns=(),
|
|
149
|
+
additional_focus_columns=(),
|
|
150
|
+
extension_columns=extension,
|
|
151
|
+
unknown_columns=unknown,
|
|
152
|
+
ambiguous_candidates=(),
|
|
153
|
+
forced=forced,
|
|
154
|
+
notes=(note,),
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
scored = sorted(
|
|
158
|
+
(_score(d, v, header_set) for d, v in candidates),
|
|
159
|
+
key=lambda c: (c.jaccard, c.mandatory_coverage),
|
|
160
|
+
reverse=True,
|
|
161
|
+
)
|
|
162
|
+
best = scored[0]
|
|
163
|
+
|
|
164
|
+
# Ambiguity: any other candidate whose similarity is within delta of the best.
|
|
165
|
+
ambiguous = tuple(
|
|
166
|
+
(c.dataset, c.version)
|
|
167
|
+
for c in scored[1:]
|
|
168
|
+
if c.jaccard > 0 and best.jaccard - c.jaccard < _AMBIGUITY_DELTA
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
notes: list[str] = []
|
|
172
|
+
dataset_name: str | None = best.dataset
|
|
173
|
+
detected_version: str | None = best.version
|
|
174
|
+
|
|
175
|
+
if not forced and best.jaccard < _DATASET_FLOOR:
|
|
176
|
+
# Header does not resemble any FOCUS schema.
|
|
177
|
+
return SchemaDetectionResult(
|
|
178
|
+
dataset=None,
|
|
179
|
+
detected_version=None,
|
|
180
|
+
confidence=CONF_LOW,
|
|
181
|
+
exact_match=False,
|
|
182
|
+
score=round(best.jaccard, 4),
|
|
183
|
+
missing_columns=(),
|
|
184
|
+
additional_focus_columns=(),
|
|
185
|
+
extension_columns=extension,
|
|
186
|
+
unknown_columns=unknown,
|
|
187
|
+
ambiguous_candidates=(),
|
|
188
|
+
forced=forced,
|
|
189
|
+
notes=("header does not match any known FOCUS dataset/version",),
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
# A FOCUS column that belongs to a *different* dataset (e.g. PaymentTerms on a Cost and
|
|
193
|
+
# Usage header) is neither "additional focus of this dataset" nor "unknown"; count it as a
|
|
194
|
+
# mismatch so it breaks exact_match and caps confidence (strict must not silently drop it).
|
|
195
|
+
foreign_focus = tuple(
|
|
196
|
+
sorted((header_set & focus_all) - registry.all_dataset_columns(best.dataset))
|
|
197
|
+
)
|
|
198
|
+
additional_focus = tuple(sorted(set(best.additional_focus) | set(foreign_focus)))
|
|
199
|
+
|
|
200
|
+
exact_match = not best.missing and not additional_focus and not unknown and not malformed
|
|
201
|
+
|
|
202
|
+
# Confidence.
|
|
203
|
+
if forced_version is not None:
|
|
204
|
+
# The version is locked by the user. Only columns that belong to *another* version of
|
|
205
|
+
# this dataset (``additional_focus``) make the forced version genuinely impossible;
|
|
206
|
+
# merely missing columns are a completeness issue the lint reports, not a reason to
|
|
207
|
+
# override an explicit choice. But a header with essentially no overlap (e.g. a
|
|
208
|
+
# non-FOCUS file) must still be rejected even when a version is forced.
|
|
209
|
+
if best.jaccard < _FORCED_OVERLAP_FLOOR:
|
|
210
|
+
confidence = CONF_LOW
|
|
211
|
+
notes.append(
|
|
212
|
+
f"header has no meaningful overlap with the forced schema "
|
|
213
|
+
f"{best.dataset} {forced_version}"
|
|
214
|
+
)
|
|
215
|
+
elif additional_focus:
|
|
216
|
+
confidence = CONF_LOW
|
|
217
|
+
notes.append(
|
|
218
|
+
f"header is incompatible with forced version {forced_version} "
|
|
219
|
+
f"(columns not in this schema: {', '.join(additional_focus)})"
|
|
220
|
+
)
|
|
221
|
+
elif best.missing or best.mandatory_coverage < 0.999:
|
|
222
|
+
confidence = CONF_MEDIUM # compatible but incomplete (lint will flag specifics)
|
|
223
|
+
else:
|
|
224
|
+
confidence = CONF_HIGH
|
|
225
|
+
elif (
|
|
226
|
+
best.jaccard >= 0.9
|
|
227
|
+
and best.mandatory_coverage >= 0.999
|
|
228
|
+
and not ambiguous
|
|
229
|
+
and not unknown
|
|
230
|
+
and not additional_focus
|
|
231
|
+
and not malformed
|
|
232
|
+
):
|
|
233
|
+
confidence = CONF_HIGH
|
|
234
|
+
elif best.jaccard >= 0.6 and best.mandatory_coverage >= 0.8:
|
|
235
|
+
confidence = CONF_MEDIUM
|
|
236
|
+
else:
|
|
237
|
+
confidence = CONF_LOW
|
|
238
|
+
|
|
239
|
+
if malformed:
|
|
240
|
+
notes.append(f"{malformed} malformed (non-string) header name(s) ignored")
|
|
241
|
+
if foreign_focus:
|
|
242
|
+
notes.append(
|
|
243
|
+
"FOCUS columns from another dataset present: " + ", ".join(foreign_focus)
|
|
244
|
+
)
|
|
245
|
+
if ambiguous and confidence == CONF_HIGH:
|
|
246
|
+
confidence = CONF_MEDIUM
|
|
247
|
+
if unknown and confidence == CONF_HIGH:
|
|
248
|
+
confidence = CONF_MEDIUM
|
|
249
|
+
notes.append(f"{len(unknown)} unknown non-x_ column(s) present")
|
|
250
|
+
elif unknown:
|
|
251
|
+
notes.append(f"{len(unknown)} unknown non-x_ column(s) present")
|
|
252
|
+
if best.additional_focus:
|
|
253
|
+
notes.append(
|
|
254
|
+
"columns from another FOCUS version present: " + ", ".join(best.additional_focus)
|
|
255
|
+
)
|
|
256
|
+
if ambiguous:
|
|
257
|
+
notes.append(
|
|
258
|
+
"close alternative schema(s): " + ", ".join(f"{d} {v}" for d, v in ambiguous)
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
return SchemaDetectionResult(
|
|
262
|
+
dataset=dataset_name,
|
|
263
|
+
detected_version=detected_version,
|
|
264
|
+
confidence=confidence,
|
|
265
|
+
exact_match=exact_match,
|
|
266
|
+
score=round(best.jaccard, 4),
|
|
267
|
+
missing_columns=best.missing,
|
|
268
|
+
additional_focus_columns=additional_focus,
|
|
269
|
+
extension_columns=extension,
|
|
270
|
+
unknown_columns=unknown,
|
|
271
|
+
ambiguous_candidates=ambiguous,
|
|
272
|
+
forced=forced,
|
|
273
|
+
notes=tuple(notes),
|
|
274
|
+
)
|