focus-data-toolkit 0.11.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- focus_data_toolkit/__init__.py +69 -0
- focus_data_toolkit/__main__.py +6 -0
- focus_data_toolkit/_version.py +8 -0
- focus_data_toolkit/cli.py +968 -0
- focus_data_toolkit/context/__init__.py +88 -0
- focus_data_toolkit/context/billing.py +54 -0
- focus_data_toolkit/context/provider.py +90 -0
- focus_data_toolkit/convert/__init__.py +708 -0
- focus_data_toolkit/convert/billing_period.py +65 -0
- focus_data_toolkit/convert/contract_applied.py +235 -0
- focus_data_toolkit/convert/contract_commitment.py +182 -0
- focus_data_toolkit/convert/cost_and_usage.py +179 -0
- focus_data_toolkit/convert/detect.py +39 -0
- focus_data_toolkit/convert/invoice_detail.py +199 -0
- focus_data_toolkit/convert/streaming.py +1030 -0
- focus_data_toolkit/errors.py +145 -0
- focus_data_toolkit/focus_json.py +68 -0
- focus_data_toolkit/generators/__init__.py +61 -0
- focus_data_toolkit/generators/_shim.py +43 -0
- focus_data_toolkit/generators/engine/__init__.py +14 -0
- focus_data_toolkit/generators/engine/context.py +12 -0
- focus_data_toolkit/generators/engine/determinism.py +117 -0
- focus_data_toolkit/generators/engine/json_focus.py +63 -0
- focus_data_toolkit/generators/engine/ladder.py +71 -0
- focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
- focus_data_toolkit/generators/engine/serialize.py +151 -0
- focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
- focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
- focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
- focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
- focus_data_toolkit/generators/providers/__init__.py +29 -0
- focus_data_toolkit/generators/providers/aws.py +186 -0
- focus_data_toolkit/generators/providers/azure.py +191 -0
- focus_data_toolkit/generators/providers/gcp.py +194 -0
- focus_data_toolkit/generators/providers/profile.py +123 -0
- focus_data_toolkit/generators/scenarios.py +178 -0
- focus_data_toolkit/generators/versions/__init__.py +17 -0
- focus_data_toolkit/generators/versions/adapter.py +41 -0
- focus_data_toolkit/generators/versions/v1_2.py +111 -0
- focus_data_toolkit/generators/versions/v1_3.py +154 -0
- focus_data_toolkit/io/__init__.py +1 -0
- focus_data_toolkit/io/atomic_writer.py +462 -0
- focus_data_toolkit/io/csv_io.py +128 -0
- focus_data_toolkit/io/parquet_io.py +528 -0
- focus_data_toolkit/io/records.py +92 -0
- focus_data_toolkit/io/row_source.py +117 -0
- focus_data_toolkit/lifecycle.py +342 -0
- focus_data_toolkit/manifest.py +114 -0
- focus_data_toolkit/model/__init__.py +43 -0
- focus_data_toolkit/model/capabilities.py +66 -0
- focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
- focus_data_toolkit/model/focus_1_4_model.json +1913 -0
- focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
- focus_data_toolkit/model/focus_json_keys.py +112 -0
- focus_data_toolkit/model/iso_4217_currencies.json +23 -0
- focus_data_toolkit/model/json_schema_check.py +205 -0
- focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
- focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
- focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
- focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
- focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
- focus_data_toolkit/model/model_provenance.json +58 -0
- focus_data_toolkit/model/validator.py +498 -0
- focus_data_toolkit/modes.py +18 -0
- focus_data_toolkit/official_validator.py +61 -0
- focus_data_toolkit/progress.py +89 -0
- focus_data_toolkit/provenance.py +106 -0
- focus_data_toolkit/py.typed +1 -0
- focus_data_toolkit/runtime.py +243 -0
- focus_data_toolkit/schema/__init__.py +17 -0
- focus_data_toolkit/schema/detection.py +274 -0
- focus_data_toolkit/schema/registry.py +127 -0
- focus_data_toolkit/storage/__init__.py +1 -0
- focus_data_toolkit/storage/external_index.py +99 -0
- focus_data_toolkit/storage/spill.py +150 -0
- focus_data_toolkit/studio/__init__.py +19 -0
- focus_data_toolkit/studio/app.py +467 -0
- focus_data_toolkit/studio/config.py +42 -0
- focus_data_toolkit/studio/frontend/app.js +214 -0
- focus_data_toolkit/studio/frontend/index.html +101 -0
- focus_data_toolkit/studio/frontend/style.css +60 -0
- focus_data_toolkit/studio/jobs.py +142 -0
- focus_data_toolkit/studio/preview.py +32 -0
- focus_data_toolkit/studio/security.py +125 -0
- focus_data_toolkit/studio/server.py +71 -0
- focus_data_toolkit/supplement/__init__.py +50 -0
- focus_data_toolkit/supplement/adapters/__init__.py +21 -0
- focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
- focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
- focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
- focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
- focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
- focus_data_toolkit/supplement/adapters/registry.py +215 -0
- focus_data_toolkit/supplement/apply.py +318 -0
- focus_data_toolkit/supplement/gaps.py +219 -0
- focus_data_toolkit/supplement/kinds.py +118 -0
- focus_data_toolkit/supplement/loader.py +409 -0
- focus_data_toolkit/supplement/spec.py +74 -0
- focus_data_toolkit/supplement/validate.py +215 -0
- focus_data_toolkit/validate/__init__.py +15 -0
- focus_data_toolkit/validate/allocation.py +333 -0
- focus_data_toolkit/validate/bundle.py +254 -0
- focus_data_toolkit/validate/codes.py +93 -0
- focus_data_toolkit/validate/corrections.py +245 -0
- focus_data_toolkit/validate/reconciliation.py +98 -0
- focus_data_toolkit/validate/referential.py +289 -0
- focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
- focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
- focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
- focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
- focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
- focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""Normative FOCUS column sets per (dataset, version).
|
|
2
|
+
|
|
3
|
+
Everything here is *computed* from the committed FOCUS 1.4 model — each column records
|
|
4
|
+
the ``version`` it was introduced in — plus a small hand-authored table of columns that
|
|
5
|
+
existed in an earlier version but were **removed** by 1.4 (and so are absent from the 1.4
|
|
6
|
+
model). This keeps the registry in lock-step with the model artifact while still being
|
|
7
|
+
able to describe 1.2 and 1.3 headers faithfully.
|
|
8
|
+
|
|
9
|
+
Source of truth for the removed-column table: the FOCUS changelog and this repository's own
|
|
10
|
+
1.2/1.3 generators. ``ProviderName`` / ``PublisherName`` were superseded by
|
|
11
|
+
``ServiceProviderName`` / ``HostProviderName`` in 1.3 and removed in 1.4.
|
|
12
|
+
|
|
13
|
+
Sanity check (see ``version_columns``): Cost and Usage yields 57 columns at 1.2, 65 at 1.3
|
|
14
|
+
and 65 at 1.4 — matching the generators (57/65) and the 1.4 model (65).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from functools import cache
|
|
20
|
+
|
|
21
|
+
from focus_data_toolkit.model import FOCUS_1_4_DATASETS, load_model, resolve_dataset
|
|
22
|
+
|
|
23
|
+
# FOCUS versions this toolkit reasons about for detection/conversion.
|
|
24
|
+
SUPPORTED_VERSIONS: tuple[str, ...] = ("1.2", "1.3", "1.4")
|
|
25
|
+
|
|
26
|
+
# dataset -> {column: (introduced_in, removed_in, mandatory_before)} for columns removed by
|
|
27
|
+
# 1.4. ``mandatory_before`` is the version at which the column stopped being required because
|
|
28
|
+
# a replacement arrived; a 1.2 Cost and Usage source needs ProviderName / PublisherName to
|
|
29
|
+
# derive the 1.4-Mandatory ServiceProviderName / HostProviderName (which do not exist until
|
|
30
|
+
# 1.3), so they are mandatory for versions < 1.3.
|
|
31
|
+
REMOVED_COLUMNS: dict[str, dict[str, tuple[str, str, str]]] = {
|
|
32
|
+
"Cost and Usage": {
|
|
33
|
+
"ProviderName": ("0.5", "1.4", "1.3"),
|
|
34
|
+
"PublisherName": ("0.5", "1.4", "1.3"),
|
|
35
|
+
},
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def version_tuple(version: str) -> tuple[int, int]:
|
|
40
|
+
"""Parse a ``"major.minor"`` (or longer) FOCUS version to a comparable tuple."""
|
|
41
|
+
parts = version.strip().split(".")
|
|
42
|
+
try:
|
|
43
|
+
return (int(parts[0]), int(parts[1]))
|
|
44
|
+
except (IndexError, ValueError) as exc: # pragma: no cover - defensive
|
|
45
|
+
raise ValueError(f"unparseable FOCUS version {version!r}") from exc
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def normalize_version(version: str) -> str:
|
|
49
|
+
"""Normalise a version string to ``"major.minor"`` (e.g. ``"1.3.0"`` -> ``"1.3"``)."""
|
|
50
|
+
major, minor = version_tuple(version)
|
|
51
|
+
return f"{major}.{minor}"
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@cache
|
|
55
|
+
def version_columns(dataset: str, version: str) -> frozenset[str]:
|
|
56
|
+
"""FOCUS columns of ``dataset`` present at ``version``.
|
|
57
|
+
|
|
58
|
+
A model column is present when it was introduced at or before ``version``; a removed
|
|
59
|
+
column is present when it was introduced at or before ``version`` and removed strictly
|
|
60
|
+
after it.
|
|
61
|
+
"""
|
|
62
|
+
cols = load_model()["datasets"][dataset]["columns"]
|
|
63
|
+
tv = version_tuple(version)
|
|
64
|
+
present = {c for c, spec in cols.items() if version_tuple(spec["version"]) <= tv}
|
|
65
|
+
for col, (intro, removed, _mandatory_before) in REMOVED_COLUMNS.get(dataset, {}).items():
|
|
66
|
+
if version_tuple(intro) <= tv < version_tuple(removed):
|
|
67
|
+
present.add(col)
|
|
68
|
+
return frozenset(present)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@cache
|
|
72
|
+
def mandatory_columns(dataset: str, version: str) -> frozenset[str]:
|
|
73
|
+
"""Mandatory FOCUS columns of ``dataset`` present at ``version``.
|
|
74
|
+
|
|
75
|
+
Feature level is taken from the 1.4 model; removed columns that were required at
|
|
76
|
+
``version`` (before a replacement arrived) are added so detection does not accept a source
|
|
77
|
+
that cannot fill the 1.4-Mandatory columns it derives.
|
|
78
|
+
"""
|
|
79
|
+
cols = load_model()["datasets"][dataset]["columns"]
|
|
80
|
+
tv = version_tuple(version)
|
|
81
|
+
mandatory = {
|
|
82
|
+
c
|
|
83
|
+
for c, spec in cols.items()
|
|
84
|
+
if version_tuple(spec["version"]) <= tv and spec.get("feature_level") == "Mandatory"
|
|
85
|
+
}
|
|
86
|
+
for col, (intro, _removed, mandatory_before) in REMOVED_COLUMNS.get(dataset, {}).items():
|
|
87
|
+
if version_tuple(intro) <= tv < version_tuple(mandatory_before):
|
|
88
|
+
mandatory.add(col)
|
|
89
|
+
return frozenset(mandatory)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@cache
|
|
93
|
+
def dataset_exists_at(dataset: str, version: str) -> bool:
|
|
94
|
+
"""Whether ``dataset`` is defined at all at ``version`` (has any column)."""
|
|
95
|
+
return bool(version_columns(dataset, version))
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@cache
|
|
99
|
+
def all_dataset_columns(dataset: str) -> frozenset[str]:
|
|
100
|
+
"""Every FOCUS column of ``dataset`` across all versions, including removed ones."""
|
|
101
|
+
cols = set(load_model()["datasets"][dataset]["columns"])
|
|
102
|
+
cols |= set(REMOVED_COLUMNS.get(dataset, {}))
|
|
103
|
+
return frozenset(cols)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@cache
|
|
107
|
+
def all_focus_columns() -> frozenset[str]:
|
|
108
|
+
"""Every FOCUS column name across every dataset and version."""
|
|
109
|
+
out: set[str] = set()
|
|
110
|
+
for dataset in FOCUS_1_4_DATASETS:
|
|
111
|
+
out |= all_dataset_columns(dataset)
|
|
112
|
+
return frozenset(out)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def candidate_schemas() -> list[tuple[str, str]]:
|
|
116
|
+
"""All ``(dataset, version)`` pairs that actually exist, in canonical order."""
|
|
117
|
+
return [
|
|
118
|
+
(dataset, version)
|
|
119
|
+
for dataset in FOCUS_1_4_DATASETS
|
|
120
|
+
for version in SUPPORTED_VERSIONS
|
|
121
|
+
if dataset_exists_at(dataset, version)
|
|
122
|
+
]
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def resolve_dataset_name(name: str) -> str:
|
|
126
|
+
"""Resolve a dataset alias (``"cau"``, ``"cost-and-usage"``, ...) to its canonical name."""
|
|
127
|
+
return resolve_dataset(name.replace("-", " "))
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Bounded-memory external state for streaming conversion (sqlite3, stdlib)."""
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""Disk-backed aggregation/dedup for streaming conversion, using stdlib ``sqlite3``.
|
|
2
|
+
|
|
3
|
+
Streaming the huge Cost and Usage file still needs a little global state — the Invoice Detail
|
|
4
|
+
sum per business grain and the distinct Billing Periods. Holding that in Python dicts would
|
|
5
|
+
scale with the number of *groups*; here it lives in a throwaway SQLite database inside the
|
|
6
|
+
atomic staging directory, so memory stays bounded.
|
|
7
|
+
|
|
8
|
+
Exactness and determinism:
|
|
9
|
+
|
|
10
|
+
* Costs are stored as **TEXT** and summed with Python ``Decimal`` during the ordered scan —
|
|
11
|
+
never ``SUM()`` in SQL — so the streamed sum is bit-for-bit the eager sum.
|
|
12
|
+
* Every finalize scan is ``ORDER BY <keys>`` under the default **BINARY** collation, which
|
|
13
|
+
compares UTF-8 bytes and so matches Python ``sorted()`` on the same string tuples.
|
|
14
|
+
|
|
15
|
+
The scratch database is disposable (``journal_mode=OFF``, ``synchronous=OFF``); durability
|
|
16
|
+
comes from the atomic writer's fsync of the finished data files, not from this DB.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import sqlite3
|
|
22
|
+
from collections.abc import Iterator
|
|
23
|
+
from decimal import Decimal
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
|
|
26
|
+
from focus_data_toolkit.convert.invoice_detail import GrainKey
|
|
27
|
+
|
|
28
|
+
# Invoice Detail business-grain columns, in key order (matches GRAIN_FIELDS).
|
|
29
|
+
_GRAIN_COLS = (
|
|
30
|
+
"issuer",
|
|
31
|
+
"invoice_id",
|
|
32
|
+
"account",
|
|
33
|
+
"currency",
|
|
34
|
+
"bp_start",
|
|
35
|
+
"bp_end",
|
|
36
|
+
"charge_category",
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class ExternalIndex:
|
|
41
|
+
"""SQLite-backed staging for Invoice Detail aggregation and Billing Period dedup."""
|
|
42
|
+
|
|
43
|
+
def __init__(self, db_path: str | Path) -> None:
|
|
44
|
+
self._conn = sqlite3.connect(str(db_path))
|
|
45
|
+
for pragma in ("journal_mode=OFF", "synchronous=OFF", "temp_store=FILE", "cache_size=-20000"):
|
|
46
|
+
self._conn.execute(f"PRAGMA {pragma}")
|
|
47
|
+
self._conn.execute(
|
|
48
|
+
"CREATE TABLE id_stage (n INTEGER PRIMARY KEY, "
|
|
49
|
+
+ ", ".join(f"{col} TEXT" for col in _GRAIN_COLS)
|
|
50
|
+
+ ", billed_cost TEXT)"
|
|
51
|
+
)
|
|
52
|
+
self._conn.execute(
|
|
53
|
+
"CREATE TABLE bp (start TEXT, end TEXT, issuer TEXT, "
|
|
54
|
+
"PRIMARY KEY (start, end, issuer)) WITHOUT ROWID"
|
|
55
|
+
)
|
|
56
|
+
self._insert_line = (
|
|
57
|
+
"INSERT INTO id_stage (" + ", ".join(_GRAIN_COLS) + ", billed_cost) VALUES ("
|
|
58
|
+
+ ", ".join("?" * (len(_GRAIN_COLS) + 1)) + ")"
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
def stage_invoice_line(self, grain_key: GrainKey, billed_cost: str) -> None:
|
|
62
|
+
"""Record one Cost and Usage line's contribution to its invoice-detail group."""
|
|
63
|
+
self._conn.execute(self._insert_line, (*grain_key, billed_cost))
|
|
64
|
+
|
|
65
|
+
def stage_billing_period(self, start: str, end: str, issuer: str) -> None:
|
|
66
|
+
"""Record a (start, end, issuer) billing period (first occurrence wins)."""
|
|
67
|
+
self._conn.execute(
|
|
68
|
+
"INSERT OR IGNORE INTO bp (start, end, issuer) VALUES (?, ?, ?)", (start, end, issuer)
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
def finalize_invoice_groups(self) -> Iterator[tuple[GrainKey, Decimal]]:
|
|
72
|
+
"""Yield ``(grain_key, summed_billed_cost)`` per group, in sorted grain order."""
|
|
73
|
+
self._conn.commit()
|
|
74
|
+
order = ", ".join(_GRAIN_COLS) + ", n"
|
|
75
|
+
cursor = self._conn.execute(
|
|
76
|
+
f"SELECT {', '.join(_GRAIN_COLS)}, billed_cost FROM id_stage ORDER BY {order}"
|
|
77
|
+
)
|
|
78
|
+
current: GrainKey | None = None
|
|
79
|
+
total = Decimal(0)
|
|
80
|
+
for row in cursor:
|
|
81
|
+
key: GrainKey = tuple(row[: len(_GRAIN_COLS)])
|
|
82
|
+
cost = row[len(_GRAIN_COLS)]
|
|
83
|
+
if current is None:
|
|
84
|
+
current = key
|
|
85
|
+
if key != current:
|
|
86
|
+
yield current, total
|
|
87
|
+
current = key
|
|
88
|
+
total = Decimal(0)
|
|
89
|
+
total += Decimal(cost or "0")
|
|
90
|
+
if current is not None:
|
|
91
|
+
yield current, total
|
|
92
|
+
|
|
93
|
+
def finalize_billing_periods(self) -> Iterator[tuple[str, str, str]]:
|
|
94
|
+
"""Yield distinct ``(start, end, issuer)`` billing periods, in sorted order."""
|
|
95
|
+
self._conn.commit()
|
|
96
|
+
yield from self._conn.execute("SELECT start, end, issuer FROM bp ORDER BY start, end, issuer")
|
|
97
|
+
|
|
98
|
+
def close(self) -> None:
|
|
99
|
+
self._conn.close()
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""Threshold-spilling string maps for streaming cross-dataset validation.
|
|
2
|
+
|
|
3
|
+
The bundle validator's checks keep per-key state (seen ids, foreign-key targets, running
|
|
4
|
+
sums) whose cardinality scales with the *large* Cost and Usage dataset. A
|
|
5
|
+
:class:`SpillableIndexPool` hands out ``str -> str`` mutable mappings that live in an
|
|
6
|
+
ordinary in-memory ``dict`` until a size threshold, then migrate transparently into a shared
|
|
7
|
+
throwaway SQLite database — so validating a bundle far larger than RAM stays bounded.
|
|
8
|
+
|
|
9
|
+
Like :mod:`focus_data_toolkit.storage.external_index`, the database is disposable scratch
|
|
10
|
+
state (``journal_mode=OFF``, ``synchronous=OFF``): durability comes from the atomic writer's
|
|
11
|
+
fsync of the published files, never from this DB. The file is created lazily on the first
|
|
12
|
+
spill, so small bundles touch the disk not at all.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import sqlite3
|
|
18
|
+
from collections.abc import Iterator, MutableMapping
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
# Keys held in memory per map before spilling to SQLite. At ~100 bytes per key/value pair
|
|
22
|
+
# this bounds each map's resident size to roughly 20 MB worst-case before it moves to disk.
|
|
23
|
+
DEFAULT_SPILL_THRESHOLD = 200_000
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class SpillableIndexPool:
|
|
27
|
+
"""Factory of :class:`SpillableMap` instances sharing one lazy scratch SQLite database."""
|
|
28
|
+
|
|
29
|
+
def __init__(self, db_path: str | Path, *, threshold: int = DEFAULT_SPILL_THRESHOLD) -> None:
|
|
30
|
+
if threshold < 1:
|
|
31
|
+
raise ValueError(f"spill threshold must be >= 1, got {threshold}")
|
|
32
|
+
self._db_path = Path(db_path)
|
|
33
|
+
self._threshold = threshold
|
|
34
|
+
self._conn: sqlite3.Connection | None = None
|
|
35
|
+
self._tables = 0
|
|
36
|
+
self._spilled = False
|
|
37
|
+
|
|
38
|
+
@property
|
|
39
|
+
def threshold(self) -> int:
|
|
40
|
+
return self._threshold
|
|
41
|
+
|
|
42
|
+
@property
|
|
43
|
+
def spilled(self) -> bool:
|
|
44
|
+
"""Whether any map has ever spilled (i.e. the scratch database was created)."""
|
|
45
|
+
return self._spilled
|
|
46
|
+
|
|
47
|
+
def make_map(self) -> SpillableMap:
|
|
48
|
+
"""Return a fresh empty ``str -> str`` mapping backed by this pool."""
|
|
49
|
+
self._tables += 1
|
|
50
|
+
return SpillableMap(self, f"kv{self._tables}")
|
|
51
|
+
|
|
52
|
+
def _connection(self) -> sqlite3.Connection:
|
|
53
|
+
if self._conn is None:
|
|
54
|
+
self._conn = sqlite3.connect(str(self._db_path))
|
|
55
|
+
self._spilled = True
|
|
56
|
+
for pragma in (
|
|
57
|
+
"journal_mode=OFF",
|
|
58
|
+
"synchronous=OFF",
|
|
59
|
+
"temp_store=FILE",
|
|
60
|
+
"cache_size=-20000",
|
|
61
|
+
):
|
|
62
|
+
self._conn.execute(f"PRAGMA {pragma}")
|
|
63
|
+
return self._conn
|
|
64
|
+
|
|
65
|
+
def close(self) -> None:
|
|
66
|
+
if self._conn is not None:
|
|
67
|
+
self._conn.close()
|
|
68
|
+
self._conn = None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class SpillableMap(MutableMapping[str, str]):
|
|
72
|
+
"""A ``str -> str`` mapping that spills from a dict to the pool's SQLite past a threshold."""
|
|
73
|
+
|
|
74
|
+
def __init__(self, pool: SpillableIndexPool, table: str) -> None:
|
|
75
|
+
self._pool = pool
|
|
76
|
+
self._table = table
|
|
77
|
+
self._mem: dict[str, str] | None = {}
|
|
78
|
+
|
|
79
|
+
def _spill(self) -> None:
|
|
80
|
+
assert self._mem is not None
|
|
81
|
+
conn = self._pool._connection()
|
|
82
|
+
conn.execute(f"CREATE TABLE {self._table} (k TEXT PRIMARY KEY, v TEXT) WITHOUT ROWID")
|
|
83
|
+
conn.executemany(
|
|
84
|
+
f"INSERT INTO {self._table} (k, v) VALUES (?, ?)", self._mem.items()
|
|
85
|
+
)
|
|
86
|
+
self._mem = None
|
|
87
|
+
|
|
88
|
+
def __setitem__(self, key: str, value: str) -> None:
|
|
89
|
+
if self._mem is not None:
|
|
90
|
+
self._mem[key] = value
|
|
91
|
+
if len(self._mem) > self._pool.threshold:
|
|
92
|
+
self._spill()
|
|
93
|
+
return
|
|
94
|
+
self._pool._connection().execute(
|
|
95
|
+
f"INSERT OR REPLACE INTO {self._table} (k, v) VALUES (?, ?)", (key, value)
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
def __getitem__(self, key: str) -> str:
|
|
99
|
+
if self._mem is not None:
|
|
100
|
+
return self._mem[key]
|
|
101
|
+
row = (
|
|
102
|
+
self._pool._connection()
|
|
103
|
+
.execute(f"SELECT v FROM {self._table} WHERE k = ?", (key,))
|
|
104
|
+
.fetchone()
|
|
105
|
+
)
|
|
106
|
+
if row is None:
|
|
107
|
+
raise KeyError(key)
|
|
108
|
+
return row[0]
|
|
109
|
+
|
|
110
|
+
def __delitem__(self, key: str) -> None:
|
|
111
|
+
if self._mem is not None:
|
|
112
|
+
del self._mem[key]
|
|
113
|
+
return
|
|
114
|
+
cursor = self._pool._connection().execute(
|
|
115
|
+
f"DELETE FROM {self._table} WHERE k = ?", (key,)
|
|
116
|
+
)
|
|
117
|
+
if cursor.rowcount == 0:
|
|
118
|
+
raise KeyError(key)
|
|
119
|
+
|
|
120
|
+
def __contains__(self, key: object) -> bool:
|
|
121
|
+
if self._mem is not None:
|
|
122
|
+
return key in self._mem
|
|
123
|
+
if not isinstance(key, str):
|
|
124
|
+
return False
|
|
125
|
+
row = (
|
|
126
|
+
self._pool._connection()
|
|
127
|
+
.execute(f"SELECT 1 FROM {self._table} WHERE k = ?", (key,))
|
|
128
|
+
.fetchone()
|
|
129
|
+
)
|
|
130
|
+
return row is not None
|
|
131
|
+
|
|
132
|
+
def __len__(self) -> int:
|
|
133
|
+
if self._mem is not None:
|
|
134
|
+
return len(self._mem)
|
|
135
|
+
return self._pool._connection().execute(
|
|
136
|
+
f"SELECT COUNT(*) FROM {self._table}"
|
|
137
|
+
).fetchone()[0]
|
|
138
|
+
|
|
139
|
+
def __iter__(self) -> Iterator[str]:
|
|
140
|
+
if self._mem is not None:
|
|
141
|
+
yield from self._mem
|
|
142
|
+
return
|
|
143
|
+
# Sorted (BINARY collation = UTF-8 byte order) for a deterministic iteration order.
|
|
144
|
+
for (key,) in self._pool._connection().execute(
|
|
145
|
+
f"SELECT k FROM {self._table} ORDER BY k"
|
|
146
|
+
):
|
|
147
|
+
yield key
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
__all__ = ["DEFAULT_SPILL_THRESHOLD", "SpillableIndexPool", "SpillableMap"]
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Focus Data Toolkit Studio — a LOCAL web UI over the same Core.
|
|
2
|
+
|
|
3
|
+
The Studio never reimplements FOCUS logic: it drives the exact SDK the CLI and Runner use
|
|
4
|
+
(``detect_focus_schema``, ``convert_files``, ``validate_dataset_bundle``, the generators), so its
|
|
5
|
+
manifests, diagnostics and checksums are identical to a CLI run. It is designed for **local,
|
|
6
|
+
single-user** use: it binds to loopback by default, requires a per-start token, validates
|
|
7
|
+
Host/Origin headers, confines file access to an allowlisted root, and processes on the bounded
|
|
8
|
+
streaming path — data never leaves the machine.
|
|
9
|
+
|
|
10
|
+
This subpackage lives behind the optional ``[studio]`` extra (FastAPI + uvicorn). The CLI's
|
|
11
|
+
``focus-toolkit ui`` command imports it lazily, so a core install without the extra still works.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from focus_data_toolkit.studio.config import StudioConfig
|
|
17
|
+
from focus_data_toolkit.studio.server import run
|
|
18
|
+
|
|
19
|
+
__all__ = ["StudioConfig", "run"]
|