focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,127 @@
1
+ """Normative FOCUS column sets per (dataset, version).
2
+
3
+ Everything here is *computed* from the committed FOCUS 1.4 model — each column records
4
+ the ``version`` it was introduced in — plus a small hand-authored table of columns that
5
+ existed in an earlier version but were **removed** by 1.4 (and so are absent from the 1.4
6
+ model). This keeps the registry in lock-step with the model artifact while still being
7
+ able to describe 1.2 and 1.3 headers faithfully.
8
+
9
+ Source of truth for the removed-column table: the FOCUS changelog and this repository's own
10
+ 1.2/1.3 generators. ``ProviderName`` / ``PublisherName`` were superseded by
11
+ ``ServiceProviderName`` / ``HostProviderName`` in 1.3 and removed in 1.4.
12
+
13
+ Sanity check (see ``version_columns``): Cost and Usage yields 57 columns at 1.2, 65 at 1.3
14
+ and 65 at 1.4 — matching the generators (57/65) and the 1.4 model (65).
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from functools import cache
20
+
21
+ from focus_data_toolkit.model import FOCUS_1_4_DATASETS, load_model, resolve_dataset
22
+
23
+ # FOCUS versions this toolkit reasons about for detection/conversion.
24
+ SUPPORTED_VERSIONS: tuple[str, ...] = ("1.2", "1.3", "1.4")
25
+
26
+ # dataset -> {column: (introduced_in, removed_in, mandatory_before)} for columns removed by
27
+ # 1.4. ``mandatory_before`` is the version at which the column stopped being required because
28
+ # a replacement arrived; a 1.2 Cost and Usage source needs ProviderName / PublisherName to
29
+ # derive the 1.4-Mandatory ServiceProviderName / HostProviderName (which do not exist until
30
+ # 1.3), so they are mandatory for versions < 1.3.
31
+ REMOVED_COLUMNS: dict[str, dict[str, tuple[str, str, str]]] = {
32
+ "Cost and Usage": {
33
+ "ProviderName": ("0.5", "1.4", "1.3"),
34
+ "PublisherName": ("0.5", "1.4", "1.3"),
35
+ },
36
+ }
37
+
38
+
39
+ def version_tuple(version: str) -> tuple[int, int]:
40
+ """Parse a ``"major.minor"`` (or longer) FOCUS version to a comparable tuple."""
41
+ parts = version.strip().split(".")
42
+ try:
43
+ return (int(parts[0]), int(parts[1]))
44
+ except (IndexError, ValueError) as exc: # pragma: no cover - defensive
45
+ raise ValueError(f"unparseable FOCUS version {version!r}") from exc
46
+
47
+
48
+ def normalize_version(version: str) -> str:
49
+ """Normalise a version string to ``"major.minor"`` (e.g. ``"1.3.0"`` -> ``"1.3"``)."""
50
+ major, minor = version_tuple(version)
51
+ return f"{major}.{minor}"
52
+
53
+
54
+ @cache
55
+ def version_columns(dataset: str, version: str) -> frozenset[str]:
56
+ """FOCUS columns of ``dataset`` present at ``version``.
57
+
58
+ A model column is present when it was introduced at or before ``version``; a removed
59
+ column is present when it was introduced at or before ``version`` and removed strictly
60
+ after it.
61
+ """
62
+ cols = load_model()["datasets"][dataset]["columns"]
63
+ tv = version_tuple(version)
64
+ present = {c for c, spec in cols.items() if version_tuple(spec["version"]) <= tv}
65
+ for col, (intro, removed, _mandatory_before) in REMOVED_COLUMNS.get(dataset, {}).items():
66
+ if version_tuple(intro) <= tv < version_tuple(removed):
67
+ present.add(col)
68
+ return frozenset(present)
69
+
70
+
71
+ @cache
72
+ def mandatory_columns(dataset: str, version: str) -> frozenset[str]:
73
+ """Mandatory FOCUS columns of ``dataset`` present at ``version``.
74
+
75
+ Feature level is taken from the 1.4 model; removed columns that were required at
76
+ ``version`` (before a replacement arrived) are added so detection does not accept a source
77
+ that cannot fill the 1.4-Mandatory columns it derives.
78
+ """
79
+ cols = load_model()["datasets"][dataset]["columns"]
80
+ tv = version_tuple(version)
81
+ mandatory = {
82
+ c
83
+ for c, spec in cols.items()
84
+ if version_tuple(spec["version"]) <= tv and spec.get("feature_level") == "Mandatory"
85
+ }
86
+ for col, (intro, _removed, mandatory_before) in REMOVED_COLUMNS.get(dataset, {}).items():
87
+ if version_tuple(intro) <= tv < version_tuple(mandatory_before):
88
+ mandatory.add(col)
89
+ return frozenset(mandatory)
90
+
91
+
92
+ @cache
93
+ def dataset_exists_at(dataset: str, version: str) -> bool:
94
+ """Whether ``dataset`` is defined at all at ``version`` (has any column)."""
95
+ return bool(version_columns(dataset, version))
96
+
97
+
98
+ @cache
99
+ def all_dataset_columns(dataset: str) -> frozenset[str]:
100
+ """Every FOCUS column of ``dataset`` across all versions, including removed ones."""
101
+ cols = set(load_model()["datasets"][dataset]["columns"])
102
+ cols |= set(REMOVED_COLUMNS.get(dataset, {}))
103
+ return frozenset(cols)
104
+
105
+
106
+ @cache
107
+ def all_focus_columns() -> frozenset[str]:
108
+ """Every FOCUS column name across every dataset and version."""
109
+ out: set[str] = set()
110
+ for dataset in FOCUS_1_4_DATASETS:
111
+ out |= all_dataset_columns(dataset)
112
+ return frozenset(out)
113
+
114
+
115
+ def candidate_schemas() -> list[tuple[str, str]]:
116
+ """All ``(dataset, version)`` pairs that actually exist, in canonical order."""
117
+ return [
118
+ (dataset, version)
119
+ for dataset in FOCUS_1_4_DATASETS
120
+ for version in SUPPORTED_VERSIONS
121
+ if dataset_exists_at(dataset, version)
122
+ ]
123
+
124
+
125
+ def resolve_dataset_name(name: str) -> str:
126
+ """Resolve a dataset alias (``"cau"``, ``"cost-and-usage"``, ...) to its canonical name."""
127
+ return resolve_dataset(name.replace("-", " "))
@@ -0,0 +1 @@
1
+ """Bounded-memory external state for streaming conversion (sqlite3, stdlib)."""
@@ -0,0 +1,99 @@
1
+ """Disk-backed aggregation/dedup for streaming conversion, using stdlib ``sqlite3``.
2
+
3
+ Streaming the huge Cost and Usage file still needs a little global state — the Invoice Detail
4
+ sum per business grain and the distinct Billing Periods. Holding that in Python dicts would
5
+ scale with the number of *groups*; here it lives in a throwaway SQLite database inside the
6
+ atomic staging directory, so memory stays bounded.
7
+
8
+ Exactness and determinism:
9
+
10
+ * Costs are stored as **TEXT** and summed with Python ``Decimal`` during the ordered scan —
11
+ never ``SUM()`` in SQL — so the streamed sum is bit-for-bit the eager sum.
12
+ * Every finalize scan is ``ORDER BY <keys>`` under the default **BINARY** collation, which
13
+ compares UTF-8 bytes and so matches Python ``sorted()`` on the same string tuples.
14
+
15
+ The scratch database is disposable (``journal_mode=OFF``, ``synchronous=OFF``); durability
16
+ comes from the atomic writer's fsync of the finished data files, not from this DB.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import sqlite3
22
+ from collections.abc import Iterator
23
+ from decimal import Decimal
24
+ from pathlib import Path
25
+
26
+ from focus_data_toolkit.convert.invoice_detail import GrainKey
27
+
28
+ # Invoice Detail business-grain columns, in key order (matches GRAIN_FIELDS).
29
+ _GRAIN_COLS = (
30
+ "issuer",
31
+ "invoice_id",
32
+ "account",
33
+ "currency",
34
+ "bp_start",
35
+ "bp_end",
36
+ "charge_category",
37
+ )
38
+
39
+
40
+ class ExternalIndex:
41
+ """SQLite-backed staging for Invoice Detail aggregation and Billing Period dedup."""
42
+
43
+ def __init__(self, db_path: str | Path) -> None:
44
+ self._conn = sqlite3.connect(str(db_path))
45
+ for pragma in ("journal_mode=OFF", "synchronous=OFF", "temp_store=FILE", "cache_size=-20000"):
46
+ self._conn.execute(f"PRAGMA {pragma}")
47
+ self._conn.execute(
48
+ "CREATE TABLE id_stage (n INTEGER PRIMARY KEY, "
49
+ + ", ".join(f"{col} TEXT" for col in _GRAIN_COLS)
50
+ + ", billed_cost TEXT)"
51
+ )
52
+ self._conn.execute(
53
+ "CREATE TABLE bp (start TEXT, end TEXT, issuer TEXT, "
54
+ "PRIMARY KEY (start, end, issuer)) WITHOUT ROWID"
55
+ )
56
+ self._insert_line = (
57
+ "INSERT INTO id_stage (" + ", ".join(_GRAIN_COLS) + ", billed_cost) VALUES ("
58
+ + ", ".join("?" * (len(_GRAIN_COLS) + 1)) + ")"
59
+ )
60
+
61
+ def stage_invoice_line(self, grain_key: GrainKey, billed_cost: str) -> None:
62
+ """Record one Cost and Usage line's contribution to its invoice-detail group."""
63
+ self._conn.execute(self._insert_line, (*grain_key, billed_cost))
64
+
65
+ def stage_billing_period(self, start: str, end: str, issuer: str) -> None:
66
+ """Record a (start, end, issuer) billing period (first occurrence wins)."""
67
+ self._conn.execute(
68
+ "INSERT OR IGNORE INTO bp (start, end, issuer) VALUES (?, ?, ?)", (start, end, issuer)
69
+ )
70
+
71
+ def finalize_invoice_groups(self) -> Iterator[tuple[GrainKey, Decimal]]:
72
+ """Yield ``(grain_key, summed_billed_cost)`` per group, in sorted grain order."""
73
+ self._conn.commit()
74
+ order = ", ".join(_GRAIN_COLS) + ", n"
75
+ cursor = self._conn.execute(
76
+ f"SELECT {', '.join(_GRAIN_COLS)}, billed_cost FROM id_stage ORDER BY {order}"
77
+ )
78
+ current: GrainKey | None = None
79
+ total = Decimal(0)
80
+ for row in cursor:
81
+ key: GrainKey = tuple(row[: len(_GRAIN_COLS)])
82
+ cost = row[len(_GRAIN_COLS)]
83
+ if current is None:
84
+ current = key
85
+ if key != current:
86
+ yield current, total
87
+ current = key
88
+ total = Decimal(0)
89
+ total += Decimal(cost or "0")
90
+ if current is not None:
91
+ yield current, total
92
+
93
+ def finalize_billing_periods(self) -> Iterator[tuple[str, str, str]]:
94
+ """Yield distinct ``(start, end, issuer)`` billing periods, in sorted order."""
95
+ self._conn.commit()
96
+ yield from self._conn.execute("SELECT start, end, issuer FROM bp ORDER BY start, end, issuer")
97
+
98
+ def close(self) -> None:
99
+ self._conn.close()
@@ -0,0 +1,150 @@
1
+ """Threshold-spilling string maps for streaming cross-dataset validation.
2
+
3
+ The bundle validator's checks keep per-key state (seen ids, foreign-key targets, running
4
+ sums) whose cardinality scales with the *large* Cost and Usage dataset. A
5
+ :class:`SpillableIndexPool` hands out ``str -> str`` mutable mappings that live in an
6
+ ordinary in-memory ``dict`` until a size threshold, then migrate transparently into a shared
7
+ throwaway SQLite database — so validating a bundle far larger than RAM stays bounded.
8
+
9
+ Like :mod:`focus_data_toolkit.storage.external_index`, the database is disposable scratch
10
+ state (``journal_mode=OFF``, ``synchronous=OFF``): durability comes from the atomic writer's
11
+ fsync of the published files, never from this DB. The file is created lazily on the first
12
+ spill, so small bundles touch the disk not at all.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import sqlite3
18
+ from collections.abc import Iterator, MutableMapping
19
+ from pathlib import Path
20
+
21
+ # Keys held in memory per map before spilling to SQLite. At ~100 bytes per key/value pair
22
+ # this bounds each map's resident size to roughly 20 MB worst-case before it moves to disk.
23
+ DEFAULT_SPILL_THRESHOLD = 200_000
24
+
25
+
26
+ class SpillableIndexPool:
27
+ """Factory of :class:`SpillableMap` instances sharing one lazy scratch SQLite database."""
28
+
29
+ def __init__(self, db_path: str | Path, *, threshold: int = DEFAULT_SPILL_THRESHOLD) -> None:
30
+ if threshold < 1:
31
+ raise ValueError(f"spill threshold must be >= 1, got {threshold}")
32
+ self._db_path = Path(db_path)
33
+ self._threshold = threshold
34
+ self._conn: sqlite3.Connection | None = None
35
+ self._tables = 0
36
+ self._spilled = False
37
+
38
+ @property
39
+ def threshold(self) -> int:
40
+ return self._threshold
41
+
42
+ @property
43
+ def spilled(self) -> bool:
44
+ """Whether any map has ever spilled (i.e. the scratch database was created)."""
45
+ return self._spilled
46
+
47
+ def make_map(self) -> SpillableMap:
48
+ """Return a fresh empty ``str -> str`` mapping backed by this pool."""
49
+ self._tables += 1
50
+ return SpillableMap(self, f"kv{self._tables}")
51
+
52
+ def _connection(self) -> sqlite3.Connection:
53
+ if self._conn is None:
54
+ self._conn = sqlite3.connect(str(self._db_path))
55
+ self._spilled = True
56
+ for pragma in (
57
+ "journal_mode=OFF",
58
+ "synchronous=OFF",
59
+ "temp_store=FILE",
60
+ "cache_size=-20000",
61
+ ):
62
+ self._conn.execute(f"PRAGMA {pragma}")
63
+ return self._conn
64
+
65
+ def close(self) -> None:
66
+ if self._conn is not None:
67
+ self._conn.close()
68
+ self._conn = None
69
+
70
+
71
+ class SpillableMap(MutableMapping[str, str]):
72
+ """A ``str -> str`` mapping that spills from a dict to the pool's SQLite past a threshold."""
73
+
74
+ def __init__(self, pool: SpillableIndexPool, table: str) -> None:
75
+ self._pool = pool
76
+ self._table = table
77
+ self._mem: dict[str, str] | None = {}
78
+
79
+ def _spill(self) -> None:
80
+ assert self._mem is not None
81
+ conn = self._pool._connection()
82
+ conn.execute(f"CREATE TABLE {self._table} (k TEXT PRIMARY KEY, v TEXT) WITHOUT ROWID")
83
+ conn.executemany(
84
+ f"INSERT INTO {self._table} (k, v) VALUES (?, ?)", self._mem.items()
85
+ )
86
+ self._mem = None
87
+
88
+ def __setitem__(self, key: str, value: str) -> None:
89
+ if self._mem is not None:
90
+ self._mem[key] = value
91
+ if len(self._mem) > self._pool.threshold:
92
+ self._spill()
93
+ return
94
+ self._pool._connection().execute(
95
+ f"INSERT OR REPLACE INTO {self._table} (k, v) VALUES (?, ?)", (key, value)
96
+ )
97
+
98
+ def __getitem__(self, key: str) -> str:
99
+ if self._mem is not None:
100
+ return self._mem[key]
101
+ row = (
102
+ self._pool._connection()
103
+ .execute(f"SELECT v FROM {self._table} WHERE k = ?", (key,))
104
+ .fetchone()
105
+ )
106
+ if row is None:
107
+ raise KeyError(key)
108
+ return row[0]
109
+
110
+ def __delitem__(self, key: str) -> None:
111
+ if self._mem is not None:
112
+ del self._mem[key]
113
+ return
114
+ cursor = self._pool._connection().execute(
115
+ f"DELETE FROM {self._table} WHERE k = ?", (key,)
116
+ )
117
+ if cursor.rowcount == 0:
118
+ raise KeyError(key)
119
+
120
+ def __contains__(self, key: object) -> bool:
121
+ if self._mem is not None:
122
+ return key in self._mem
123
+ if not isinstance(key, str):
124
+ return False
125
+ row = (
126
+ self._pool._connection()
127
+ .execute(f"SELECT 1 FROM {self._table} WHERE k = ?", (key,))
128
+ .fetchone()
129
+ )
130
+ return row is not None
131
+
132
+ def __len__(self) -> int:
133
+ if self._mem is not None:
134
+ return len(self._mem)
135
+ return self._pool._connection().execute(
136
+ f"SELECT COUNT(*) FROM {self._table}"
137
+ ).fetchone()[0]
138
+
139
+ def __iter__(self) -> Iterator[str]:
140
+ if self._mem is not None:
141
+ yield from self._mem
142
+ return
143
+ # Sorted (BINARY collation = UTF-8 byte order) for a deterministic iteration order.
144
+ for (key,) in self._pool._connection().execute(
145
+ f"SELECT k FROM {self._table} ORDER BY k"
146
+ ):
147
+ yield key
148
+
149
+
150
+ __all__ = ["DEFAULT_SPILL_THRESHOLD", "SpillableIndexPool", "SpillableMap"]
@@ -0,0 +1,19 @@
1
+ """Focus Data Toolkit Studio — a LOCAL web UI over the same Core.
2
+
3
+ The Studio never reimplements FOCUS logic: it drives the exact SDK the CLI and Runner use
4
+ (``detect_focus_schema``, ``convert_files``, ``validate_dataset_bundle``, the generators), so its
5
+ manifests, diagnostics and checksums are identical to a CLI run. It is designed for **local,
6
+ single-user** use: it binds to loopback by default, requires a per-start token, validates
7
+ Host/Origin headers, confines file access to an allowlisted root, and processes on the bounded
8
+ streaming path — data never leaves the machine.
9
+
10
+ This subpackage lives behind the optional ``[studio]`` extra (FastAPI + uvicorn). The CLI's
11
+ ``focus-toolkit ui`` command imports it lazily, so a core install without the extra still works.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from focus_data_toolkit.studio.config import StudioConfig
17
+ from focus_data_toolkit.studio.server import run
18
+
19
+ __all__ = ["StudioConfig", "run"]