focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,968 @@
1
+ """focus-toolkit command line interface.
2
+
3
+ Subcommands:
4
+
5
+ * ``generate`` — emit provider-realistic FOCUS 1.2/1.3 sample CSVs.
6
+ * ``convert`` — convert a FOCUS 1.2/1.3 source (CSV or Parquet) into the four
7
+ FOCUS 1.4 datasets, optionally completed by ``--supplement`` client facts.
8
+ * ``gaps`` — report exactly which facts a client must supply for the four
9
+ FOCUS 1.4 datasets to be produced factually from a given source.
10
+ * ``supplements`` — pre-flight ``validate`` supplement files against a source, and
11
+ list the provider-native export ``adapters`` (AWS / Azure / GCP).
12
+ * ``detect`` — detect the FOCUS dataset / version of a file's header.
13
+ * ``validate`` — validate a produced file against the built-in FOCUS 1.4 model,
14
+ or run the official FinOps validator (``--official``).
15
+ * ``validate-bundle`` — run the cross-dataset validation gate over a bundle of
16
+ FOCUS 1.4 datasets (explicit per-dataset files or an auto-detected directory).
17
+ * ``version`` — print the toolkit version and optional-extra availability.
18
+ * ``ui`` — launch the local Studio web UI (needs the ``[studio]`` extra;
19
+ localhost-only by default).
20
+ * ``clean`` — recover interrupted publishes and remove leftover staging
21
+ directories.
22
+
23
+ Exit codes (``convert``): 0 ok; 1 lint/bundle/write failure; 2 invalid input/args;
24
+ 3 strict mode left datasets NOT_PRODUCED; 4 synthetic assumptions present; 5 disk
25
+ budget / free-space exhaustion; 130 cancelled (SIGINT/SIGTERM). ``--exit-policy
26
+ pipeline`` maps 3 and 4 to 0 for orchestrators.
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ import argparse
32
+ import contextlib
33
+ import json
34
+ import shutil
35
+ import signal
36
+ import sys
37
+ import threading
38
+ from pathlib import Path
39
+
40
+ from focus_data_toolkit.convert import (
41
+ OUTPUT_FORMATS,
42
+ AtomicWriteError,
43
+ ConversionCancelled,
44
+ ConversionError,
45
+ DestinationExistsError,
46
+ OnExists,
47
+ convert_files,
48
+ convert_to_focus_1_4,
49
+ write_result,
50
+ )
51
+ from focus_data_toolkit.generators import FOCUS_VERSIONS, PROVIDERS, get_generator
52
+ from focus_data_toolkit.io.parquet_io import COMPRESSIONS
53
+ from focus_data_toolkit.io.records import MalformedRecordError
54
+ from focus_data_toolkit.io.row_source import read_source_rows
55
+ from focus_data_toolkit.manifest import render as render_manifest
56
+ from focus_data_toolkit.model.capabilities import KNOWN_CONDITIONS, CapabilityProfile
57
+ from focus_data_toolkit.model.validator import lint_focus_1_4_structure, resolve_dataset
58
+ from focus_data_toolkit.modes import Mode
59
+ from focus_data_toolkit.runtime import ResourceLimitError, parse_size
60
+
61
+
62
+ def _parse_size(value: str | None) -> int | None:
63
+ """Parse a byte size for --target-file-size (e.g. ``128MB``, ``512KB``, or a byte count)."""
64
+ try:
65
+ return parse_size(value)
66
+ except ValueError as exc:
67
+ raise ConversionError(
68
+ f"invalid --target-file-size {value!r}: use e.g. 128MB, 512KB, or a byte count"
69
+ ) from exc
70
+
71
+
72
+ def _apply_exit_policy(code: int, policy: str) -> int:
73
+ """Map functional exit codes to a pipeline-friendly scheme when requested.
74
+
75
+ In ``pipeline`` mode a functional-but-complete outcome — ``3`` (strict mode left some
76
+ datasets NOT_PRODUCED) or ``4`` (synthetic assumptions present) — is reported as success
77
+ (``0``), so orchestrators (Kubernetes / Airflow / Jenkins / AWS Batch), which treat any
78
+ non-zero code as failure, do not mark a legitimate run failed. Genuine failures (``1`` /
79
+ ``2`` / ``5`` / ``130``) stay non-zero. ``detailed`` (default) keeps the historic codes;
80
+ the full functional status is always in the manifest and the ``_run.json`` sidecar.
81
+ """
82
+ if policy == "pipeline" and code in (3, 4):
83
+ return 0
84
+ return code
85
+
86
+
87
+ @contextlib.contextmanager
88
+ def _cancel_on_signals():
89
+ """Yield a threading.Event set by SIGINT/SIGTERM, restoring prior handlers on exit.
90
+
91
+ A batch run in a container receives SIGTERM on ``docker stop`` / pod eviction and SIGINT
92
+ on Ctrl-C; turning both into a cooperative cancel lets the streaming engine unwind cleanly
93
+ (staging removed, nothing published) instead of dying mid-write. Signal handlers can only
94
+ be installed on the main thread, so off-main-thread callers (e.g. a server) silently get an
95
+ event they can set themselves.
96
+ """
97
+ event = threading.Event()
98
+
99
+ def _handler(_signum, _frame):
100
+ event.set()
101
+
102
+ previous: dict = {}
103
+ for sig in (signal.SIGINT, signal.SIGTERM):
104
+ try:
105
+ previous[sig] = signal.signal(sig, _handler)
106
+ except (ValueError, OSError): # not on the main thread
107
+ pass
108
+ try:
109
+ yield event
110
+ finally:
111
+ for sig, prev in previous.items():
112
+ with contextlib.suppress(ValueError, OSError):
113
+ signal.signal(sig, prev)
114
+
115
+
116
+ def _stderr_progress():
117
+ """Return a ProgressCallback that renders a single throttled status line on stderr."""
118
+ state = {"width": 0}
119
+
120
+ def render(event) -> None:
121
+ pct = "" if event.fraction is None else f" {event.fraction * 100:5.1f}%"
122
+ total = "" if event.total is None else f"/{event.total:,}"
123
+ message = f" - {event.message}" if event.message else ""
124
+ line = f"{event.phase} {event.completed:,}{total} {event.unit}{pct}{message}"
125
+ pad = max(0, state["width"] - len(line))
126
+ state["width"] = len(line)
127
+ sys.stderr.write("\r" + line + " " * pad)
128
+ sys.stderr.flush()
129
+
130
+ return render
131
+
132
+
133
+ def _cmd_generate(args: argparse.Namespace) -> int:
134
+ module = get_generator(args.provider, args.focus_version)
135
+ out_dir = Path(args.out)
136
+ out_dir.mkdir(parents=True, exist_ok=True)
137
+ suffix = args.focus_version.replace(".", "_")
138
+
139
+ cau = out_dir / f"focus_{suffix}_cost_and_usage_{args.provider}.csv"
140
+ cau.write_bytes(module.generate_csv_bytes(args.rows, args.seed))
141
+ print(f"wrote {cau} ({args.rows} rows, seed {args.seed})")
142
+
143
+ if args.focus_version == "1.3":
144
+ cc = out_dir / f"focus_{suffix}_contract_commitment_{args.provider}.csv"
145
+ cc.write_bytes(module.generate_contract_commitment_csv_bytes(args.rows, args.seed))
146
+ print(f"wrote {cc}")
147
+ return 0
148
+
149
+
150
+ def _read_header(path: str) -> tuple[str, ...]:
151
+ """Read only the header of a CSV (gzip auto-detected) or Parquet file."""
152
+ from focus_data_toolkit.io.row_source import open_row_source
153
+
154
+ reader = open_row_source(path)
155
+ try:
156
+ return reader.source_columns
157
+ finally:
158
+ reader.close()
159
+
160
+
161
+ def _cmd_gaps(args: argparse.Namespace) -> int:
162
+ from focus_data_toolkit.convert import _resolve_source_version
163
+ from focus_data_toolkit.supplement import compute_gaps
164
+
165
+ try:
166
+ header = _read_header(args.cost_and_usage)
167
+ version, _detection = _resolve_source_version(
168
+ header,
169
+ source_version=args.source_version,
170
+ source_dataset=args.source_dataset,
171
+ mode=Mode.STRICT,
172
+ )
173
+ cc_header = (
174
+ _read_header(args.contract_commitment) if args.contract_commitment else None
175
+ )
176
+ except (ConversionError, MalformedRecordError) as exc:
177
+ # MalformedRecordError: unreadable source header (malformed CSV, corrupt Parquet,
178
+ # or the missing-pyarrow install hint) — a CLI error, not a traceback.
179
+ print(f"error: {exc}", file=sys.stderr)
180
+ return 2
181
+ report = compute_gaps(header, version, cc_columns=cc_header)
182
+ payload = (
183
+ json.dumps(report.as_dict(), indent=2, sort_keys=True) + "\n"
184
+ if args.format == "json"
185
+ else report.render_text()
186
+ )
187
+ if args.out:
188
+ Path(args.out).write_text(payload, encoding="utf-8")
189
+ print(f"wrote {args.out}")
190
+ else:
191
+ print(payload, end="")
192
+ return 0
193
+
194
+
195
+ def _supplement_specs(args: argparse.Namespace) -> list:
196
+ """Collect supplement file specs from --supplement / --supplements-dir."""
197
+ from focus_data_toolkit.supplement import load_bundle_dir, parse_supplement_arg
198
+
199
+ specs = [parse_supplement_arg(arg) for arg in (args.supplement or [])]
200
+ if getattr(args, "supplements_dir", None):
201
+ specs.extend(load_bundle_dir(args.supplements_dir))
202
+ return specs
203
+
204
+
205
+ def _cmd_supplements_validate(args: argparse.Namespace) -> int:
206
+ from focus_data_toolkit.supplement import (
207
+ SupplementBundle,
208
+ SupplementError,
209
+ source_key_sets,
210
+ validate_supplements,
211
+ )
212
+ from focus_data_toolkit.supplement.validate import has_blocking_errors
213
+
214
+ try:
215
+ specs = _supplement_specs(args)
216
+ if not specs:
217
+ print("error: provide --supplement and/or --supplements-dir", file=sys.stderr)
218
+ return 2
219
+ bundle = SupplementBundle.load(specs)
220
+ except SupplementError as exc:
221
+ print(f"error: {exc}", file=sys.stderr)
222
+ return 2
223
+ try:
224
+ cau_rows = read_source_rows(args.cost_and_usage)
225
+ cc_rows = (
226
+ read_source_rows(args.contract_commitment) if args.contract_commitment else None
227
+ )
228
+ except MalformedRecordError as exc:
229
+ print(f"error: {exc}", file=sys.stderr)
230
+ return 2
231
+ diagnostics = validate_supplements(bundle, source_key_sets(cau_rows, cc_rows))
232
+ for diag in diagnostics:
233
+ print(f"{diag.severity} {diag.code}: {diag.message}")
234
+ for key, value in diag.context.items():
235
+ print(f" {key}: {value}")
236
+ if has_blocking_errors(diagnostics):
237
+ print("supplements NOT usable: fix the ERROR diagnostics above", file=sys.stderr)
238
+ return 1
239
+ kinds = ", ".join(sorted(bundle.tables)) or "-"
240
+ print(f"supplements OK ({kinds}); see FDT-SUPP-010 entries for coverage")
241
+ return 0
242
+
243
+
244
+ def _cmd_supplements_adapters(args: argparse.Namespace) -> int:
245
+ from focus_data_toolkit.supplement.adapters import load_adapters
246
+
247
+ adapters = load_adapters()
248
+ if not adapters:
249
+ print("no provider adapters available")
250
+ return 0
251
+ for name in sorted(adapters):
252
+ a = adapters[name]
253
+ print(f"{name} (v{a.version}) -> {a.target_kind}")
254
+ print(f" source: {a.provenance.get('source', '?')}")
255
+ print(f" doc: {a.provenance.get('doc_url', '?')}")
256
+ return 0
257
+
258
+
259
+ def _capabilities(args: argparse.Namespace) -> CapabilityProfile:
260
+ """Build the capability profile from repeated ``--supports`` flags."""
261
+ return CapabilityProfile(frozenset(args.supports), source="cli") if args.supports \
262
+ else CapabilityProfile.none()
263
+
264
+
265
+ def _cmd_convert_stream(args: argparse.Namespace, mode: Mode) -> int:
266
+ """Bounded-memory streaming conversion (required for Parquet output / large inputs)."""
267
+ from focus_data_toolkit.supplement import SupplementError
268
+
269
+ partition_by = [c.strip() for c in (args.partition_by or "").split(",") if c.strip()]
270
+ progress = _stderr_progress() if getattr(args, "progress", False) else None
271
+ try:
272
+ target_file_size = _parse_size(args.target_file_size)
273
+ with _cancel_on_signals() as cancel_event:
274
+ out = convert_files(
275
+ args.cost_and_usage,
276
+ args.out,
277
+ contract_commitment=args.contract_commitment,
278
+ source_version=args.source_version,
279
+ source_dataset=args.source_dataset,
280
+ mode=mode,
281
+ validate=not args.no_validate,
282
+ on_exists=OnExists(args.on_exists),
283
+ keep_temp=args.keep_temp,
284
+ output_format=args.output_format,
285
+ partition_by=partition_by,
286
+ compression=args.compression,
287
+ target_file_size=target_file_size,
288
+ capabilities=_capabilities(args),
289
+ supplements=_load_supplements(args),
290
+ progress=progress,
291
+ cancel=cancel_event.is_set,
292
+ )
293
+ except ConversionCancelled:
294
+ # Subclass of ConversionError — must be caught first. Nothing was published.
295
+ if progress:
296
+ sys.stderr.write("\n")
297
+ print("cancelled: no output written", file=sys.stderr)
298
+ return 130
299
+ except ResourceLimitError as exc:
300
+ if progress:
301
+ sys.stderr.write("\n")
302
+ print(f"error: [{exc.diagnostic.code}] {exc.diagnostic.message}", file=sys.stderr)
303
+ return 5
304
+ except SupplementError as exc:
305
+ print(f"error: {exc}", file=sys.stderr)
306
+ return 2
307
+ except (ConversionError, DestinationExistsError, MalformedRecordError) as exc:
308
+ # MalformedRecordError covers a malformed CSV record and a missing PyArrow (the clear
309
+ # install hint) — surface both as a normal CLI error, not a traceback.
310
+ print(f"error: {exc}", file=sys.stderr)
311
+ return 2
312
+ except AtomicWriteError as exc:
313
+ print(f"error: {exc}", file=sys.stderr)
314
+ return 1
315
+
316
+ if progress:
317
+ sys.stderr.write("\n")
318
+
319
+ from focus_data_toolkit.manifest import NOT_PRODUCED
320
+
321
+ published_manifest = out / "focus_1_4_manifest.json"
322
+ if args.manifest: # honour --manifest for the streaming path too (parity with eager)
323
+ Path(args.manifest).write_text(
324
+ published_manifest.read_text(encoding="utf-8"), encoding="utf-8"
325
+ )
326
+ manifest = json.loads(published_manifest.read_text(encoding="utf-8"))
327
+ for diag in manifest.get("diagnostics", []):
328
+ print(f"note {diag.get('code')}: {diag.get('message')}", file=sys.stderr)
329
+ print(f"wrote {out}/ (format {args.output_format}, mode {mode})")
330
+ for name, entry in manifest["datasets"].items():
331
+ if entry.get("status") == NOT_PRODUCED:
332
+ print(f"not produced [{name}]: {entry.get('reason', 'unavailable')}")
333
+ if mode is Mode.SYNTHETIC and manifest.get("assumptions_present"):
334
+ print(
335
+ "WARNING: synthetic mode — datasets marked PRODUCED_SYNTHETIC contain ASSUMED "
336
+ "values and are NOT fully FOCUS-conformant. See the manifest.",
337
+ file=sys.stderr,
338
+ )
339
+ return _apply_exit_policy(4, args.exit_policy)
340
+ if mode is Mode.STRICT and any(
341
+ e.get("status") == NOT_PRODUCED for e in manifest["datasets"].values()
342
+ ):
343
+ return _apply_exit_policy(3, args.exit_policy)
344
+ return 0
345
+
346
+
347
+ def _load_supplements(args: argparse.Namespace):
348
+ """Load the supplement bundle from CLI args (None when no supplements given)."""
349
+ from focus_data_toolkit.supplement import SupplementBundle
350
+
351
+ specs = _supplement_specs(args)
352
+ return SupplementBundle.load(specs) if specs else None
353
+
354
+
355
+ def _cmd_convert(args: argparse.Namespace) -> int:
356
+ mode = Mode(args.mode)
357
+ # Parquet output, explicit --stream, or partitioning go through the bounded-memory streaming
358
+ # engine; the eager path (rich per-dataset reporting) stays the default for CSV output.
359
+ if args.output_format == "parquet" or args.stream or args.partition_by:
360
+ return _cmd_convert_stream(args, mode)
361
+
362
+ from focus_data_toolkit.supplement import SupplementError
363
+
364
+ try:
365
+ cau_rows = read_source_rows(args.cost_and_usage)
366
+ cc_rows = (
367
+ read_source_rows(args.contract_commitment) if args.contract_commitment else None
368
+ )
369
+ result = convert_to_focus_1_4(
370
+ cau_rows,
371
+ cc_rows,
372
+ source_version=args.source_version,
373
+ source_dataset=args.source_dataset,
374
+ mode=mode,
375
+ validate=not args.no_validate,
376
+ capabilities=_capabilities(args),
377
+ supplements=_load_supplements(args),
378
+ )
379
+ except SupplementError as exc:
380
+ print(f"error: {exc}", file=sys.stderr)
381
+ return 2
382
+ except (ConversionError, MalformedRecordError) as exc:
383
+ # MalformedRecordError covers a malformed source record and a missing PyArrow
384
+ # (its clear install hint) for Parquet input — a CLI error, not a traceback.
385
+ print(f"error: {exc}", file=sys.stderr)
386
+ return 2
387
+
388
+ confidence = result.detection.confidence if result.detection else "?"
389
+ print(
390
+ f"source detected: FOCUS {result.source_version} "
391
+ f"(dataset {result.detection.dataset if result.detection else '?'}, "
392
+ f"confidence {confidence}, mode: {mode})"
393
+ )
394
+ for diag in result.diagnostics:
395
+ print(f"note {diag.code}: {diag.message}", file=sys.stderr)
396
+
397
+ # Mandatory lint gate: a lint-failing result is never written to disk.
398
+ if not args.no_validate and not result.ok:
399
+ for name, report in result.reports.items():
400
+ status = "lint OK" if report.ok else f"{len(report.violations)} violation(s)"
401
+ print(f"lint [{name}]: {status}")
402
+ for message in report.messages()[:20]:
403
+ print(f" {message}")
404
+ print("output not written: mandatory lint failed", file=sys.stderr)
405
+ return 1
406
+
407
+ try:
408
+ written = write_result(
409
+ result, args.out, on_exists=OnExists(args.on_exists), keep_temp=args.keep_temp,
410
+ validate_bundle=not args.no_validate,
411
+ )
412
+ except DestinationExistsError as exc:
413
+ print(f"error: {exc}", file=sys.stderr)
414
+ return 2
415
+ except AtomicWriteError as exc:
416
+ print(f"error: {exc}", file=sys.stderr)
417
+ return 1
418
+
419
+ if args.manifest:
420
+ Path(args.manifest).write_text(render_manifest(result.manifest), encoding="utf-8")
421
+ for path in written:
422
+ print(f"wrote {path}")
423
+ for name in result.not_produced:
424
+ entry = result.manifest["datasets"][name]
425
+ print(f"not produced [{name}]: {entry.get('reason', 'unavailable')}")
426
+ if not args.no_validate:
427
+ for name in result.reports:
428
+ print(f"lint [{name}]: lint OK")
429
+
430
+ if mode is Mode.SYNTHETIC and result.assumptions_present:
431
+ print(
432
+ "WARNING: synthetic mode — datasets marked PRODUCED_SYNTHETIC contain ASSUMED "
433
+ "values and are NOT fully FOCUS-conformant. See the manifest.",
434
+ file=sys.stderr,
435
+ )
436
+ return _apply_exit_policy(4, args.exit_policy)
437
+ if mode is Mode.STRICT and result.not_produced:
438
+ return _apply_exit_policy(3, args.exit_policy)
439
+ return 0
440
+
441
+
442
+ def _cmd_validate(args: argparse.Namespace) -> int:
443
+ if args.official:
444
+ from focus_data_toolkit.official_validator import run_official_validator
445
+
446
+ if not args.focus_version:
447
+ print("--official requires --focus-version (e.g. 1.2.0.1)", file=sys.stderr)
448
+ return 2
449
+ return run_official_validator(args.file, args.focus_version)
450
+
451
+ dataset = resolve_dataset(args.dataset.replace("-", " "))
452
+ try:
453
+ rows = read_source_rows(args.file, dataset=dataset)
454
+ except MalformedRecordError as exc:
455
+ print(f"error: {exc}", file=sys.stderr)
456
+ return 2
457
+ report = lint_focus_1_4_structure(dataset, rows, profile=_capabilities(args))
458
+ status = (
459
+ f"structural+semantic lint OK ({', '.join(report.levels_passed)})"
460
+ if report.ok
461
+ else f"{len(report.violations)} violation(s)"
462
+ )
463
+ print(f"{args.file}: {status}")
464
+ if report.ok:
465
+ print(" note: structural lint only — not a full FOCUS 1.4 conformance check")
466
+ for message in report.messages()[:50]:
467
+ print(f" {message}")
468
+ return 0 if report.ok else 1
469
+
470
+
471
+ def _cmd_detect(args: argparse.Namespace) -> int:
472
+ from focus_data_toolkit.schema import detect_focus_schema
473
+
474
+ try:
475
+ header = _read_header(args.file)
476
+ except MalformedRecordError as exc:
477
+ print(f"error: {exc}", file=sys.stderr)
478
+ return 2
479
+ try:
480
+ forced_dataset = resolve_dataset(args.dataset.replace("-", " ")) if args.dataset else None
481
+ result = detect_focus_schema(header, dataset=forced_dataset, version=args.version)
482
+ except ValueError as exc:
483
+ print(f"error: {exc}", file=sys.stderr)
484
+ return 2
485
+
486
+ if args.format == "json":
487
+ print(json.dumps(result.as_dict(), indent=2, sort_keys=True))
488
+ return 0
489
+ print(
490
+ f"{args.file}: dataset={result.dataset or '?'} "
491
+ f"version={result.detected_version or '?'} "
492
+ f"confidence={result.confidence} (score {result.score:.3f})"
493
+ )
494
+ if result.missing_columns:
495
+ print(f" missing: {', '.join(result.missing_columns)}")
496
+ if result.unknown_columns:
497
+ print(f" unknown: {', '.join(result.unknown_columns)}")
498
+ if result.extension_columns:
499
+ print(f" x_ extensions: {', '.join(result.extension_columns)}")
500
+ for note in result.notes:
501
+ print(f" note: {note}")
502
+ return 0
503
+
504
+
505
+ class _FileRows:
506
+ """Re-iterable row source over a bundle input file (CSV/Parquet) for ``validate-bundle``.
507
+
508
+ ``validate_dataset_bundle`` makes several independent forward passes, so each ``__iter__``
509
+ opens a fresh reader — the dataset is never materialised, keeping memory bounded.
510
+ """
511
+
512
+ def __init__(self, path: str, dataset: str) -> None:
513
+ self._path = path
514
+ self._dataset = dataset
515
+
516
+ def __iter__(self):
517
+ from focus_data_toolkit.io.row_source import open_row_source
518
+
519
+ with contextlib.closing(open_row_source(self._path, dataset=self._dataset)) as reader:
520
+ for record in reader:
521
+ yield record.values
522
+
523
+
524
+ def _detect_bundle_dir(directory: str) -> dict[str, str]:
525
+ """Map each FOCUS dataset in ``directory`` to its file path (auto-detected from headers).
526
+
527
+ Raises ``ValueError`` when two files look like the same dataset (ambiguous) or none are FOCUS.
528
+ """
529
+ from focus_data_toolkit.schema import detect_focus_schema
530
+
531
+ base = Path(directory)
532
+ if not base.is_dir():
533
+ raise ValueError(f"not a directory: {directory}")
534
+ found: dict[str, Path] = {}
535
+ for path in sorted(base.iterdir()):
536
+ if path.name.startswith(".") or path.name == "SHA256SUMS":
537
+ continue
538
+ if path.is_file() and path.suffix.lower() not in (".csv", ".gz", ".parquet"):
539
+ continue
540
+ # A directory is a candidate only if it looks like a Hive-partitioned Parquet dataset
541
+ # (COL=value subdirectories) — which open_row_source reads natively. Without this, a
542
+ # partitioned Cost and Usage dataset would be skipped and its cross-dataset checks lost.
543
+ if path.is_dir() and not _looks_partitioned(path):
544
+ continue
545
+ try:
546
+ header = _read_header(str(path))
547
+ except (MalformedRecordError, OSError):
548
+ continue
549
+ result = detect_focus_schema(header)
550
+ if result.dataset is None or result.confidence == "LOW":
551
+ continue
552
+ if result.detected_version != "1.4":
553
+ continue # validate-bundle validates a FOCUS 1.4 bundle; skip 1.2/1.3 exports
554
+ if result.dataset in found:
555
+ raise ValueError(
556
+ f"ambiguous bundle: {found[result.dataset].name} and {path.name} both look "
557
+ f"like {result.dataset}; use the per-dataset file flags instead"
558
+ )
559
+ found[result.dataset] = path
560
+ if not found:
561
+ raise ValueError(f"no FOCUS 1.4 datasets detected under {directory}")
562
+ return {name: str(path) for name, path in found.items()}
563
+
564
+
565
+ def _looks_partitioned(directory: Path) -> bool:
566
+ """Whether ``directory`` has ``COL=value`` subdirectories (a Hive-partitioned dataset root)."""
567
+ try:
568
+ return any(child.is_dir() and "=" in child.name for child in directory.iterdir())
569
+ except OSError:
570
+ return False
571
+
572
+
573
+ def _cmd_validate_bundle(args: argparse.Namespace) -> int:
574
+ import tempfile
575
+
576
+ from focus_data_toolkit.storage.spill import SpillableIndexPool
577
+ from focus_data_toolkit.validate.bundle import validate_dataset_bundle
578
+
579
+ explicit = {
580
+ "Cost and Usage": args.cost_and_usage,
581
+ "Contract Commitment": args.contract_commitment,
582
+ "Billing Period": args.billing_period,
583
+ "Invoice Detail": args.invoice_detail,
584
+ }
585
+ given = {name: path for name, path in explicit.items() if path}
586
+ if args.directory and given:
587
+ print(
588
+ "error: use either --directory or the per-dataset file flags, not both",
589
+ file=sys.stderr,
590
+ )
591
+ return 2
592
+ if not args.directory and not given:
593
+ print(
594
+ "error: provide --directory or at least one per-dataset file flag (e.g. "
595
+ "--cost-and-usage FILE)",
596
+ file=sys.stderr,
597
+ )
598
+ return 2
599
+
600
+ try:
601
+ mapping = _detect_bundle_dir(args.directory) if args.directory else given
602
+ except (MalformedRecordError, ValueError) as exc:
603
+ print(f"error: {exc}", file=sys.stderr)
604
+ return 2
605
+
606
+ bundle = {name: _FileRows(path, name) for name, path in mapping.items()}
607
+ tmpdir = tempfile.mkdtemp(prefix="fdt-bundle-")
608
+ spill = SpillableIndexPool(Path(tmpdir) / "_bundle.sqlite")
609
+ try:
610
+ report = validate_dataset_bundle(bundle, index_factory=spill.make_map)
611
+ except MalformedRecordError as exc:
612
+ print(f"error: {exc}", file=sys.stderr)
613
+ return 2
614
+ finally:
615
+ spill.close()
616
+ shutil.rmtree(tmpdir, ignore_errors=True)
617
+
618
+ if args.report:
619
+ Path(args.report).write_text(
620
+ json.dumps(report.as_dict(), indent=2, sort_keys=True) + "\n", encoding="utf-8"
621
+ )
622
+ # stderr, so `--format json` keeps stdout a single parseable JSON document.
623
+ print(f"wrote {args.report}", file=sys.stderr)
624
+ if args.format == "json":
625
+ print(json.dumps(report.as_dict(), indent=2, sort_keys=True))
626
+ else:
627
+ print(f"validated: {', '.join(sorted(mapping))}")
628
+ print(report.format())
629
+ return 0 if report.ok else 1
630
+
631
+
632
+ def _cmd_version(args: argparse.Namespace) -> int:
633
+ from focus_data_toolkit import __version__
634
+
635
+ try:
636
+ import pyarrow # noqa: F401
637
+
638
+ parquet = "available"
639
+ except ModuleNotFoundError:
640
+ parquet = "not installed (pip install 'focus-data-toolkit[parquet]')"
641
+ print(f"focus-data-toolkit {__version__}")
642
+ print(f" python {sys.version.split()[0]}")
643
+ print(f" parquet: {parquet}")
644
+ return 0
645
+
646
+
647
+ def _cmd_ui(args: argparse.Namespace) -> int:
648
+ try:
649
+ from focus_data_toolkit import studio
650
+ except ImportError:
651
+ print(
652
+ "error: the Studio web UI needs the [studio] extra: "
653
+ "pip install 'focus-data-toolkit[studio]' (or [studio-all] for Parquet)",
654
+ file=sys.stderr,
655
+ )
656
+ return 2
657
+ try:
658
+ max_upload = _parse_size(args.max_upload)
659
+ except ConversionError as exc:
660
+ print(f"error: {exc}", file=sys.stderr)
661
+ return 2
662
+ return studio.run(
663
+ host=args.host,
664
+ port=args.port,
665
+ root=args.root,
666
+ work_dir=args.work_dir,
667
+ allow_remote=args.allow_remote,
668
+ open_browser=not args.no_open_browser,
669
+ max_upload_bytes=max_upload or 200 * 1000 * 1000,
670
+ )
671
+
672
+
673
+ def _cmd_clean(args: argparse.Namespace) -> int:
674
+ from focus_data_toolkit.io.atomic_writer import clean_leftovers
675
+
676
+ # The directory itself may be missing precisely because a crash interrupted a replace
677
+ # mid-swap — recovery restores it from the journal — so only its parent must exist.
678
+ target = Path(args.out)
679
+ if not target.exists() and not target.parent.is_dir():
680
+ print(f"error: neither {target} nor its parent directory exists", file=sys.stderr)
681
+ return 2
682
+ actions = clean_leftovers(target)
683
+ for action in actions:
684
+ print(action)
685
+ if not actions:
686
+ print("nothing to clean")
687
+ return 0
688
+
689
+
690
+ def build_parser() -> argparse.ArgumentParser:
691
+ parser = argparse.ArgumentParser(
692
+ prog="focus-toolkit",
693
+ description="Generate FOCUS 1.2/1.3 sample data, convert it to FOCUS 1.4, validate it.",
694
+ )
695
+ sub = parser.add_subparsers(dest="command", required=True)
696
+
697
+ gen = sub.add_parser("generate", help="generate provider-realistic FOCUS 1.2/1.3 CSVs")
698
+ gen.add_argument("--provider", choices=PROVIDERS, required=True)
699
+ gen.add_argument("--focus-version", choices=FOCUS_VERSIONS, required=True)
700
+ gen.add_argument("--rows", type=int, default=1000)
701
+ gen.add_argument("--seed", type=int, default=1202)
702
+ gen.add_argument("--out", default="out", help="output directory (default: ./out)")
703
+ gen.set_defaults(func=_cmd_generate)
704
+
705
+ conv = sub.add_parser("convert", help="convert FOCUS 1.2/1.3 towards the FOCUS 1.4 datasets")
706
+ conv.add_argument(
707
+ "--cost-and-usage", required=True,
708
+ help="FOCUS 1.2/1.3 Cost and Usage source (CSV, gzip ok, or Parquet)",
709
+ )
710
+ conv.add_argument(
711
+ "--contract-commitment",
712
+ help="optional FOCUS 1.3 Contract Commitment source (CSV or Parquet, 13 columns)"
713
+ )
714
+ conv.add_argument("--out", default="focus-1.4", help="output directory (default: ./focus-1.4)")
715
+ conv.add_argument(
716
+ "--source-version",
717
+ help="force the FOCUS source version (1.2 or 1.3) instead of auto-detecting it",
718
+ )
719
+ conv.add_argument(
720
+ "--source-dataset",
721
+ help="force the source dataset (e.g. cost-and-usage) instead of auto-detecting it",
722
+ )
723
+ conv.add_argument(
724
+ "--mode",
725
+ choices=[m.value for m in Mode],
726
+ default=Mode.STRICT.value,
727
+ help="strict (default): never invent provider facts; synthetic: generate assumed "
728
+ "values for demos/tests (labelled synthetic, not fully conformant)",
729
+ )
730
+ conv.add_argument(
731
+ "--manifest", help="also write the conversion manifest JSON to this path"
732
+ )
733
+ conv.add_argument(
734
+ "--no-validate", action="store_true",
735
+ help="skip the built-in FOCUS 1.4 structural lint and the cross-dataset bundle "
736
+ "validation gate (the skip is recorded in the manifest)",
737
+ )
738
+ conv.add_argument(
739
+ "--supports",
740
+ action="append",
741
+ default=[],
742
+ choices=sorted(KNOWN_CONDITIONS),
743
+ metavar="CONDITION",
744
+ help="declare a FOCUS applicability condition the source supports (repeatable); "
745
+ "conditionally-required columns are enforced only for declared conditions "
746
+ f"(known: {', '.join(sorted(KNOWN_CONDITIONS))})",
747
+ )
748
+ conv.add_argument(
749
+ "--supplement",
750
+ action="append",
751
+ default=[],
752
+ metavar="FILE[:KIND]",
753
+ help="supplemental client facts (CSV/JSON, gzip ok); repeatable; ':KIND' forces "
754
+ "the kind; see 'fdt gaps' and docs/supplements.md",
755
+ )
756
+ conv.add_argument(
757
+ "--supplements-dir", help="directory containing a supplements.json bundle manifest"
758
+ )
759
+ conv.add_argument(
760
+ "--on-exists",
761
+ choices=[e.value for e in OnExists],
762
+ default=OnExists.REFUSE.value,
763
+ help="policy when the output directory already exists: refuse (default), replace "
764
+ "(atomic swap) or version (new versioned subdirectory)",
765
+ )
766
+ conv.add_argument(
767
+ "--keep-temp",
768
+ action="store_true",
769
+ help="keep the staging directory on error for diagnosis",
770
+ )
771
+ conv.add_argument(
772
+ "--progress",
773
+ action="store_true",
774
+ help="print bounded-memory conversion progress (per phase) to stderr; applies to the "
775
+ "streaming engine (--stream / --output-format parquet / --partition-by)",
776
+ )
777
+ conv.add_argument(
778
+ "--exit-policy",
779
+ choices=("detailed", "pipeline"),
780
+ default="detailed",
781
+ help="detailed (default): distinct exit codes (3=strict incomplete, 4=synthetic "
782
+ "assumptions). pipeline: exit 0 for any functional completion, non-zero only on a real "
783
+ "failure — for orchestrators (Kubernetes/Airflow/Jenkins/AWS Batch) that treat any "
784
+ "non-zero code as failure",
785
+ )
786
+ conv.add_argument(
787
+ "--output-format",
788
+ choices=list(OUTPUT_FORMATS),
789
+ default="csv",
790
+ help="output format: csv (default, byte-exact) or parquet (value-exact decimal128; "
791
+ "uses the streaming engine and requires the [parquet] extra)",
792
+ )
793
+ conv.add_argument(
794
+ "--stream",
795
+ action="store_true",
796
+ help="use the bounded-memory streaming engine (implied by --output-format parquet); "
797
+ "recommended for large client files",
798
+ )
799
+ conv.add_argument(
800
+ "--partition-by",
801
+ help="Parquet only: comma-separated low-cardinality String/Date-Time Cost and Usage "
802
+ "columns to Hive-partition the dataset by (e.g. BillingCurrency,InvoiceIssuerName)",
803
+ )
804
+ conv.add_argument(
805
+ "--compression",
806
+ choices=list(COMPRESSIONS),
807
+ default="snappy",
808
+ help="Parquet compression codec (default: snappy)",
809
+ )
810
+ conv.add_argument(
811
+ "--target-file-size",
812
+ help="Parquet only: approximate max part-file size per partition (e.g. 128MB); rolls to "
813
+ "a new part file once exceeded",
814
+ )
815
+ conv.set_defaults(func=_cmd_convert)
816
+
817
+ gaps = sub.add_parser(
818
+ "gaps",
819
+ help="report exactly which facts a client must supply to produce the four "
820
+ "FOCUS 1.4 datasets factually from this source",
821
+ )
822
+ gaps.add_argument(
823
+ "--cost-and-usage", required=True,
824
+ help="FOCUS 1.2/1.3 Cost and Usage source (CSV, gzip ok, or Parquet)",
825
+ )
826
+ gaps.add_argument(
827
+ "--contract-commitment", help="optional FOCUS 1.3 Contract Commitment CSV (13 columns)"
828
+ )
829
+ gaps.add_argument(
830
+ "--source-version", help="force the FOCUS source version (1.2 or 1.3)"
831
+ )
832
+ gaps.add_argument(
833
+ "--source-dataset", help="force the source dataset instead of auto-detecting it"
834
+ )
835
+ gaps.add_argument("--format", choices=("text", "json"), default="text")
836
+ gaps.add_argument("--out", help="write the report to this path instead of stdout")
837
+ gaps.set_defaults(func=_cmd_gaps)
838
+
839
+ supp = sub.add_parser(
840
+ "supplements", help="work with supplemental client data (see docs/supplements.md)"
841
+ )
842
+ supp_sub = supp.add_subparsers(dest="supplements_command", required=True)
843
+ supp_val = supp_sub.add_parser(
844
+ "validate", help="pre-flight check supplement files against a source"
845
+ )
846
+ supp_val.add_argument(
847
+ "--cost-and-usage", required=True,
848
+ help="FOCUS 1.2/1.3 Cost and Usage source (CSV, gzip ok, or Parquet)",
849
+ )
850
+ supp_val.add_argument(
851
+ "--contract-commitment", help="optional FOCUS 1.3 Contract Commitment CSV"
852
+ )
853
+ supp_val.add_argument(
854
+ "--supplement",
855
+ action="append",
856
+ default=[],
857
+ metavar="FILE[:KIND]",
858
+ help="supplement file (CSV/JSON, gzip ok); repeatable; ':KIND' forces the kind",
859
+ )
860
+ supp_val.add_argument(
861
+ "--supplements-dir", help="directory containing a supplements.json bundle manifest"
862
+ )
863
+ supp_val.set_defaults(func=_cmd_supplements_validate)
864
+ supp_adapters = supp_sub.add_parser(
865
+ "adapters", help="list the provider-native export adapters (AWS/Azure/GCP)"
866
+ )
867
+ supp_adapters.set_defaults(func=_cmd_supplements_adapters)
868
+
869
+ val = sub.add_parser("validate", help="validate a produced CSV/Parquet file")
870
+ val.add_argument("file", help="CSV or Parquet file to validate")
871
+ val.add_argument(
872
+ "--dataset",
873
+ default="Cost and Usage",
874
+ help="FOCUS 1.4 dataset name for the built-in validator "
875
+ "(cost-and-usage, contract-commitment, billing-period, invoice-detail)",
876
+ )
877
+ val.add_argument(
878
+ "--supports",
879
+ action="append",
880
+ default=[],
881
+ choices=sorted(KNOWN_CONDITIONS),
882
+ metavar="CONDITION",
883
+ help="declare a FOCUS applicability condition the source supports (repeatable)",
884
+ )
885
+ val.add_argument(
886
+ "--official", action="store_true", help="run the official FinOps focus-validator instead"
887
+ )
888
+ val.add_argument(
889
+ "--focus-version", help="rule-model version for --official (e.g. 1.2.0.1)"
890
+ )
891
+ val.set_defaults(func=_cmd_validate)
892
+
893
+ det = sub.add_parser(
894
+ "detect", help="detect the FOCUS dataset/version of a file's header"
895
+ )
896
+ det.add_argument("file", help="CSV (gzip ok) or Parquet file")
897
+ det.add_argument("--dataset", help="force the dataset instead of auto-detecting it")
898
+ det.add_argument("--version", help="force the FOCUS version (1.2/1.3/1.4)")
899
+ det.add_argument("--format", choices=("text", "json"), default="text")
900
+ det.set_defaults(func=_cmd_detect)
901
+
902
+ vb = sub.add_parser(
903
+ "validate-bundle",
904
+ help="cross-dataset validation of a FOCUS 1.4 bundle (referential integrity, "
905
+ "reconciliation, allocation, corrections, lifecycle)",
906
+ )
907
+ vb.add_argument(
908
+ "--directory",
909
+ help="directory of FOCUS 1.4 dataset files; datasets are auto-detected from headers "
910
+ "(mutually exclusive with the per-dataset flags below)",
911
+ )
912
+ vb.add_argument("--cost-and-usage", help="Cost and Usage file (CSV, gzip ok, or Parquet)")
913
+ vb.add_argument("--contract-commitment", help="Contract Commitment file")
914
+ vb.add_argument("--billing-period", help="Billing Period file")
915
+ vb.add_argument("--invoice-detail", help="Invoice Detail file")
916
+ vb.add_argument("--report", help="write the bundle report JSON to this path")
917
+ vb.add_argument("--format", choices=("text", "json"), default="text")
918
+ vb.set_defaults(func=_cmd_validate_bundle)
919
+
920
+ ver = sub.add_parser(
921
+ "version", help="print the toolkit version and optional-extra availability"
922
+ )
923
+ ver.set_defaults(func=_cmd_version)
924
+
925
+ ui = sub.add_parser(
926
+ "ui",
927
+ help="launch the local Studio web UI (needs the [studio] extra); localhost-only by default",
928
+ )
929
+ ui.add_argument("--host", default="127.0.0.1", help="bind address (default: 127.0.0.1)")
930
+ ui.add_argument("--port", type=int, default=8765, help="port (default: 8765)")
931
+ ui.add_argument(
932
+ "--root", default=".", help="directory the UI may read source files from (default: .)"
933
+ )
934
+ ui.add_argument("--work-dir", help="scratch/output directory (default: a fresh temp dir)")
935
+ ui.add_argument(
936
+ "--allow-remote",
937
+ action="store_true",
938
+ help="permit binding a non-loopback --host (the access token is still required); "
939
+ "expose only on trusted networks",
940
+ )
941
+ ui.add_argument(
942
+ "--no-open-browser", action="store_true", help="do not open a browser on start"
943
+ )
944
+ ui.add_argument(
945
+ "--max-upload", default="200MB", help="max browser upload size (default: 200MB)"
946
+ )
947
+ ui.set_defaults(func=_cmd_ui)
948
+
949
+ clean = sub.add_parser(
950
+ "clean",
951
+ help="recover interrupted publishes and remove leftover staging/trash directories",
952
+ )
953
+ clean.add_argument(
954
+ "--out", required=True,
955
+ help="directory to clean (an output directory or the directory containing outputs); "
956
+ "run only when no conversion is publishing there",
957
+ )
958
+ clean.set_defaults(func=_cmd_clean)
959
+ return parser
960
+
961
+
962
+ def main(argv: list[str] | None = None) -> int:
963
+ args = build_parser().parse_args(argv)
964
+ return args.func(args)
965
+
966
+
967
+ if __name__ == "__main__":
968
+ raise SystemExit(main())