focus-data-toolkit 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. focus_data_toolkit/__init__.py +69 -0
  2. focus_data_toolkit/__main__.py +6 -0
  3. focus_data_toolkit/_version.py +8 -0
  4. focus_data_toolkit/cli.py +968 -0
  5. focus_data_toolkit/context/__init__.py +88 -0
  6. focus_data_toolkit/context/billing.py +54 -0
  7. focus_data_toolkit/context/provider.py +90 -0
  8. focus_data_toolkit/convert/__init__.py +708 -0
  9. focus_data_toolkit/convert/billing_period.py +65 -0
  10. focus_data_toolkit/convert/contract_applied.py +235 -0
  11. focus_data_toolkit/convert/contract_commitment.py +182 -0
  12. focus_data_toolkit/convert/cost_and_usage.py +179 -0
  13. focus_data_toolkit/convert/detect.py +39 -0
  14. focus_data_toolkit/convert/invoice_detail.py +199 -0
  15. focus_data_toolkit/convert/streaming.py +1030 -0
  16. focus_data_toolkit/errors.py +145 -0
  17. focus_data_toolkit/focus_json.py +68 -0
  18. focus_data_toolkit/generators/__init__.py +61 -0
  19. focus_data_toolkit/generators/_shim.py +43 -0
  20. focus_data_toolkit/generators/engine/__init__.py +14 -0
  21. focus_data_toolkit/generators/engine/context.py +12 -0
  22. focus_data_toolkit/generators/engine/determinism.py +117 -0
  23. focus_data_toolkit/generators/engine/json_focus.py +63 -0
  24. focus_data_toolkit/generators/engine/ladder.py +71 -0
  25. focus_data_toolkit/generators/engine/scenarios_core.py +380 -0
  26. focus_data_toolkit/generators/engine/serialize.py +151 -0
  27. focus_data_toolkit/generators/generate_aws_focus_1_2.py +19 -0
  28. focus_data_toolkit/generators/generate_aws_focus_1_3.py +20 -0
  29. focus_data_toolkit/generators/generate_azure_focus_1_2.py +17 -0
  30. focus_data_toolkit/generators/generate_azure_focus_1_3.py +17 -0
  31. focus_data_toolkit/generators/generate_gcp_focus_1_2.py +17 -0
  32. focus_data_toolkit/generators/generate_gcp_focus_1_3.py +17 -0
  33. focus_data_toolkit/generators/providers/__init__.py +29 -0
  34. focus_data_toolkit/generators/providers/aws.py +186 -0
  35. focus_data_toolkit/generators/providers/azure.py +191 -0
  36. focus_data_toolkit/generators/providers/gcp.py +194 -0
  37. focus_data_toolkit/generators/providers/profile.py +123 -0
  38. focus_data_toolkit/generators/scenarios.py +178 -0
  39. focus_data_toolkit/generators/versions/__init__.py +17 -0
  40. focus_data_toolkit/generators/versions/adapter.py +41 -0
  41. focus_data_toolkit/generators/versions/v1_2.py +111 -0
  42. focus_data_toolkit/generators/versions/v1_3.py +154 -0
  43. focus_data_toolkit/io/__init__.py +1 -0
  44. focus_data_toolkit/io/atomic_writer.py +462 -0
  45. focus_data_toolkit/io/csv_io.py +128 -0
  46. focus_data_toolkit/io/parquet_io.py +528 -0
  47. focus_data_toolkit/io/records.py +92 -0
  48. focus_data_toolkit/io/row_source.py +117 -0
  49. focus_data_toolkit/lifecycle.py +342 -0
  50. focus_data_toolkit/manifest.py +114 -0
  51. focus_data_toolkit/model/__init__.py +43 -0
  52. focus_data_toolkit/model/capabilities.py +66 -0
  53. focus_data_toolkit/model/focus_1_4_decimal_scale.json +10 -0
  54. focus_data_toolkit/model/focus_1_4_model.json +1913 -0
  55. focus_data_toolkit/model/focus_1_4_servicesubcategory.json +84 -0
  56. focus_data_toolkit/model/focus_json_keys.py +112 -0
  57. focus_data_toolkit/model/iso_4217_currencies.json +23 -0
  58. focus_data_toolkit/model/json_schema_check.py +205 -0
  59. focus_data_toolkit/model/json_schemas/allocatedmethoddetailsobjectschema.json +82 -0
  60. focus_data_toolkit/model/json_schemas/commitmentprogrameligibilitydetailsobjectschema.json +41 -0
  61. focus_data_toolkit/model/json_schemas/contractappliedobjectschema.json +104 -0
  62. focus_data_toolkit/model/json_schemas/contractcommitmentapplicabilityobjectschema.json +290 -0
  63. focus_data_toolkit/model/json_schemas/json_schemas_provenance.json +38 -0
  64. focus_data_toolkit/model/model_provenance.json +58 -0
  65. focus_data_toolkit/model/validator.py +498 -0
  66. focus_data_toolkit/modes.py +18 -0
  67. focus_data_toolkit/official_validator.py +61 -0
  68. focus_data_toolkit/progress.py +89 -0
  69. focus_data_toolkit/provenance.py +106 -0
  70. focus_data_toolkit/py.typed +1 -0
  71. focus_data_toolkit/runtime.py +243 -0
  72. focus_data_toolkit/schema/__init__.py +17 -0
  73. focus_data_toolkit/schema/detection.py +274 -0
  74. focus_data_toolkit/schema/registry.py +127 -0
  75. focus_data_toolkit/storage/__init__.py +1 -0
  76. focus_data_toolkit/storage/external_index.py +99 -0
  77. focus_data_toolkit/storage/spill.py +150 -0
  78. focus_data_toolkit/studio/__init__.py +19 -0
  79. focus_data_toolkit/studio/app.py +467 -0
  80. focus_data_toolkit/studio/config.py +42 -0
  81. focus_data_toolkit/studio/frontend/app.js +214 -0
  82. focus_data_toolkit/studio/frontend/index.html +101 -0
  83. focus_data_toolkit/studio/frontend/style.css +60 -0
  84. focus_data_toolkit/studio/jobs.py +142 -0
  85. focus_data_toolkit/studio/preview.py +32 -0
  86. focus_data_toolkit/studio/security.py +125 -0
  87. focus_data_toolkit/studio/server.py +71 -0
  88. focus_data_toolkit/supplement/__init__.py +50 -0
  89. focus_data_toolkit/supplement/adapters/__init__.py +21 -0
  90. focus_data_toolkit/supplement/adapters/adapters_provenance.json +39 -0
  91. focus_data_toolkit/supplement/adapters/aws_invoice_summary.json +24 -0
  92. focus_data_toolkit/supplement/adapters/aws_savings_plans.json +31 -0
  93. focus_data_toolkit/supplement/adapters/azure_invoice.json +25 -0
  94. focus_data_toolkit/supplement/adapters/gcp_compute_commitments.json +28 -0
  95. focus_data_toolkit/supplement/adapters/registry.py +215 -0
  96. focus_data_toolkit/supplement/apply.py +318 -0
  97. focus_data_toolkit/supplement/gaps.py +219 -0
  98. focus_data_toolkit/supplement/kinds.py +118 -0
  99. focus_data_toolkit/supplement/loader.py +409 -0
  100. focus_data_toolkit/supplement/spec.py +74 -0
  101. focus_data_toolkit/supplement/validate.py +215 -0
  102. focus_data_toolkit/validate/__init__.py +15 -0
  103. focus_data_toolkit/validate/allocation.py +333 -0
  104. focus_data_toolkit/validate/bundle.py +254 -0
  105. focus_data_toolkit/validate/codes.py +93 -0
  106. focus_data_toolkit/validate/corrections.py +245 -0
  107. focus_data_toolkit/validate/reconciliation.py +98 -0
  108. focus_data_toolkit/validate/referential.py +289 -0
  109. focus_data_toolkit-0.11.0.dist-info/METADATA +519 -0
  110. focus_data_toolkit-0.11.0.dist-info/RECORD +116 -0
  111. focus_data_toolkit-0.11.0.dist-info/WHEEL +5 -0
  112. focus_data_toolkit-0.11.0.dist-info/entry_points.txt +2 -0
  113. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSE +21 -0
  114. focus_data_toolkit-0.11.0.dist-info/licenses/LICENSES/CC-BY-4.0.txt +156 -0
  115. focus_data_toolkit-0.11.0.dist-info/licenses/NOTICE +60 -0
  116. focus_data_toolkit-0.11.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,467 @@
1
+ """The Studio FastAPI application: security middleware + a thin API over the Core SDK.
2
+
3
+ Every route delegates to the same SDK the CLI uses — ``detect_focus_schema``, ``convert_files``,
4
+ the generators, ``open_row_source`` — so nothing here reimplements FOCUS logic and the outputs
5
+ match a CLI run byte-for-byte. The app is created via :func:`create_app` so it can be driven by a
6
+ test client without starting a server.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import asyncio
12
+ import contextlib
13
+ import csv
14
+ import io
15
+ import json
16
+ import logging
17
+ from pathlib import Path
18
+ from typing import Any
19
+
20
+ from fastapi import FastAPI, File, Request, UploadFile
21
+ from fastapi.responses import (
22
+ FileResponse,
23
+ HTMLResponse,
24
+ JSONResponse,
25
+ PlainTextResponse,
26
+ Response,
27
+ StreamingResponse,
28
+ )
29
+ from fastapi.staticfiles import StaticFiles
30
+
31
+ from focus_data_toolkit import __version__
32
+ from focus_data_toolkit.studio.config import MAX_PREVIEW_LIMIT, StudioConfig
33
+ from focus_data_toolkit.studio.jobs import Job, JobManager
34
+ from focus_data_toolkit.studio.preview import sampled_page
35
+ from focus_data_toolkit.studio.security import (
36
+ PathOutsideRoot,
37
+ host_header_allowed,
38
+ origin_allowed,
39
+ resolve_within_root,
40
+ token_matches,
41
+ )
42
+
43
+ _FRONTEND = Path(__file__).resolve().parent / "frontend"
44
+ _MANIFEST_NAME = "focus_1_4_manifest.json"
45
+ _CHECKSUMS_NAME = "SHA256SUMS"
46
+
47
+ # Error detail is logged server-side; API responses carry only generic, non-revealing messages
48
+ # (no exception text / stack info flows to the client).
49
+ _LOG = logging.getLogger("focus_data_toolkit.studio")
50
+
51
+
52
+ def _parquet_available() -> bool:
53
+ try:
54
+ import pyarrow # noqa: F401
55
+
56
+ return True
57
+ except ModuleNotFoundError:
58
+ return False
59
+
60
+
61
+ def create_app(config: StudioConfig, jobs: JobManager | None = None) -> FastAPI:
62
+ app = FastAPI(title="Focus Data Toolkit Studio", docs_url=None, redoc_url=None, openapi_url=None)
63
+ jm = jobs or JobManager(
64
+ config.work_dir or config.root / ".fdt-studio-work",
65
+ max_concurrency=config.max_concurrency,
66
+ ttl_seconds=config.job_ttl_seconds,
67
+ )
68
+ app.state.jobs = jm
69
+
70
+ @app.middleware("http")
71
+ async def _guard(request: Request, call_next: Any) -> Response:
72
+ # Loopback defense-in-depth (skipped for an explicit remote bind, which is token-gated):
73
+ # validate Host (anti DNS-rebinding) and, for state-changing API calls, Origin (anti-CSRF).
74
+ if not config.allow_remote:
75
+ if not host_header_allowed(request.headers.get("host"), config.host, config.port):
76
+ return JSONResponse({"error": "host not allowed"}, status_code=400)
77
+ path = request.url.path
78
+ is_api = path.startswith("/api/")
79
+ if is_api and not config.allow_remote and request.method not in ("GET", "HEAD", "OPTIONS"):
80
+ if not origin_allowed(request.headers.get("origin"), config.host, config.port):
81
+ return JSONResponse({"error": "origin not allowed"}, status_code=403)
82
+ # Token gates every API call (header for fetch, query param for EventSource/downloads).
83
+ if is_api:
84
+ token = request.headers.get("x-fdt-token") or request.query_params.get("token")
85
+ if not token_matches(config.token, token):
86
+ return JSONResponse({"error": "missing or invalid token"}, status_code=401)
87
+ return await call_next(request)
88
+
89
+ if _FRONTEND.is_dir():
90
+ app.mount("/static", StaticFiles(directory=str(_FRONTEND)), name="static")
91
+
92
+ # --- shell ---------------------------------------------------------------------------
93
+ @app.get("/")
94
+ async def index() -> Response:
95
+ html = _FRONTEND / "index.html"
96
+ if not html.is_file():
97
+ return HTMLResponse("<h1>Studio frontend missing</h1>", status_code=500)
98
+ return FileResponse(str(html))
99
+
100
+ # --- metadata ------------------------------------------------------------------------
101
+ @app.get("/api/health")
102
+ async def health() -> Response:
103
+ return JSONResponse({"version": __version__, "parquet": _parquet_available()})
104
+
105
+ @app.get("/api/config")
106
+ async def get_config() -> Response:
107
+ from focus_data_toolkit.generators import FOCUS_VERSIONS, PROVIDERS
108
+
109
+ return JSONResponse(
110
+ {
111
+ "root": str(config.root),
112
+ "max_upload_bytes": config.max_upload_bytes,
113
+ "max_generate_rows": config.max_generate_rows,
114
+ "providers": list(PROVIDERS),
115
+ "focus_versions": list(FOCUS_VERSIONS),
116
+ "modes": ["strict", "synthetic"],
117
+ "output_formats": ["csv", "parquet"] if _parquet_available() else ["csv"],
118
+ "parquet": _parquet_available(),
119
+ }
120
+ )
121
+
122
+ # --- browse the allowlisted root -----------------------------------------------------
123
+ @app.get("/api/files")
124
+ async def list_files(subpath: str = "") -> Response:
125
+ try:
126
+ target = resolve_within_root(subpath or ".", config.root)
127
+ except PathOutsideRoot as exc:
128
+ _LOG.warning("rejected file listing outside root: %s", exc)
129
+ return JSONResponse({"error": "path is outside the allowed root"}, status_code=400)
130
+ if not target.is_dir():
131
+ return JSONResponse({"error": "not a directory"}, status_code=400)
132
+ entries = []
133
+ for child in sorted(target.iterdir(), key=lambda p: (not p.is_dir(), p.name.lower())):
134
+ if child.name.startswith("."):
135
+ continue
136
+ try:
137
+ size = child.stat().st_size if child.is_file() else None
138
+ except OSError:
139
+ size = None
140
+ entries.append({"name": child.name, "is_dir": child.is_dir(), "size": size})
141
+ rel = target.relative_to(config.root) if target != config.root else Path()
142
+ return JSONResponse({"path": str(rel), "entries": entries})
143
+
144
+ # --- detect --------------------------------------------------------------------------
145
+ @app.post("/api/detect")
146
+ async def detect(request: Request) -> Response:
147
+ body = await request.json()
148
+ try:
149
+ source = _resolve_source(config, jm, body)
150
+ except PathOutsideRoot as exc:
151
+ _LOG.warning("rejected detect source: %s", exc)
152
+ return JSONResponse({"error": "path is outside the allowed root"}, status_code=400)
153
+ if source is None:
154
+ return JSONResponse({"error": "provide a path or source"}, status_code=400)
155
+ from focus_data_toolkit.io.records import MalformedRecordError
156
+ from focus_data_toolkit.io.row_source import open_row_source
157
+ from focus_data_toolkit.schema import detect_focus_schema
158
+
159
+ try:
160
+ with contextlib.closing(open_row_source(str(source))) as reader:
161
+ header = reader.source_columns
162
+ result = detect_focus_schema(header)
163
+ except (MalformedRecordError, PathOutsideRoot, OSError) as exc:
164
+ _LOG.warning("detect failed: %s", exc)
165
+ return JSONResponse({"error": "could not read the source file"}, status_code=400)
166
+ return JSONResponse(result.as_dict())
167
+
168
+ # --- managed sources: upload + generate ----------------------------------------------
169
+ @app.post("/api/upload")
170
+ async def upload(file: UploadFile = File(...)) -> Response:
171
+ source_id, dest_dir = jm.new_source_dir()
172
+ name = Path(file.filename or "upload.csv").name
173
+ target = dest_dir / name
174
+ written = 0
175
+ with open(target, "wb") as out:
176
+ while True:
177
+ chunk = await file.read(1024 * 1024)
178
+ if not chunk:
179
+ break
180
+ written += len(chunk)
181
+ if written > config.max_upload_bytes:
182
+ out.close()
183
+ target.unlink(missing_ok=True)
184
+ return JSONResponse(
185
+ {"error": f"upload exceeds the {config.max_upload_bytes}-byte limit"},
186
+ status_code=413,
187
+ )
188
+ out.write(chunk)
189
+ return JSONResponse({"source_id": source_id, "source_name": name, "size": written})
190
+
191
+ @app.post("/api/generate")
192
+ async def generate(request: Request) -> Response:
193
+ from starlette.concurrency import run_in_threadpool
194
+
195
+ body = await request.json()
196
+ provider = str(body.get("provider", "aws"))
197
+ version = str(body.get("focus_version", "1.3"))
198
+ seed = int(body.get("seed", 1202))
199
+ rows = int(body.get("rows", 1000))
200
+ if rows < 1 or rows > config.max_generate_rows:
201
+ return JSONResponse(
202
+ {"error": f"rows must be between 1 and {config.max_generate_rows} (Studio cap; "
203
+ "use the CLI/Runner for larger synthetic sets)"},
204
+ status_code=400,
205
+ )
206
+ from focus_data_toolkit.generators import FOCUS_VERSIONS, PROVIDERS, get_generator
207
+
208
+ if provider not in PROVIDERS or version not in FOCUS_VERSIONS:
209
+ return JSONResponse({"error": "unknown provider or focus_version"}, status_code=400)
210
+
211
+ source_id, dest_dir = jm.new_source_dir()
212
+ suffix = version.replace(".", "_")
213
+
214
+ def _emit() -> dict:
215
+ module = get_generator(provider, version)
216
+ cau_name = f"focus_{suffix}_cost_and_usage_{provider}.csv"
217
+ (dest_dir / cau_name).write_bytes(module.generate_csv_bytes(rows, seed))
218
+ names = [cau_name]
219
+ if version == "1.3":
220
+ cc_name = f"focus_{suffix}_contract_commitment_{provider}.csv"
221
+ (dest_dir / cc_name).write_bytes(
222
+ module.generate_contract_commitment_csv_bytes(rows, seed)
223
+ )
224
+ names.append(cc_name)
225
+ return {"cau": cau_name, "names": names}
226
+
227
+ made = await run_in_threadpool(_emit)
228
+ return JSONResponse(
229
+ {"source_id": source_id, "source_name": made["cau"], "files": made["names"], "rows": rows}
230
+ )
231
+
232
+ # --- conversion jobs -----------------------------------------------------------------
233
+ @app.post("/api/jobs")
234
+ async def create_job(request: Request) -> Response:
235
+ from focus_data_toolkit.convert import ConversionCancelled, OnExists
236
+ from focus_data_toolkit.model.capabilities import CapabilityProfile
237
+ from focus_data_toolkit.runtime import ResourceLimitError
238
+
239
+ body = await request.json()
240
+ try:
241
+ source = _resolve_source(config, jm, body)
242
+ contract = _resolve_source(config, jm, body, prefix="contract_")
243
+ except PathOutsideRoot as exc:
244
+ _LOG.warning("rejected job source outside root: %s", exc)
245
+ return JSONResponse({"error": "source path is outside the allowed root"}, status_code=400)
246
+ if source is None:
247
+ return JSONResponse({"error": "provide a source (path or source_id)"}, status_code=400)
248
+ mode = str(body.get("mode", "strict"))
249
+ output_format = str(body.get("output_format", "csv"))
250
+ supports = [str(s) for s in body.get("supports", [])]
251
+ try:
252
+ on_exists = OnExists(str(body.get("on_exists", "refuse")))
253
+ except ValueError:
254
+ return JSONResponse({"error": "invalid on_exists"}, status_code=400)
255
+ caps = CapabilityProfile.of(*supports) if supports else None
256
+
257
+ def run(job: Job) -> None:
258
+ from focus_data_toolkit.convert import convert_files
259
+
260
+ try:
261
+ convert_files(
262
+ str(source),
263
+ str(job.out_dir),
264
+ contract_commitment=str(contract) if contract else None,
265
+ mode=mode,
266
+ output_format=output_format,
267
+ on_exists=on_exists,
268
+ capabilities=caps,
269
+ progress=lambda event: job.events.append(event.as_dict()),
270
+ cancel=job.cancel.is_set,
271
+ )
272
+ job.status = "succeeded"
273
+ except ConversionCancelled:
274
+ job.status = "cancelled"
275
+ except ResourceLimitError as exc:
276
+ job.status, job.error, job.error_code = "failed", exc.diagnostic.message, exc.diagnostic.code
277
+ except Exception as exc: # ConversionError / AtomicWriteError / MalformedRecord / ...
278
+ job.status, job.error = "failed", f"{type(exc).__name__}: {exc}"
279
+
280
+ job = jm.submit_convert(run)
281
+ return JSONResponse({"job_id": job.id}, status_code=202)
282
+
283
+ @app.get("/api/jobs/{job_id}")
284
+ async def job_status(job_id: str) -> Response:
285
+ job = jm.get(job_id)
286
+ if job is None:
287
+ return JSONResponse({"error": "unknown job"}, status_code=404)
288
+ return JSONResponse(job.summary())
289
+
290
+ @app.get("/api/jobs/{job_id}/events")
291
+ async def job_events(job_id: str) -> Response:
292
+ if jm.get(job_id) is None:
293
+ return JSONResponse({"error": "unknown job"}, status_code=404)
294
+
295
+ async def stream() -> Any:
296
+ cursor = 0
297
+ while True:
298
+ job = jm.get(job_id)
299
+ if job is None:
300
+ break
301
+ while cursor < len(job.events):
302
+ yield f"data: {json.dumps(job.events[cursor])}\n\n"
303
+ cursor += 1
304
+ if job.done:
305
+ yield f"event: done\ndata: {json.dumps(job.summary())}\n\n"
306
+ break
307
+ await asyncio.sleep(0.25)
308
+
309
+ return StreamingResponse(stream(), media_type="text/event-stream")
310
+
311
+ @app.post("/api/jobs/{job_id}/cancel")
312
+ async def job_cancel(job_id: str) -> Response:
313
+ job = jm.get(job_id)
314
+ if job is None:
315
+ return JSONResponse({"error": "unknown job"}, status_code=404)
316
+ job.cancel.set()
317
+ return JSONResponse({"ok": True}, status_code=202)
318
+
319
+ @app.get("/api/jobs/{job_id}/result")
320
+ async def job_result(job_id: str) -> Response:
321
+ job = jm.get(job_id)
322
+ if job is None:
323
+ return JSONResponse({"error": "unknown job"}, status_code=404)
324
+ files: list[dict] = []
325
+ manifest: dict | None = None
326
+ if job.out_dir.is_dir():
327
+ for child in sorted(job.out_dir.iterdir()):
328
+ files.append(
329
+ {"name": child.name, "is_dir": child.is_dir(),
330
+ "size": child.stat().st_size if child.is_file() else None}
331
+ )
332
+ manifest = _read_manifest(job.out_dir)
333
+ return JSONResponse(
334
+ {
335
+ "status": job.status,
336
+ "error": job.error,
337
+ "error_code": job.error_code,
338
+ "files": files,
339
+ "datasets": (manifest or {}).get("datasets"),
340
+ "diagnostics": (manifest or {}).get("diagnostics", []),
341
+ "assumptions_present": (manifest or {}).get("assumptions_present"),
342
+ }
343
+ )
344
+
345
+ @app.get("/api/jobs/{job_id}/preview")
346
+ async def job_preview(job_id: str, file: str, offset: int = 0, limit: int = 50) -> Response:
347
+ try:
348
+ path = jm.job_file(job_id, file)
349
+ except (KeyError, PathOutsideRoot) as exc:
350
+ _LOG.warning("rejected preview file: %s", exc)
351
+ return JSONResponse({"error": "invalid file"}, status_code=400)
352
+ if not path.exists():
353
+ return JSONResponse({"error": "no such produced file"}, status_code=404)
354
+ limit = max(1, min(limit, MAX_PREVIEW_LIMIT))
355
+ from focus_data_toolkit.io.records import MalformedRecordError
356
+
357
+ try:
358
+ page = sampled_page(path, offset=offset, limit=limit)
359
+ except (MalformedRecordError, OSError) as exc:
360
+ _LOG.warning("preview failed: %s", exc)
361
+ return JSONResponse({"error": "could not read the file"}, status_code=400)
362
+ return JSONResponse(page)
363
+
364
+ @app.get("/api/jobs/{job_id}/manifest")
365
+ async def job_manifest(job_id: str) -> Response:
366
+ job = jm.get(job_id)
367
+ if job is None or not (job.out_dir / _MANIFEST_NAME).is_file():
368
+ return JSONResponse({"error": "no manifest"}, status_code=404)
369
+ return FileResponse(str(job.out_dir / _MANIFEST_NAME), media_type="application/json")
370
+
371
+ @app.get("/api/jobs/{job_id}/checksums")
372
+ async def job_checksums(job_id: str) -> Response:
373
+ job = jm.get(job_id)
374
+ if job is None or not (job.out_dir / _CHECKSUMS_NAME).is_file():
375
+ return JSONResponse({"error": "no checksums"}, status_code=404)
376
+ return PlainTextResponse((job.out_dir / _CHECKSUMS_NAME).read_text(encoding="utf-8"))
377
+
378
+ @app.get("/api/jobs/{job_id}/diagnostics")
379
+ async def job_diagnostics(job_id: str, format: str = "json") -> Response:
380
+ job = jm.get(job_id)
381
+ manifest = _read_manifest(job.out_dir) if job else None
382
+ if manifest is None:
383
+ return JSONResponse({"error": "no manifest"}, status_code=404)
384
+ diags = manifest.get("diagnostics", [])
385
+ if format == "csv":
386
+ return PlainTextResponse(_diagnostics_csv(diags), media_type="text/csv")
387
+ return JSONResponse(diags)
388
+
389
+ @app.get("/api/jobs/{job_id}/download")
390
+ async def job_download(job_id: str, file: str) -> Response:
391
+ try:
392
+ path = jm.job_file(job_id, file)
393
+ except (KeyError, PathOutsideRoot) as exc:
394
+ _LOG.warning("rejected download file: %s", exc)
395
+ return JSONResponse({"error": "invalid file"}, status_code=400)
396
+ if not path.is_file():
397
+ return JSONResponse({"error": "not a downloadable file"}, status_code=404)
398
+ return FileResponse(str(path), filename=path.name, media_type="application/octet-stream")
399
+
400
+ @app.get("/api/jobs/{job_id}/summary.html")
401
+ async def job_summary_html(job_id: str) -> Response:
402
+ job = jm.get(job_id)
403
+ manifest = _read_manifest(job.out_dir) if job else None
404
+ if manifest is None:
405
+ return HTMLResponse("<p>No result yet.</p>", status_code=404)
406
+ return HTMLResponse(_summary_html(manifest))
407
+
408
+ return app
409
+
410
+
411
+ # --- helpers ----------------------------------------------------------------------------
412
+ def _resolve_source(
413
+ config: StudioConfig, jm: JobManager, body: dict, *, prefix: str = ""
414
+ ) -> Path | None:
415
+ """Resolve a source from a path (under root) or a managed (id, name) pair; None if absent."""
416
+ path = body.get(f"{prefix}path")
417
+ if path:
418
+ return resolve_within_root(str(path), config.root)
419
+ source_id = body.get(f"{prefix}source_id")
420
+ source_name = body.get(f"{prefix}source_name")
421
+ if source_id and source_name:
422
+ return jm.source_file(str(source_id), str(source_name))
423
+ return None
424
+
425
+
426
+ def _read_manifest(out_dir: Path) -> dict | None:
427
+ manifest = out_dir / _MANIFEST_NAME
428
+ if not manifest.is_file():
429
+ return None
430
+ try:
431
+ return json.loads(manifest.read_text(encoding="utf-8"))
432
+ except (OSError, ValueError):
433
+ return None
434
+
435
+
436
+ def _diagnostics_csv(diags: list[dict]) -> str:
437
+ columns = ["rule_id", "severity", "message", "dataset", "column", "line_number", "suggestion"]
438
+ buffer = io.StringIO()
439
+ writer = csv.DictWriter(buffer, fieldnames=columns, extrasaction="ignore")
440
+ writer.writeheader()
441
+ for diag in diags:
442
+ writer.writerow({key: diag.get(key, "") for key in columns})
443
+ return buffer.getvalue()
444
+
445
+
446
+ def _summary_html(manifest: dict) -> str:
447
+ import html
448
+
449
+ rows = []
450
+ for name, entry in (manifest.get("datasets") or {}).items():
451
+ rows.append(
452
+ f"<tr><td>{html.escape(name)}</td><td>{html.escape(str(entry.get('status')))}</td>"
453
+ f"<td>{html.escape(str(entry.get('conformance')))}</td>"
454
+ f"<td>{html.escape(str(entry.get('row_count', '')))}</td></tr>"
455
+ )
456
+ diags = manifest.get("diagnostics", [])
457
+ return (
458
+ "<!doctype html><meta charset='utf-8'><title>FOCUS conversion summary</title>"
459
+ "<h1>FOCUS conversion summary</h1>"
460
+ f"<p>source {html.escape(str(manifest.get('source_version')))} → "
461
+ f"target {html.escape(str(manifest.get('target_version')))}, "
462
+ f"mode {html.escape(str(manifest.get('mode')))}, "
463
+ f"assumptions_present={html.escape(str(manifest.get('assumptions_present')))}.</p>"
464
+ "<table border='1' cellpadding='4'><tr><th>Dataset</th><th>Status</th>"
465
+ "<th>Conformance</th><th>Rows</th></tr>" + "".join(rows) + "</table>"
466
+ f"<p>{len(diags)} diagnostic(s).</p>"
467
+ )
@@ -0,0 +1,42 @@
1
+ """Studio runtime configuration (bind address, allowlisted root, limits, per-start token)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+ from pathlib import Path
7
+
8
+ from focus_data_toolkit.studio.security import new_token
9
+
10
+ # Defaults chosen for a local single-user tool.
11
+ DEFAULT_PORT = 8765
12
+ DEFAULT_MAX_UPLOAD_BYTES = 200 * 1000 * 1000 # 200 MB — uploads are the *secondary* path
13
+ DEFAULT_MAX_GENERATE_ROWS = 100_000 # generation is eager (in-memory); cap it in the UI
14
+ DEFAULT_PREVIEW_LIMIT = 50
15
+ MAX_PREVIEW_LIMIT = 500
16
+ DEFAULT_JOB_TTL_SECONDS = 24 * 3600
17
+
18
+
19
+ @dataclass
20
+ class StudioConfig:
21
+ """Resolved configuration for a Studio server instance."""
22
+
23
+ host: str = "127.0.0.1"
24
+ port: int = DEFAULT_PORT
25
+ root: Path = field(default_factory=Path.cwd)
26
+ work_dir: Path | None = None # scratch/output root; defaults to a temp dir under the system temp
27
+ allow_remote: bool = False
28
+ max_upload_bytes: int = DEFAULT_MAX_UPLOAD_BYTES
29
+ max_generate_rows: int = DEFAULT_MAX_GENERATE_ROWS
30
+ job_ttl_seconds: int = DEFAULT_JOB_TTL_SECONDS
31
+ max_concurrency: int = 1 # one conversion at a time by default
32
+ token: str = field(default_factory=new_token)
33
+
34
+ def __post_init__(self) -> None:
35
+ self.root = Path(self.root).resolve()
36
+ if self.work_dir is not None:
37
+ self.work_dir = Path(self.work_dir).resolve()
38
+
39
+ def url(self) -> str:
40
+ """The URL a user opens — includes the token so only the launcher can drive the API."""
41
+ host = "127.0.0.1" if self.host in ("0.0.0.0", "::", "") else self.host
42
+ return f"http://{host}:{self.port}/?token={self.token}"