prepro-auto 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- app/__init__.py +0 -0
- app/api/__init__.py +0 -0
- app/api/router.py +18 -0
- app/api/routes/__init__.py +0 -0
- app/api/routes/datasets.py +487 -0
- app/api/routes/decisions.py +302 -0
- app/api/routes/execution.py +116 -0
- app/api/routes/export.py +69 -0
- app/api/routes/health.py +159 -0
- app/api/routes/ml.py +208 -0
- app/api/routes/transform.py +339 -0
- app/core/__init__.py +0 -0
- app/core/config.py +139 -0
- app/core/logging.py +47 -0
- app/core/memory.py +143 -0
- app/db/__init__.py +0 -0
- app/db/session.py +73 -0
- app/main.py +124 -0
- app/ml/__init__.py +82 -0
- app/ml/encoders.py +97 -0
- app/ml/export.py +183 -0
- app/ml/inference.py +52 -0
- app/ml/leakage.py +90 -0
- app/ml/metrics.py +84 -0
- app/ml/model_zoo.py +431 -0
- app/ml/narrate.py +84 -0
- app/ml/problem.py +101 -0
- app/ml/recipe.py +301 -0
- app/ml/trainer.py +370 -0
- app/ml/types.py +129 -0
- app/models/__init__.py +18 -0
- app/models/decision.py +78 -0
- app/models/job.py +92 -0
- app/models/ml_run.py +43 -0
- app/models/snapshot.py +81 -0
- app/notebook.py +511 -0
- app/preprocessing/__init__.py +0 -0
- app/preprocessing/abbreviations.py +200 -0
- app/preprocessing/correlation.py +152 -0
- app/preprocessing/encoding.py +128 -0
- app/preprocessing/imputation.py +445 -0
- app/preprocessing/missing_values.py +293 -0
- app/preprocessing/outliers.py +216 -0
- app/preprocessing/profiler.py +169 -0
- app/preprocessing/scaling.py +175 -0
- app/preprocessing/semantics.py +190 -0
- app/preprocessing/transforms.py +571 -0
- app/preprocessing/type_inference.py +223 -0
- app/schemas/__init__.py +0 -0
- app/schemas/decision.py +52 -0
- app/schemas/execution.py +50 -0
- app/schemas/job.py +77 -0
- app/services/__init__.py +0 -0
- app/services/dashboard_service.py +112 -0
- app/services/dataset_inspector.py +78 -0
- app/services/decision_alternatives.py +230 -0
- app/services/drift_service.py +181 -0
- app/services/encoding_service.py +108 -0
- app/services/execution_service.py +187 -0
- app/services/export_service.py +434 -0
- app/services/file_reader.py +163 -0
- app/services/llm_client.py +292 -0
- app/services/llm_transform_service.py +505 -0
- app/services/missing_value_service.py +125 -0
- app/services/ml_service.py +318 -0
- app/services/outlier_service.py +164 -0
- app/services/privacy.py +72 -0
- app/services/profiling_service.py +62 -0
- app/services/scaling_service.py +191 -0
- app/services/storage.py +129 -0
- app/services/transform_service.py +111 -0
- app/services/versioning_service.py +173 -0
- app/services/visualization_service.py +171 -0
- app/web/__init__.py +1 -0
- app/web/review.html +275 -0
- app/web/workbench.html +2281 -0
- app/web/workbench_ml.html +595 -0
- app/workers/__init__.py +0 -0
- app/workers/celery_app.py +33 -0
- app/workers/tasks.py +48 -0
- prepro_auto/__init__.py +115 -0
- prepro_auto/_ml.py +236 -0
- prepro_auto/cli.py +38 -0
- prepro_auto-1.0.0.dist-info/METADATA +600 -0
- prepro_auto-1.0.0.dist-info/RECORD +89 -0
- prepro_auto-1.0.0.dist-info/WHEEL +5 -0
- prepro_auto-1.0.0.dist-info/entry_points.txt +2 -0
- prepro_auto-1.0.0.dist-info/licenses/LICENSE +21 -0
- prepro_auto-1.0.0.dist-info/top_level.txt +2 -0
app/__init__.py
ADDED
|
File without changes
|
app/api/__init__.py
ADDED
|
File without changes
|
app/api/router.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Aggregates all route modules into a single APIRouter that main.py mounts
|
|
3
|
+
under the API_PREFIX (e.g. /api/v1).
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from fastapi import APIRouter
|
|
7
|
+
|
|
8
|
+
from app.api.routes import health, datasets, decisions, execution, export, transform
|
|
9
|
+
from app.api.routes import ml
|
|
10
|
+
|
|
11
|
+
api_router = APIRouter()
|
|
12
|
+
api_router.include_router(health.router)
|
|
13
|
+
api_router.include_router(datasets.router)
|
|
14
|
+
api_router.include_router(decisions.router)
|
|
15
|
+
api_router.include_router(execution.router)
|
|
16
|
+
api_router.include_router(export.router)
|
|
17
|
+
api_router.include_router(transform.router)
|
|
18
|
+
api_router.include_router(ml.router)
|
|
File without changes
|
|
@@ -0,0 +1,487 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Dataset routes: upload a file and inspect a job.
|
|
3
|
+
|
|
4
|
+
POST /api/v1/datasets/upload
|
|
5
|
+
Accepts a multipart file + optional domain. Validates extension and size,
|
|
6
|
+
stores the raw bytes, creates a PipelineJob row, returns the job_id.
|
|
7
|
+
|
|
8
|
+
GET /api/v1/datasets/{job_id}
|
|
9
|
+
Returns the full job detail.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
import uuid
|
|
13
|
+
|
|
14
|
+
from fastapi import APIRouter, UploadFile, File, Form, Depends, HTTPException, status
|
|
15
|
+
from sqlalchemy import select
|
|
16
|
+
from sqlalchemy.orm import Session
|
|
17
|
+
|
|
18
|
+
from app.core.config import settings
|
|
19
|
+
from app.core.logging import get_logger
|
|
20
|
+
from app.db.session import get_db
|
|
21
|
+
from app.models.job import PipelineJob, JobStatus, Domain
|
|
22
|
+
from app.schemas.job import JobCreateResponse, JobDetail, DatasetPreview, ProfileResponse
|
|
23
|
+
from app.services.storage import storage
|
|
24
|
+
from app.services.file_reader import read_dataframe, FileReadError
|
|
25
|
+
from app.services.dataset_inspector import inspect_dataframe
|
|
26
|
+
from app.services.profiling_service import run_profiling_sync
|
|
27
|
+
|
|
28
|
+
log = get_logger(__name__)
|
|
29
|
+
router = APIRouter(prefix="/datasets", tags=["datasets"])
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _extract_extension(filename: str) -> str:
|
|
33
|
+
if "." not in filename:
|
|
34
|
+
return ""
|
|
35
|
+
return filename.rsplit(".", 1)[-1].lower()
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@router.post(
|
|
39
|
+
"/upload",
|
|
40
|
+
response_model=JobCreateResponse,
|
|
41
|
+
status_code=status.HTTP_201_CREATED,
|
|
42
|
+
)
|
|
43
|
+
async def upload_dataset(
|
|
44
|
+
file: UploadFile = File(...),
|
|
45
|
+
domain: str = Form(default="general"),
|
|
46
|
+
db: Session = Depends(get_db),
|
|
47
|
+
):
|
|
48
|
+
# ── Validate extension ─────────────────────────────────
|
|
49
|
+
ext = _extract_extension(file.filename or "")
|
|
50
|
+
if ext not in settings.allowed_extensions_set:
|
|
51
|
+
# Be specific for the formats people most often try to upload by mistake.
|
|
52
|
+
# Generic "use a CSV instead" is unhelpful; tell them which tool to use.
|
|
53
|
+
format_hint = {
|
|
54
|
+
"pdf": "PDFs aren't tabular data. Extract the table first with `pdfplumber` "
|
|
55
|
+
"(text PDFs) or `camelot-py` (PDFs with clear table borders), save as CSV, "
|
|
56
|
+
"then upload that. Example: `df = pdfplumber.open('f.pdf').pages[0].extract_table()`",
|
|
57
|
+
"docx": "Word documents aren't tabular. Open in Excel, save as .xlsx or .csv, then upload.",
|
|
58
|
+
"doc": "Word documents aren't tabular. Open in Excel, save as .xlsx or .csv, then upload.",
|
|
59
|
+
"html": "HTML pages aren't directly tabular. Use `pd.read_html('page.html')[0].to_csv(...)` "
|
|
60
|
+
"first, then upload the CSV.",
|
|
61
|
+
"htm": "HTML pages aren't directly tabular. Use `pd.read_html('page.html')[0].to_csv(...)` "
|
|
62
|
+
"first, then upload the CSV.",
|
|
63
|
+
"txt": "Plain text isn't reliably tabular. If it's CSV-like, rename to .csv and retry.",
|
|
64
|
+
}.get(ext, "")
|
|
65
|
+
msg = f"Unsupported file type '.{ext}'. PrePro Auto handles tabular files: " \
|
|
66
|
+
f"{sorted(settings.allowed_extensions_set)}."
|
|
67
|
+
if format_hint:
|
|
68
|
+
msg += f" {format_hint}"
|
|
69
|
+
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=msg)
|
|
70
|
+
|
|
71
|
+
# ── Validate domain ────────────────────────────────────
|
|
72
|
+
try:
|
|
73
|
+
domain_enum = Domain(domain.lower())
|
|
74
|
+
except ValueError:
|
|
75
|
+
raise HTTPException(
|
|
76
|
+
status_code=status.HTTP_400_BAD_REQUEST,
|
|
77
|
+
detail=f"Invalid domain '{domain}'. Use 'general' or 'finance'.",
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
# ── Read bytes and check size ──────────────────────────
|
|
81
|
+
data = await file.read()
|
|
82
|
+
size = len(data)
|
|
83
|
+
if size == 0:
|
|
84
|
+
raise HTTPException(
|
|
85
|
+
status_code=status.HTTP_400_BAD_REQUEST,
|
|
86
|
+
detail="Uploaded file is empty.",
|
|
87
|
+
)
|
|
88
|
+
if size > settings.effective_max_upload_bytes:
|
|
89
|
+
eff = settings.effective_max_upload_mb
|
|
90
|
+
raise HTTPException(
|
|
91
|
+
status_code=status.HTTP_413_REQUEST_ENTITY_TOO_LARGE,
|
|
92
|
+
detail=(
|
|
93
|
+
f"File is {size / (1024 * 1024):.0f} MB, which exceeds the current "
|
|
94
|
+
f"{eff} MB limit. This limit is based on available memory because the "
|
|
95
|
+
f"dataset is processed in RAM. For larger files, free up memory, run on "
|
|
96
|
+
f"a machine with more RAM, or pre-sample the data."
|
|
97
|
+
),
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
# ── Store raw file ─────────────────────────────────────
|
|
101
|
+
job_id = str(uuid.uuid4())
|
|
102
|
+
storage_key = f"raw/{job_id}.{ext}"
|
|
103
|
+
storage.save_bytes(storage_key, data)
|
|
104
|
+
|
|
105
|
+
# ── Create job record ──────────────────────────────────
|
|
106
|
+
job = PipelineJob(
|
|
107
|
+
id=job_id,
|
|
108
|
+
original_filename=file.filename or f"upload.{ext}",
|
|
109
|
+
file_extension=ext,
|
|
110
|
+
storage_key=storage_key,
|
|
111
|
+
file_size_bytes=size,
|
|
112
|
+
status=JobStatus.UPLOADED,
|
|
113
|
+
domain=domain_enum,
|
|
114
|
+
)
|
|
115
|
+
db.add(job)
|
|
116
|
+
db.commit()
|
|
117
|
+
db.refresh(job)
|
|
118
|
+
|
|
119
|
+
# If the file is within the streamable limit but too big for the heavy
|
|
120
|
+
# in-memory steps on this machine, note it (the UI also warns the user).
|
|
121
|
+
from app.core.memory import fits_heavy_ops
|
|
122
|
+
heavy = fits_heavy_ops(size)
|
|
123
|
+
if not heavy["ok"]:
|
|
124
|
+
log.warning(
|
|
125
|
+
f"Job {job_id[:8]}: {heavy['file_mb']} MB exceeds heavy-op limit "
|
|
126
|
+
f"({heavy['heavy_max_mb']} MB) — KNN/MICE/IsolationForest may not fit in RAM."
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
log.info(
|
|
130
|
+
f"Upload stored: job={job_id[:8]} file={file.filename!r} "
|
|
131
|
+
f"size={size} domain={domain_enum.value}"
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
# NOTE: In Week 3 we will trigger the profiler Celery task here:
|
|
135
|
+
# from app.workers.tasks import profile_dataset
|
|
136
|
+
# profile_dataset.delay(job_id)
|
|
137
|
+
|
|
138
|
+
return JobCreateResponse(
|
|
139
|
+
job_id=job.id,
|
|
140
|
+
original_filename=job.original_filename,
|
|
141
|
+
file_extension=job.file_extension,
|
|
142
|
+
file_size_bytes=job.file_size_bytes,
|
|
143
|
+
status=job.status,
|
|
144
|
+
domain=job.domain,
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
@router.get("/{job_id}", response_model=JobDetail)
|
|
149
|
+
def get_dataset(job_id: str, db: Session = Depends(get_db)):
|
|
150
|
+
job = db.get(PipelineJob, job_id)
|
|
151
|
+
if job is None:
|
|
152
|
+
raise HTTPException(
|
|
153
|
+
status_code=status.HTTP_404_NOT_FOUND,
|
|
154
|
+
detail=f"Job {job_id} not found.",
|
|
155
|
+
)
|
|
156
|
+
return job
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
@router.get("/{job_id}/preview", response_model=DatasetPreview)
|
|
160
|
+
def preview_dataset(
|
|
161
|
+
job_id: str,
|
|
162
|
+
rows: int = 10,
|
|
163
|
+
db: Session = Depends(get_db),
|
|
164
|
+
):
|
|
165
|
+
"""
|
|
166
|
+
Load the stored raw file, parse it into a DataFrame, and return its shape
|
|
167
|
+
plus a preview of the first `rows` rows.
|
|
168
|
+
|
|
169
|
+
As a side effect, persists row_count and column_count onto the job record
|
|
170
|
+
(this is the first time we actually learn the dataset's shape).
|
|
171
|
+
"""
|
|
172
|
+
job = db.get(PipelineJob, job_id)
|
|
173
|
+
if job is None:
|
|
174
|
+
raise HTTPException(
|
|
175
|
+
status_code=status.HTTP_404_NOT_FOUND,
|
|
176
|
+
detail=f"Job {job_id} not found.",
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
# Clamp preview rows to a sane range.
|
|
180
|
+
rows = max(1, min(rows, 100))
|
|
181
|
+
|
|
182
|
+
# Load the raw bytes from storage.
|
|
183
|
+
try:
|
|
184
|
+
data = storage.load_bytes(job.storage_key)
|
|
185
|
+
except Exception as e: # noqa: BLE001
|
|
186
|
+
raise HTTPException(
|
|
187
|
+
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
|
|
188
|
+
detail=f"Could not load stored file: {e}",
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
# Parse into a DataFrame.
|
|
192
|
+
try:
|
|
193
|
+
df = read_dataframe(data, job.file_extension)
|
|
194
|
+
except FileReadError as e:
|
|
195
|
+
# Record the failure on the job so the developer can see why.
|
|
196
|
+
job.status = JobStatus.FAILED
|
|
197
|
+
job.error_message = str(e)
|
|
198
|
+
db.commit()
|
|
199
|
+
raise HTTPException(
|
|
200
|
+
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
|
|
201
|
+
detail=str(e),
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
summary = inspect_dataframe(df, preview_rows=rows)
|
|
205
|
+
|
|
206
|
+
# Persist the shape onto the job record.
|
|
207
|
+
job.row_count = summary["row_count"]
|
|
208
|
+
job.column_count = summary["column_count"]
|
|
209
|
+
db.commit()
|
|
210
|
+
|
|
211
|
+
log.info(
|
|
212
|
+
f"Preview generated: job={job_id[:8]} "
|
|
213
|
+
f"shape=({summary['row_count']}, {summary['column_count']})"
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
return DatasetPreview(job_id=job_id, **summary)
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
@router.post("/{job_id}/profile", response_model=ProfileResponse)
|
|
220
|
+
def profile_dataset_endpoint(
|
|
221
|
+
job_id: str,
|
|
222
|
+
background: bool = False,
|
|
223
|
+
db: Session = Depends(get_db),
|
|
224
|
+
):
|
|
225
|
+
"""
|
|
226
|
+
Run the statistical profiler on a dataset.
|
|
227
|
+
|
|
228
|
+
By default this runs SYNCHRONOUSLY (background=false) — no Redis/Celery
|
|
229
|
+
needed, perfect for local development. The response includes the full
|
|
230
|
+
profile.
|
|
231
|
+
|
|
232
|
+
Set background=true to dispatch the work to a Celery worker instead (needs
|
|
233
|
+
Redis + a running worker). In that case the response returns immediately
|
|
234
|
+
with status=profiling and you poll GET .../profile for the result.
|
|
235
|
+
"""
|
|
236
|
+
job = db.get(PipelineJob, job_id)
|
|
237
|
+
if job is None:
|
|
238
|
+
raise HTTPException(
|
|
239
|
+
status_code=status.HTTP_404_NOT_FOUND,
|
|
240
|
+
detail=f"Job {job_id} not found.",
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
if background:
|
|
244
|
+
# Dispatch to Celery. Import here so local sync mode never needs Redis.
|
|
245
|
+
from app.workers.tasks import profile_dataset as profile_task
|
|
246
|
+
profile_task.delay(job_id)
|
|
247
|
+
job.status = JobStatus.PROFILING
|
|
248
|
+
db.commit()
|
|
249
|
+
return ProfileResponse(
|
|
250
|
+
job_id=job_id,
|
|
251
|
+
status=JobStatus.PROFILING,
|
|
252
|
+
quality_score=None,
|
|
253
|
+
row_count=job.row_count,
|
|
254
|
+
column_count=job.column_count,
|
|
255
|
+
)
|
|
256
|
+
|
|
257
|
+
# Synchronous path (default).
|
|
258
|
+
try:
|
|
259
|
+
job = run_profiling_sync(job_id, db)
|
|
260
|
+
except FileReadError as e:
|
|
261
|
+
raise HTTPException(
|
|
262
|
+
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
|
|
263
|
+
detail=str(e),
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
profile = job.profile_json or {}
|
|
267
|
+
return ProfileResponse(
|
|
268
|
+
job_id=job_id,
|
|
269
|
+
status=job.status,
|
|
270
|
+
quality_score=job.quality_score,
|
|
271
|
+
row_count=job.row_count,
|
|
272
|
+
column_count=job.column_count,
|
|
273
|
+
type_breakdown=profile.get("type_breakdown"),
|
|
274
|
+
columns=profile.get("columns"),
|
|
275
|
+
)
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
@router.get("/{job_id}/profile", response_model=ProfileResponse)
|
|
279
|
+
def get_profile(job_id: str, db: Session = Depends(get_db)):
|
|
280
|
+
"""Retrieve a previously-computed profile."""
|
|
281
|
+
job = db.get(PipelineJob, job_id)
|
|
282
|
+
if job is None:
|
|
283
|
+
raise HTTPException(
|
|
284
|
+
status_code=status.HTTP_404_NOT_FOUND,
|
|
285
|
+
detail=f"Job {job_id} not found.",
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
profile = job.profile_json or {}
|
|
289
|
+
return ProfileResponse(
|
|
290
|
+
job_id=job_id,
|
|
291
|
+
status=job.status,
|
|
292
|
+
quality_score=job.quality_score,
|
|
293
|
+
row_count=job.row_count,
|
|
294
|
+
column_count=job.column_count,
|
|
295
|
+
type_breakdown=profile.get("type_breakdown"),
|
|
296
|
+
columns=profile.get("columns"),
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
@router.get("/{job_id}/comparison")
|
|
301
|
+
def dataset_comparison(job_id: str, db: Session = Depends(get_db)):
|
|
302
|
+
"""
|
|
303
|
+
Before/after comparison for the workbench.
|
|
304
|
+
|
|
305
|
+
Returns the original (raw) data's quality and per-column types alongside the
|
|
306
|
+
cleaned data's quality and types, plus a small sample of each. Lets a user
|
|
307
|
+
see exactly what preprocessing changed: type conversions, dropped/added
|
|
308
|
+
columns, and the overall quality-score movement.
|
|
309
|
+
"""
|
|
310
|
+
import io as _io
|
|
311
|
+
import pandas as _pd
|
|
312
|
+
from app.preprocessing.profiler import profile_dataframe
|
|
313
|
+
from app.models.snapshot import PipelineSnapshot
|
|
314
|
+
|
|
315
|
+
job = db.get(PipelineJob, job_id)
|
|
316
|
+
if job is None:
|
|
317
|
+
raise HTTPException(status_code=404, detail=f"Job {job_id} not found.")
|
|
318
|
+
|
|
319
|
+
# ── BEFORE: the raw uploaded file ──────────────────────
|
|
320
|
+
raw_df = read_dataframe(storage.load_bytes(job.storage_key), job.file_extension)
|
|
321
|
+
before = profile_dataframe(raw_df, use_llm=False)
|
|
322
|
+
|
|
323
|
+
# ── AFTER: the ACTIVE snapshot version (respects undo/redo) ──
|
|
324
|
+
from app.services.versioning_service import load_active_dataframe
|
|
325
|
+
active = job.active_version or 0
|
|
326
|
+
if active > 0:
|
|
327
|
+
clean_df = load_active_dataframe(job, db)
|
|
328
|
+
after = profile_dataframe(clean_df, use_llm=False)
|
|
329
|
+
cleaned = True
|
|
330
|
+
version = active
|
|
331
|
+
else:
|
|
332
|
+
clean_df = raw_df
|
|
333
|
+
after = before
|
|
334
|
+
cleaned = False
|
|
335
|
+
version = 0
|
|
336
|
+
|
|
337
|
+
def _types(profile):
|
|
338
|
+
return {
|
|
339
|
+
c["name"]: (c.get("refined_type") or c.get("semantic_type", "unknown"))
|
|
340
|
+
for c in profile["columns"]
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
def _sample(df, n=5):
|
|
344
|
+
head = df.head(n)
|
|
345
|
+
return {
|
|
346
|
+
"columns": [str(c) for c in df.columns],
|
|
347
|
+
"rows": [
|
|
348
|
+
{str(c): (None if _pd.isna(r[c]) else _json_cell(r[c])) for c in df.columns}
|
|
349
|
+
for _, r in head.iterrows()
|
|
350
|
+
],
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
return {
|
|
354
|
+
"job_id": job_id,
|
|
355
|
+
"cleaned": cleaned,
|
|
356
|
+
"snapshot_version": version,
|
|
357
|
+
"before": {
|
|
358
|
+
"quality_score": before["quality_score"],
|
|
359
|
+
"row_count": before["row_count"],
|
|
360
|
+
"column_count": before["column_count"],
|
|
361
|
+
"types": _types(before),
|
|
362
|
+
"sample": _sample(raw_df),
|
|
363
|
+
},
|
|
364
|
+
"after": {
|
|
365
|
+
"quality_score": after["quality_score"],
|
|
366
|
+
"row_count": after["row_count"],
|
|
367
|
+
"column_count": after["column_count"],
|
|
368
|
+
"types": _types(after),
|
|
369
|
+
"sample": _sample(clean_df),
|
|
370
|
+
},
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
def _json_cell(v):
|
|
375
|
+
"""Make a single cell JSON-safe (numpy scalars -> python)."""
|
|
376
|
+
try:
|
|
377
|
+
import numpy as _np
|
|
378
|
+
if isinstance(v, (_np.integer,)):
|
|
379
|
+
return int(v)
|
|
380
|
+
if isinstance(v, (_np.floating,)):
|
|
381
|
+
f = float(v)
|
|
382
|
+
return None if f != f else f # NaN guard
|
|
383
|
+
if isinstance(v, (_np.bool_,)):
|
|
384
|
+
return bool(v)
|
|
385
|
+
except Exception:
|
|
386
|
+
pass
|
|
387
|
+
return v
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
@router.get("/{job_id}/view")
|
|
391
|
+
def view_dataset(
|
|
392
|
+
job_id: str,
|
|
393
|
+
rows: int = 25,
|
|
394
|
+
offset: int = 0,
|
|
395
|
+
db: Session = Depends(get_db),
|
|
396
|
+
):
|
|
397
|
+
"""
|
|
398
|
+
Live view of the dataset at the job's ACTIVE version (respects undo/redo).
|
|
399
|
+
Returns columns (with dtypes), a row sample, shape, and the version label so
|
|
400
|
+
the UI's View section always reflects the current state after any change.
|
|
401
|
+
|
|
402
|
+
Supports windowed paging via `offset`/`rows` so the UI can lazily "Load more"
|
|
403
|
+
rows in batches instead of rendering the whole frame at once. `total_rows`
|
|
404
|
+
tells the UI when it has reached the end.
|
|
405
|
+
"""
|
|
406
|
+
import pandas as _pd
|
|
407
|
+
from app.services.versioning_service import load_active_dataframe, history as _history
|
|
408
|
+
|
|
409
|
+
job = db.get(PipelineJob, job_id)
|
|
410
|
+
if job is None:
|
|
411
|
+
raise HTTPException(status_code=404, detail=f"Job {job_id} not found.")
|
|
412
|
+
|
|
413
|
+
rows = max(1, min(rows, 200))
|
|
414
|
+
offset = max(0, offset)
|
|
415
|
+
df = load_active_dataframe(job, db)
|
|
416
|
+
total_rows = int(df.shape[0])
|
|
417
|
+
window = df.iloc[offset:offset + rows]
|
|
418
|
+
|
|
419
|
+
cols = [{"name": str(c), "dtype": str(df[c].dtype)} for c in df.columns]
|
|
420
|
+
sample = [
|
|
421
|
+
{str(c): (None if _pd.isna(r[c]) else _json_cell(r[c])) for c in df.columns}
|
|
422
|
+
for _, r in window.iterrows()
|
|
423
|
+
]
|
|
424
|
+
h = _history(job, db)
|
|
425
|
+
|
|
426
|
+
# Live quality score of the CURRENT version + the raw baseline, so the UI can
|
|
427
|
+
# show the score and turn green when the data has improved over raw.
|
|
428
|
+
from app.preprocessing.profiler import profile_dataframe
|
|
429
|
+
try:
|
|
430
|
+
current_quality = profile_dataframe(df, use_llm=False)["quality_score"]
|
|
431
|
+
except Exception: # noqa: BLE001 - quality is best-effort, never blocks the view
|
|
432
|
+
current_quality = None
|
|
433
|
+
baseline_quality = job.quality_score # set at profile time on the raw data
|
|
434
|
+
|
|
435
|
+
returned = offset + len(sample)
|
|
436
|
+
return {
|
|
437
|
+
"job_id": job_id,
|
|
438
|
+
"active_version": h["active_version"],
|
|
439
|
+
"version_label": next(
|
|
440
|
+
(v["label"] for v in h["versions"] if v["version"] == h["active_version"]),
|
|
441
|
+
"Raw data",
|
|
442
|
+
),
|
|
443
|
+
"row_count": total_rows,
|
|
444
|
+
"total_rows": total_rows,
|
|
445
|
+
"offset": offset,
|
|
446
|
+
"returned": returned,
|
|
447
|
+
"has_more": returned < total_rows,
|
|
448
|
+
"column_count": int(df.shape[1]),
|
|
449
|
+
"quality_score": current_quality,
|
|
450
|
+
"baseline_quality": baseline_quality,
|
|
451
|
+
"improved": (
|
|
452
|
+
current_quality is not None and baseline_quality is not None
|
|
453
|
+
and current_quality > baseline_quality
|
|
454
|
+
),
|
|
455
|
+
"columns": cols,
|
|
456
|
+
"sample": sample,
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
|
|
460
|
+
@router.get("/{job_id}/history")
|
|
461
|
+
def get_history(job_id: str, db: Session = Depends(get_db)):
|
|
462
|
+
"""Snapshot history + undo/redo availability for the job."""
|
|
463
|
+
from app.services.versioning_service import history as _history
|
|
464
|
+
job = db.get(PipelineJob, job_id)
|
|
465
|
+
if job is None:
|
|
466
|
+
raise HTTPException(status_code=404, detail=f"Job {job_id} not found.")
|
|
467
|
+
return _history(job, db)
|
|
468
|
+
|
|
469
|
+
|
|
470
|
+
@router.post("/{job_id}/undo")
|
|
471
|
+
def undo_step(job_id: str, db: Session = Depends(get_db)):
|
|
472
|
+
"""Move the active version back one step (undo)."""
|
|
473
|
+
from app.services.versioning_service import undo as _undo
|
|
474
|
+
job = db.get(PipelineJob, job_id)
|
|
475
|
+
if job is None:
|
|
476
|
+
raise HTTPException(status_code=404, detail=f"Job {job_id} not found.")
|
|
477
|
+
return _undo(job, db)
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
@router.post("/{job_id}/redo")
|
|
481
|
+
def redo_step(job_id: str, db: Session = Depends(get_db)):
|
|
482
|
+
"""Move the active version forward one step (redo)."""
|
|
483
|
+
from app.services.versioning_service import redo as _redo
|
|
484
|
+
job = db.get(PipelineJob, job_id)
|
|
485
|
+
if job is None:
|
|
486
|
+
raise HTTPException(status_code=404, detail=f"Job {job_id} not found.")
|
|
487
|
+
return _redo(job, db)
|