metacls 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
metacls/__init__.py ADDED
@@ -0,0 +1,8 @@
1
+ """MetaCLS — bulk metadata scrubbing for PDF, Office and image files.
2
+
3
+ The remediation companion to MetaScout: MetaScout finds documents leaking
4
+ metadata, MetaCLS strips it and produces a before/after proof report.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ __version__ = "0.3.0"
metacls/_theme.py ADDED
@@ -0,0 +1,130 @@
1
+ """MetaCLS's own visual identity — one source of truth for the CSS used
2
+ by both the local web UI (`web.py`) and the HTML report
3
+ (`report/templates/`).
4
+
5
+ Deliberately *not* MetaScout's look. MetaScout is blue-on-black, dark
6
+ only. MetaCLS is emerald/slate with a "swept clean" motif and a proper
7
+ light theme (follows `prefers-color-scheme`). Keeping the palette here,
8
+ rather than copy-pasted into two files, is what keeps the two surfaces
9
+ looking like one product.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ # Design tokens. Dark is the primary look; the light block is a full
14
+ # re-theme, not an afterthought.
15
+ TOKENS_CSS = """
16
+ :root {
17
+ --bg: #0c1110;
18
+ --panel: #141b1a;
19
+ --panel-2: #1a2321;
20
+ --border: #263230;
21
+ --text: #e9efed;
22
+ --muted: #93a5a0;
23
+ --accent: #2dd4a7; /* emerald-teal — MetaCLS's signature */
24
+ --accent-ink:#04140f;
25
+ --link: #5ad1c4;
26
+ --good: #2dd4a7;
27
+ --warn: #f6c454;
28
+ --bad: #f2778d;
29
+ --sweep: linear-gradient(100deg, rgba(45,212,167,.16), rgba(45,212,167,0) 60%);
30
+ }
31
+ @media (prefers-color-scheme: light) {
32
+ :root {
33
+ --bg: #f4f7f6;
34
+ --panel: #ffffff;
35
+ --panel-2: #eef3f1;
36
+ --border: #dde5e3;
37
+ --text: #13201d;
38
+ --muted: #5c6b67;
39
+ --accent: #0f9e78;
40
+ --accent-ink:#ffffff;
41
+ --link: #0c7c6a;
42
+ --good: #0f9e78;
43
+ --warn: #b57d12;
44
+ --bad: #c2405a;
45
+ --sweep: linear-gradient(100deg, rgba(15,158,120,.10), rgba(15,158,120,0) 60%);
46
+ }
47
+ }
48
+ """
49
+
50
+ # Shared element + component styles. The `msc-` prefix keeps these from
51
+ # colliding with anything in a report card's own escaped content.
52
+ BASE_CSS = """
53
+ * { box-sizing: border-box; }
54
+ body { margin:0; background:var(--bg); color:var(--text);
55
+ font-family:"Inter",-apple-system,"Segoe UI",Roboto,sans-serif; font-size:14px;
56
+ line-height:1.5; -webkit-font-smoothing:antialiased; }
57
+ a { color:var(--link); }
58
+ code, .mono { font-family:ui-monospace,"SF Mono",SFMono-Regular,Menlo,monospace; }
59
+
60
+ .msc-header { padding:26px 40px 20px; border-bottom:1px solid var(--border);
61
+ background:var(--sweep); background-repeat:no-repeat; }
62
+ .msc-mark { font-weight:800; font-size:21px; letter-spacing:-.01em; }
63
+ .msc-mark b { color:var(--accent); font-weight:800; }
64
+ .msc-mark::before { content:""; display:inline-block; width:22px; height:3px; border-radius:2px;
65
+ background:var(--accent); margin:0 10px 5px 0; vertical-align:middle; }
66
+ .msc-tag { color:var(--muted); font-size:12.5px; margin-top:5px; }
67
+ .msc-nav { margin-top:12px; display:flex; gap:18px; }
68
+ .msc-nav a { color:var(--muted); text-decoration:none; font-size:12.5px; font-weight:600;
69
+ padding-bottom:3px; border-bottom:2px solid transparent; }
70
+ .msc-nav a.on, .msc-nav a:hover { color:var(--accent); border-bottom-color:var(--accent); }
71
+ .msc-lang a { color:var(--muted); text-decoration:none; border:1px solid var(--border);
72
+ border-radius:999px; padding:3px 10px; font-size:11.5px; font-weight:700; margin-left:6px; }
73
+ .msc-lang a.on { background:var(--accent); color:var(--accent-ink); border-color:var(--accent); }
74
+
75
+ .msc-main { padding:26px 40px 72px; max-width:880px; margin:0 auto; }
76
+
77
+ .msc-hero { border:1px solid var(--border); border-radius:14px; overflow:hidden;
78
+ background:var(--panel); margin-bottom:22px; }
79
+ .msc-hero-band { padding:18px 22px; background:var(--sweep), var(--panel-2);
80
+ border-bottom:1px solid var(--border); display:flex; align-items:baseline; gap:12px; flex-wrap:wrap; }
81
+ .msc-hero-state { font-size:19px; font-weight:800; letter-spacing:.02em; }
82
+ .msc-hero-state.ok { color:var(--good); }
83
+ .msc-hero-state.warn { color:var(--warn); }
84
+ .msc-hero-state.bad { color:var(--bad); }
85
+ .msc-hero-sub { color:var(--muted); font-size:12.5px; }
86
+ .msc-stats { display:flex; flex-wrap:wrap; gap:10px; padding:16px 22px; }
87
+ .msc-stat { flex:1; min-width:120px; background:var(--panel-2); border:1px solid var(--border);
88
+ border-radius:10px; padding:12px 14px; }
89
+ .msc-stat .n { font-size:20px; font-weight:800; }
90
+ .msc-stat .l { color:var(--muted); font-size:11.5px; margin-top:2px; letter-spacing:.02em; }
91
+ .msc-stat.warn .n { color:var(--warn); }
92
+ .msc-stat.bad .n { color:var(--bad); }
93
+
94
+ .msc-card { background:var(--panel); border:1px solid var(--border); border-left:3px solid var(--border);
95
+ border-radius:12px; padding:18px 20px; margin-bottom:14px; }
96
+ .msc-card.is-cleaned { border-left-color:var(--good); }
97
+ .msc-card.is-skipped { border-left-color:var(--warn); }
98
+ .msc-card.is-error { border-left-color:var(--bad); }
99
+ .msc-card.is-unsupported { border-left-color:var(--muted); }
100
+ .msc-file { font-family:ui-monospace,"SF Mono",SFMono-Regular,Menlo,monospace; font-size:13px;
101
+ word-break:break-all; }
102
+ .msc-meta { color:var(--muted); font-size:12px; }
103
+
104
+ .msc-pill { display:inline-block; font-size:10.5px; font-weight:800; padding:2px 9px;
105
+ border-radius:999px; text-transform:uppercase; letter-spacing:.05em; }
106
+ .msc-pill.cleaned{ background:color-mix(in srgb,var(--good) 18%,transparent); color:var(--good); }
107
+ .msc-pill.skipped{ background:color-mix(in srgb,var(--warn) 18%,transparent); color:var(--warn); }
108
+ .msc-pill.error{ background:color-mix(in srgb,var(--bad) 18%,transparent); color:var(--bad); }
109
+ .msc-pill.unsupported{ background:color-mix(in srgb,var(--muted) 18%,transparent); color:var(--muted); }
110
+
111
+ table.msc-diff { width:100%; border-collapse:collapse; margin-top:10px; font-size:12.5px; }
112
+ table.msc-diff th, table.msc-diff td { text-align:left; padding:6px 10px;
113
+ border-bottom:1px solid var(--border); vertical-align:top; }
114
+ table.msc-diff th { color:var(--muted); font-weight:600; }
115
+ table.msc-diff td.was { font-family:ui-monospace,SFMono-Regular,Menlo,monospace; color:var(--muted);
116
+ text-decoration:line-through; text-decoration-color:color-mix(in srgb,var(--bad) 60%,transparent);
117
+ word-break:break-word; }
118
+ .msc-kept { color:var(--good); font-size:12px; margin-top:10px; }
119
+ .msc-residual { color:var(--bad); font-size:12px; margin-top:10px; font-weight:600; }
120
+ .msc-empty { color:var(--muted); font-size:12.5px; margin-top:8px; }
121
+
122
+ .msc-btn { background:var(--accent); color:var(--accent-ink); border:0; border-radius:9px;
123
+ padding:12px 22px; font-weight:800; font-size:14px; cursor:pointer; }
124
+ .msc-btn:disabled { opacity:.6; cursor:wait; }
125
+ .msc-actions a { margin-right:16px; font-weight:600; font-size:13px; }
126
+ """
127
+
128
+
129
+ def full_css() -> str:
130
+ return TOKENS_CSS + BASE_CSS
@@ -0,0 +1,5 @@
1
+ from __future__ import annotations
2
+
3
+ from .app import create_app
4
+
5
+ __all__ = ["create_app"]
metacls/api/app.py ADDED
@@ -0,0 +1,276 @@
1
+ from __future__ import annotations
2
+
3
+ import io
4
+ import json
5
+ import os
6
+ import secrets
7
+ import shutil
8
+ import tempfile
9
+ import time
10
+ import zipfile
11
+ from contextlib import asynccontextmanager
12
+ from datetime import datetime, timezone
13
+
14
+ from fastapi import Depends, FastAPI, File, Form, HTTPException, Request, UploadFile
15
+ from fastapi.responses import HTMLResponse, Response, StreamingResponse
16
+ from werkzeug.utils import secure_filename
17
+
18
+ from .. import __version__
19
+ from ..cleaner import clean_file_list
20
+ from ..config import CONTAINER_EXTENSIONS, MEDIA_EXTENSIONS, CleanConfig
21
+ from ..engines import format_support
22
+ from ..models import BatchReport
23
+ from .jobs import Job, JobQueueFull, JobStore
24
+ from .schemas import (
25
+ FormatsResponse,
26
+ HealthResponse,
27
+ JobCreated,
28
+ JobLogResponse,
29
+ JobStatusResponse,
30
+ JobSummary,
31
+ )
32
+
33
+ _CHUNK = 1024 * 1024
34
+
35
+
36
+ def _rm(path: str) -> None:
37
+ try:
38
+ os.remove(path)
39
+ except OSError:
40
+ pass
41
+
42
+
43
+ def _summary(job: Job) -> JobSummary | None:
44
+ return JobSummary(**job.summary) if job.summary else None
45
+
46
+
47
+ def _links(request: Request, job_id: str) -> dict[str, str]:
48
+ return {
49
+ "self": str(request.url_for("get_job", job_id=job_id)),
50
+ "log": str(request.url_for("get_job_log", job_id=job_id)),
51
+ "report_json": str(request.url_for("get_job_report_json", job_id=job_id)),
52
+ "report_html": str(request.url_for("get_job_report_html", job_id=job_id)),
53
+ "download": str(request.url_for("get_job_download", job_id=job_id)),
54
+ }
55
+
56
+
57
+ def _status(job: Job, request: Request) -> JobStatusResponse:
58
+ return JobStatusResponse(
59
+ job_id=job.job_id, status=job.status, run_id=job.run_id, file_count=job.file_count,
60
+ created_at=job.created_at.isoformat(),
61
+ started_at=job.started_at.isoformat() if job.started_at else None,
62
+ finished_at=job.finished_at.isoformat() if job.finished_at else None,
63
+ error=job.error,
64
+ summary=_summary(job),
65
+ links=_links(request, job.job_id),
66
+ )
67
+
68
+
69
+ def _require(store: JobStore, job_id: str) -> Job:
70
+ job = store.get(job_id)
71
+ if job is None:
72
+ raise HTTPException(404, f"No job {job_id!r}. Jobs are in-memory and don't survive a restart.")
73
+ return job
74
+
75
+
76
+ def _require_done(store: JobStore, job_id: str) -> Job:
77
+ job = _require(store, job_id)
78
+ if job.status != "done":
79
+ raise HTTPException(409, {"job_id": job.job_id, "status": job.status, "error": job.error,
80
+ "message": "Not ready — poll GET /v1/clean/{job_id} until status is 'done'."})
81
+ return job
82
+
83
+
84
+ def create_app(*, output_dir: str = "./metacls_cleaned", max_workers: int = 2,
85
+ max_pending: int = 50, api_key: str | None = None,
86
+ max_upload_mb: int = 200, max_files: int = 50,
87
+ run_ttl_days: int = 0, log_json: bool = False) -> FastAPI:
88
+ """MetaCLS REST API — POST files, poll the job, pull the cleaned
89
+ files + report as a zip. Every job runs in a bounded background thread
90
+ pool so POST returns immediately; max_pending caps queued+running jobs.
91
+
92
+ `api_key`, when set, is required on every `/v1/*` route except the
93
+ open capability routes `/v1/health` and `/v1/formats` — as
94
+ `X-API-Key: <key>` or `Authorization: Bearer <key>`. `max_upload_mb` /
95
+ `max_files` bound a single request.
96
+ """
97
+ os.makedirs(output_dir, exist_ok=True)
98
+ store = JobStore(output_dir=output_dir, max_workers=max_workers, max_pending=max_pending,
99
+ run_ttl_days=run_ttl_days)
100
+ max_upload_bytes = max_upload_mb * 1024 * 1024
101
+
102
+ def require_key(request: Request) -> None:
103
+ if not api_key:
104
+ return
105
+ supplied = request.headers.get("x-api-key")
106
+ if not supplied:
107
+ auth = request.headers.get("authorization", "")
108
+ if auth.lower().startswith("bearer "):
109
+ supplied = auth[7:]
110
+ if not (supplied and secrets.compare_digest(supplied, api_key)):
111
+ raise HTTPException(401, "missing or invalid API key")
112
+
113
+ guard = [Depends(require_key)]
114
+
115
+ @asynccontextmanager
116
+ async def lifespan(_app: FastAPI):
117
+ yield
118
+ store.shutdown()
119
+
120
+ app = FastAPI(
121
+ title="MetaCLS API",
122
+ version=__version__,
123
+ description="Job-based bulk metadata scrubbing for PDF/Office/image files. "
124
+ "Set an API key before exposing this — see the README.",
125
+ lifespan=lifespan,
126
+ )
127
+
128
+ if log_json:
129
+ @app.middleware("http")
130
+ async def _access_log(request: Request, call_next):
131
+ t0 = time.monotonic()
132
+ response = await call_next(request)
133
+ print(json.dumps({
134
+ "ts": datetime.now(timezone.utc).isoformat(),
135
+ "method": request.method, "path": request.url.path,
136
+ "status": response.status_code,
137
+ "ms": round((time.monotonic() - t0) * 1000, 1),
138
+ "client": request.client.host if request.client else None,
139
+ }), flush=True)
140
+ return response
141
+
142
+ @app.get("/v1/health", response_model=HealthResponse, tags=["meta"])
143
+ def health() -> HealthResponse:
144
+ active = sum(1 for j in store.list() if j.status in ("queued", "running"))
145
+ return HealthResponse(status="ok", version=__version__, active_jobs=active)
146
+
147
+ @app.get("/v1/formats", response_model=FormatsResponse, tags=["meta"])
148
+ def formats() -> FormatsResponse:
149
+ """What this server can scrub: every supported extension, the
150
+ extensions each engine handles, and which optional pieces
151
+ (exiftool, LibreOffice, mutagen, py7zr, …) are installed here."""
152
+ return FormatsResponse(**format_support())
153
+
154
+ @app.post("/v1/clean", response_model=JobCreated, status_code=202, tags=["clean"],
155
+ dependencies=guard)
156
+ async def create_clean(
157
+ request: Request,
158
+ files: list[UploadFile] = File(..., description="Files to scrub."),
159
+ keep_title: bool = Form(False),
160
+ keep_color_profile: bool = Form(True),
161
+ recurse: bool = Form(False, description="Descend into .zip/.tar/.7z/.eml members."),
162
+ media: bool = Form(False, description="Also scrub uploaded audio/video files."),
163
+ in_place: bool = Form(False, description="Ignored — the API always returns cleaned copies."),
164
+ report_lang: str = Form("en"),
165
+ ) -> JobCreated:
166
+ uploads = [f for f in files if f.filename]
167
+ if not uploads:
168
+ raise HTTPException(422, "No files provided.")
169
+ if len(uploads) > max_files:
170
+ raise HTTPException(413, f"too many files ({len(uploads)} > {max_files})")
171
+
172
+ # Stream each upload straight to a temp file (never the whole batch
173
+ # in memory), enforcing the size cap as we go. The worker later
174
+ # moves them into the job's run dir.
175
+ staged: list[tuple[str, str]] = [] # (original name, temp path)
176
+ total = 0
177
+ try:
178
+ for f in uploads:
179
+ tmp = tempfile.NamedTemporaryFile(prefix="metacls-up-", delete=False)
180
+ try:
181
+ while chunk := await f.read(_CHUNK):
182
+ total += len(chunk)
183
+ if total > max_upload_bytes:
184
+ raise HTTPException(413, f"upload exceeds {max_upload_mb} MB")
185
+ tmp.write(chunk)
186
+ finally:
187
+ tmp.close()
188
+ staged.append((secure_filename(f.filename or "") or "file", tmp.name))
189
+ except Exception:
190
+ for _n, p in staged:
191
+ _rm(p)
192
+ raise
193
+
194
+ def work(log, run_path: str) -> BatchReport:
195
+ up_dir = os.path.join(run_path, "uploads")
196
+ os.makedirs(up_dir, exist_ok=True)
197
+ paths: list[str] = []
198
+ for name, tmp_path in staged:
199
+ dest = os.path.join(up_dir, name)
200
+ stem, ext = os.path.splitext(dest)
201
+ n = 1
202
+ while dest in paths or os.path.exists(dest):
203
+ dest = f"{stem}({n}){ext}"
204
+ n += 1
205
+ shutil.move(tmp_path, dest)
206
+ paths.append(dest)
207
+ filetypes = list(CleanConfig().filetypes)
208
+ if media:
209
+ filetypes = sorted(set(filetypes) | MEDIA_EXTENSIONS)
210
+ if recurse:
211
+ filetypes = sorted(set(filetypes) | CONTAINER_EXTENSIONS)
212
+ cfg = CleanConfig(
213
+ filetypes=filetypes,
214
+ output_dir=os.path.join(run_path, "cleaned"),
215
+ keep_fields=["Title"] if keep_title else [],
216
+ keep_color_profile=keep_color_profile,
217
+ recurse=recurse,
218
+ )
219
+ return clean_file_list(paths, cfg, base_dir=up_dir, log=log)
220
+
221
+ try:
222
+ job = store.submit(report_lang=report_lang if report_lang in ("en", "tr") else "en",
223
+ file_count=len(staged), work=work)
224
+ except JobQueueFull as exc:
225
+ for _n, p in staged:
226
+ _rm(p)
227
+ raise HTTPException(429, str(exc)) from exc
228
+
229
+ return JobCreated(job_id=job.job_id, status="queued", run_id=job.run_id,
230
+ created_at=job.created_at.isoformat(), links=_links(request, job.job_id))
231
+
232
+ @app.get("/v1/clean", response_model=list[JobStatusResponse], tags=["clean"], dependencies=guard)
233
+ def list_jobs(request: Request) -> list[JobStatusResponse]:
234
+ return [_status(j, request) for j in store.list()]
235
+
236
+ @app.get("/v1/clean/{job_id}", response_model=JobStatusResponse, tags=["clean"], name="get_job", dependencies=guard)
237
+ def get_job(job_id: str, request: Request) -> JobStatusResponse:
238
+ return _status(_require(store, job_id), request)
239
+
240
+ @app.get("/v1/clean/{job_id}/log", response_model=JobLogResponse, tags=["clean"], name="get_job_log", dependencies=guard)
241
+ def get_job_log(job_id: str) -> JobLogResponse:
242
+ job = _require(store, job_id)
243
+ return JobLogResponse(job_id=job.job_id, status=job.status, lines=list(job.log_lines))
244
+
245
+ @app.get("/v1/clean/{job_id}/report.json", tags=["clean"], name="get_job_report_json", dependencies=guard)
246
+ def get_job_report_json(job_id: str) -> Response:
247
+ job = _require_done(store, job_id)
248
+ with open(os.path.join(job.run_path, "report.json"), encoding="utf-8") as fh:
249
+ return Response(fh.read(), media_type="application/json")
250
+
251
+ @app.get("/v1/clean/{job_id}/report.html", response_class=HTMLResponse, tags=["clean"],
252
+ name="get_job_report_html", dependencies=guard)
253
+ def get_job_report_html(job_id: str) -> str:
254
+ job = _require_done(store, job_id)
255
+ with open(os.path.join(job.run_path, "report.html"), encoding="utf-8") as fh:
256
+ return fh.read()
257
+
258
+ @app.get("/v1/clean/{job_id}/download", tags=["clean"], name="get_job_download", dependencies=guard)
259
+ def get_job_download(job_id: str) -> StreamingResponse:
260
+ job = _require_done(store, job_id)
261
+ buf = io.BytesIO()
262
+ with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
263
+ cleaned = os.path.join(job.run_path, "cleaned")
264
+ for root, _, names in os.walk(cleaned):
265
+ for name in names:
266
+ full = os.path.join(root, name)
267
+ zf.write(full, os.path.join("cleaned", os.path.relpath(full, cleaned)))
268
+ for meta in ("report.json", "report.html"):
269
+ p = os.path.join(job.run_path, meta)
270
+ if os.path.isfile(p):
271
+ zf.write(p, meta)
272
+ buf.seek(0)
273
+ headers = {"Content-Disposition": f'attachment; filename="metacls-{job.run_id}.zip"'}
274
+ return StreamingResponse(buf, media_type="application/zip", headers=headers)
275
+
276
+ return app
metacls/api/jobs.py ADDED
@@ -0,0 +1,230 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import os
5
+ import shutil
6
+ import sqlite3
7
+ import threading
8
+ import uuid
9
+ from collections.abc import Callable
10
+ from concurrent.futures import ThreadPoolExecutor
11
+ from datetime import datetime, timedelta, timezone
12
+
13
+ from ..models import BatchReport
14
+ from ..report import render_html_report, render_json_report
15
+
16
+ # work(log, run_path) -> BatchReport. Takes run_path explicitly (assigned
17
+ # before the job starts) so the worker never races on an attribute the
18
+ # route handler sets only after submit() returns.
19
+ JobWork = Callable[[Callable[[str], None], str], BatchReport]
20
+
21
+
22
+ class JobQueueFull(RuntimeError):
23
+ """Raised by JobStore.submit() when max_pending jobs are already
24
+ queued or running — a bounded 429 instead of unbounded growth.
25
+ """
26
+
27
+
28
+ class Job:
29
+ __slots__ = ("job_id", "run_id", "output_dir", "report_lang", "file_count", "status",
30
+ "created_at", "started_at", "finished_at", "error", "log_lines", "summary")
31
+
32
+ def __init__(self, *, job_id: str, run_id: str, output_dir: str, report_lang: str,
33
+ file_count: int = 0, status: str = "queued",
34
+ created_at: datetime | None = None, started_at: datetime | None = None,
35
+ finished_at: datetime | None = None, error: str | None = None,
36
+ summary: dict | None = None) -> None:
37
+ self.job_id = job_id
38
+ self.run_id = run_id
39
+ self.output_dir = output_dir
40
+ self.report_lang = report_lang
41
+ self.file_count = file_count
42
+ self.status = status
43
+ self.created_at = created_at or datetime.now(timezone.utc)
44
+ self.started_at = started_at
45
+ self.finished_at = finished_at
46
+ self.error = error
47
+ self.summary = summary # JobSummary dict, or None until done
48
+ self.log_lines: list[str] = [] # never persisted — process-local only
49
+
50
+ @property
51
+ def run_path(self) -> str:
52
+ return os.path.join(self.output_dir, self.run_id)
53
+
54
+
55
+ class JobStore:
56
+ """Job registry backed by a bounded thread pool and a SQLite file
57
+ (`<output_dir>/jobs.db`) so job status survives a process restart.
58
+ Log lines stay in memory only; a job's report.json / report.html and
59
+ cleaned files are on disk under `<output_dir>/<run_id>/`.
60
+ """
61
+
62
+ _LOG_LIMIT = 1000
63
+
64
+ def __init__(self, *, output_dir: str, max_workers: int = 2, max_jobs_kept: int = 500,
65
+ max_pending: int = 50, run_ttl_days: int = 0) -> None:
66
+ self.output_dir = output_dir
67
+ os.makedirs(output_dir, exist_ok=True)
68
+ self._executor = ThreadPoolExecutor(max_workers=max_workers, thread_name_prefix="metacls-job")
69
+ self._jobs: dict[str, Job] = {}
70
+ self._lock = threading.Lock()
71
+ self._max_jobs_kept = max_jobs_kept
72
+ self._max_pending = max_pending
73
+ self._run_ttl_days = run_ttl_days
74
+
75
+ self._db = sqlite3.connect(os.path.join(output_dir, "jobs.db"), check_same_thread=False)
76
+ self._db.execute("""
77
+ CREATE TABLE IF NOT EXISTS jobs (
78
+ job_id TEXT PRIMARY KEY, run_id TEXT, report_lang TEXT, file_count INTEGER,
79
+ status TEXT, created_at TEXT, started_at TEXT, finished_at TEXT,
80
+ error TEXT, summary_json TEXT
81
+ )""")
82
+ self._db.commit()
83
+ self._recover()
84
+ if run_ttl_days > 0:
85
+ self._sweep_expired()
86
+
87
+ # ------------------------------------------------------------------ persistence
88
+
89
+ def _recover(self) -> None:
90
+ rows = self._db.execute(
91
+ "SELECT job_id, run_id, report_lang, file_count, status, created_at, started_at, "
92
+ "finished_at, error, summary_json FROM jobs"
93
+ ).fetchall()
94
+ for (job_id, run_id, lang, fc, status, created, started, finished, error, summary_json) in rows:
95
+ if status in ("queued", "running"):
96
+ status, error = "error", (error or "interrupted by a service restart")
97
+ finished = finished or _now_iso()
98
+ self._db.execute("UPDATE jobs SET status=?, error=?, finished_at=? WHERE job_id=?",
99
+ (status, error, finished, job_id))
100
+ self._jobs[job_id] = Job(
101
+ job_id=job_id, run_id=run_id, output_dir=self.output_dir, report_lang=lang or "en",
102
+ file_count=fc or 0, status=status, created_at=_parse(created),
103
+ started_at=_parse(started), finished_at=_parse(finished), error=error,
104
+ summary=json.loads(summary_json) if summary_json else None,
105
+ )
106
+ self._db.commit()
107
+
108
+ def _persist(self, job: Job) -> None:
109
+ with self._lock:
110
+ self._db.execute(
111
+ "INSERT OR REPLACE INTO jobs VALUES (?,?,?,?,?,?,?,?,?,?)",
112
+ (job.job_id, job.run_id, job.report_lang, job.file_count, job.status,
113
+ _iso(job.created_at), _iso(job.started_at), _iso(job.finished_at), job.error,
114
+ json.dumps(job.summary) if job.summary else None),
115
+ )
116
+ self._db.commit()
117
+
118
+ def _sweep_expired(self) -> None:
119
+ cutoff = datetime.now(timezone.utc) - timedelta(days=self._run_ttl_days)
120
+ for job in list(self._jobs.values()):
121
+ if job.status in ("done", "error") and job.created_at < cutoff:
122
+ shutil.rmtree(job.run_path, ignore_errors=True)
123
+ with self._lock:
124
+ self._jobs.pop(job.job_id, None)
125
+ self._db.execute("DELETE FROM jobs WHERE job_id=?", (job.job_id,))
126
+ self._db.commit()
127
+
128
+ # ------------------------------------------------------------------ api
129
+
130
+ def submit(self, *, report_lang: str, file_count: int, work: JobWork) -> Job:
131
+ with self._lock:
132
+ pending = sum(1 for j in self._jobs.values() if j.status in ("queued", "running"))
133
+ if pending >= self._max_pending:
134
+ raise JobQueueFull(
135
+ f"{pending} job(s) already queued or running (limit {self._max_pending})."
136
+ )
137
+ job_id = uuid.uuid4().hex[:12]
138
+ run_id = f"api-{datetime.now().strftime('%Y%m%d-%H%M%S')}-{job_id[:8]}"
139
+ job = Job(job_id=job_id, run_id=run_id, output_dir=self.output_dir,
140
+ report_lang=report_lang, file_count=file_count)
141
+ self._jobs[job_id] = job
142
+ self._evict_locked()
143
+ self._persist(job)
144
+ self._executor.submit(self._run, job, work)
145
+ return job
146
+
147
+ def get(self, job_id: str) -> Job | None:
148
+ with self._lock:
149
+ return self._jobs.get(job_id)
150
+
151
+ def list(self) -> list[Job]:
152
+ with self._lock:
153
+ return sorted(self._jobs.values(), key=lambda j: j.created_at, reverse=True)
154
+
155
+ def shutdown(self) -> None:
156
+ self._executor.shutdown(wait=False, cancel_futures=True)
157
+ try:
158
+ self._db.close()
159
+ except sqlite3.Error:
160
+ pass
161
+
162
+ # ------------------------------------------------------------------ internals
163
+
164
+ def _evict_locked(self) -> None:
165
+ if len(self._jobs) <= self._max_jobs_kept:
166
+ return
167
+ finished = sorted(
168
+ (j for j in self._jobs.values() if j.status in ("done", "error")),
169
+ key=lambda j: j.created_at,
170
+ )
171
+ for j in finished:
172
+ if len(self._jobs) <= self._max_jobs_kept:
173
+ break
174
+ self._jobs.pop(j.job_id, None)
175
+ self._db.execute("DELETE FROM jobs WHERE job_id=?", (j.job_id,))
176
+ self._db.commit()
177
+
178
+ def _push_log(self, job: Job, message: str) -> None:
179
+ with self._lock:
180
+ job.log_lines.append(message)
181
+ if len(job.log_lines) > self._LOG_LIMIT:
182
+ job.log_lines = job.log_lines[-self._LOG_LIMIT:]
183
+
184
+ def _run(self, job: Job, work: JobWork) -> None:
185
+ job.status = "running"
186
+ job.started_at = datetime.now(timezone.utc)
187
+ self._persist(job)
188
+ try:
189
+ report = work(lambda m: self._push_log(job, m), job.run_path)
190
+ os.makedirs(job.run_path, exist_ok=True)
191
+ with open(os.path.join(job.run_path, "report.json"), "w", encoding="utf-8") as fh:
192
+ fh.write(render_json_report(report))
193
+ with open(os.path.join(job.run_path, "report.html"), "w", encoding="utf-8") as fh:
194
+ fh.write(render_html_report(report, lang=job.report_lang))
195
+ job.summary = _summary_dict(report)
196
+ job.status = "done"
197
+ except Exception as exc: # noqa: BLE001
198
+ job.error = str(exc)
199
+ job.status = "error"
200
+ self._push_log(job, f"! job failed: {exc}")
201
+ finally:
202
+ job.finished_at = datetime.now(timezone.utc)
203
+ self._persist(job)
204
+
205
+
206
+ def _summary_dict(report: BatchReport) -> dict:
207
+ return {
208
+ "files": len(report.results),
209
+ "by_status": report.counts,
210
+ "fields_removed": report.fields_removed,
211
+ "files_with_residual": len(report.files_with_residual),
212
+ "files_errored": len(report.errored),
213
+ }
214
+
215
+
216
+ def _now_iso() -> str:
217
+ return datetime.now(timezone.utc).isoformat()
218
+
219
+
220
+ def _iso(dt: datetime | None) -> str | None:
221
+ return dt.isoformat() if dt else None
222
+
223
+
224
+ def _parse(s: str | None) -> datetime | None:
225
+ if not s:
226
+ return None
227
+ try:
228
+ return datetime.fromisoformat(s)
229
+ except ValueError:
230
+ return None