metacls 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- metacls/__init__.py +8 -0
- metacls/_theme.py +130 -0
- metacls/api/__init__.py +5 -0
- metacls/api/app.py +276 -0
- metacls/api/jobs.py +230 -0
- metacls/api/schemas.py +81 -0
- metacls/cleaner.py +260 -0
- metacls/cli.py +571 -0
- metacls/config.py +160 -0
- metacls/diff.py +61 -0
- metacls/engines/__init__.py +124 -0
- metacls/engines/base.py +55 -0
- metacls/engines/container.py +387 -0
- metacls/engines/ebml_riff.py +304 -0
- metacls/engines/exiftool.py +197 -0
- metacls/engines/image.py +86 -0
- metacls/engines/legacy_office.py +178 -0
- metacls/engines/media.py +154 -0
- metacls/engines/office.py +429 -0
- metacls/engines/ole2.py +195 -0
- metacls/engines/pdf.py +503 -0
- metacls/engines/svg.py +141 -0
- metacls/models.py +79 -0
- metacls/py.typed +0 -0
- metacls/report/__init__.py +165 -0
- metacls/report/templates/report.html.jinja +139 -0
- metacls/scanner.py +82 -0
- metacls/watch.py +307 -0
- metacls/web.py +386 -0
- metacls-0.3.0.dist-info/METADATA +383 -0
- metacls-0.3.0.dist-info/RECORD +35 -0
- metacls-0.3.0.dist-info/WHEEL +5 -0
- metacls-0.3.0.dist-info/entry_points.txt +2 -0
- metacls-0.3.0.dist-info/licenses/LICENSE +21 -0
- metacls-0.3.0.dist-info/top_level.txt +1 -0
metacls/__init__.py
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""MetaCLS — bulk metadata scrubbing for PDF, Office and image files.
|
|
2
|
+
|
|
3
|
+
The remediation companion to MetaScout: MetaScout finds documents leaking
|
|
4
|
+
metadata, MetaCLS strips it and produces a before/after proof report.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
__version__ = "0.3.0"
|
metacls/_theme.py
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""MetaCLS's own visual identity — one source of truth for the CSS used
|
|
2
|
+
by both the local web UI (`web.py`) and the HTML report
|
|
3
|
+
(`report/templates/`).
|
|
4
|
+
|
|
5
|
+
Deliberately *not* MetaScout's look. MetaScout is blue-on-black, dark
|
|
6
|
+
only. MetaCLS is emerald/slate with a "swept clean" motif and a proper
|
|
7
|
+
light theme (follows `prefers-color-scheme`). Keeping the palette here,
|
|
8
|
+
rather than copy-pasted into two files, is what keeps the two surfaces
|
|
9
|
+
looking like one product.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
# Design tokens. Dark is the primary look; the light block is a full
|
|
14
|
+
# re-theme, not an afterthought.
|
|
15
|
+
TOKENS_CSS = """
|
|
16
|
+
:root {
|
|
17
|
+
--bg: #0c1110;
|
|
18
|
+
--panel: #141b1a;
|
|
19
|
+
--panel-2: #1a2321;
|
|
20
|
+
--border: #263230;
|
|
21
|
+
--text: #e9efed;
|
|
22
|
+
--muted: #93a5a0;
|
|
23
|
+
--accent: #2dd4a7; /* emerald-teal — MetaCLS's signature */
|
|
24
|
+
--accent-ink:#04140f;
|
|
25
|
+
--link: #5ad1c4;
|
|
26
|
+
--good: #2dd4a7;
|
|
27
|
+
--warn: #f6c454;
|
|
28
|
+
--bad: #f2778d;
|
|
29
|
+
--sweep: linear-gradient(100deg, rgba(45,212,167,.16), rgba(45,212,167,0) 60%);
|
|
30
|
+
}
|
|
31
|
+
@media (prefers-color-scheme: light) {
|
|
32
|
+
:root {
|
|
33
|
+
--bg: #f4f7f6;
|
|
34
|
+
--panel: #ffffff;
|
|
35
|
+
--panel-2: #eef3f1;
|
|
36
|
+
--border: #dde5e3;
|
|
37
|
+
--text: #13201d;
|
|
38
|
+
--muted: #5c6b67;
|
|
39
|
+
--accent: #0f9e78;
|
|
40
|
+
--accent-ink:#ffffff;
|
|
41
|
+
--link: #0c7c6a;
|
|
42
|
+
--good: #0f9e78;
|
|
43
|
+
--warn: #b57d12;
|
|
44
|
+
--bad: #c2405a;
|
|
45
|
+
--sweep: linear-gradient(100deg, rgba(15,158,120,.10), rgba(15,158,120,0) 60%);
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
# Shared element + component styles. The `msc-` prefix keeps these from
|
|
51
|
+
# colliding with anything in a report card's own escaped content.
|
|
52
|
+
BASE_CSS = """
|
|
53
|
+
* { box-sizing: border-box; }
|
|
54
|
+
body { margin:0; background:var(--bg); color:var(--text);
|
|
55
|
+
font-family:"Inter",-apple-system,"Segoe UI",Roboto,sans-serif; font-size:14px;
|
|
56
|
+
line-height:1.5; -webkit-font-smoothing:antialiased; }
|
|
57
|
+
a { color:var(--link); }
|
|
58
|
+
code, .mono { font-family:ui-monospace,"SF Mono",SFMono-Regular,Menlo,monospace; }
|
|
59
|
+
|
|
60
|
+
.msc-header { padding:26px 40px 20px; border-bottom:1px solid var(--border);
|
|
61
|
+
background:var(--sweep); background-repeat:no-repeat; }
|
|
62
|
+
.msc-mark { font-weight:800; font-size:21px; letter-spacing:-.01em; }
|
|
63
|
+
.msc-mark b { color:var(--accent); font-weight:800; }
|
|
64
|
+
.msc-mark::before { content:""; display:inline-block; width:22px; height:3px; border-radius:2px;
|
|
65
|
+
background:var(--accent); margin:0 10px 5px 0; vertical-align:middle; }
|
|
66
|
+
.msc-tag { color:var(--muted); font-size:12.5px; margin-top:5px; }
|
|
67
|
+
.msc-nav { margin-top:12px; display:flex; gap:18px; }
|
|
68
|
+
.msc-nav a { color:var(--muted); text-decoration:none; font-size:12.5px; font-weight:600;
|
|
69
|
+
padding-bottom:3px; border-bottom:2px solid transparent; }
|
|
70
|
+
.msc-nav a.on, .msc-nav a:hover { color:var(--accent); border-bottom-color:var(--accent); }
|
|
71
|
+
.msc-lang a { color:var(--muted); text-decoration:none; border:1px solid var(--border);
|
|
72
|
+
border-radius:999px; padding:3px 10px; font-size:11.5px; font-weight:700; margin-left:6px; }
|
|
73
|
+
.msc-lang a.on { background:var(--accent); color:var(--accent-ink); border-color:var(--accent); }
|
|
74
|
+
|
|
75
|
+
.msc-main { padding:26px 40px 72px; max-width:880px; margin:0 auto; }
|
|
76
|
+
|
|
77
|
+
.msc-hero { border:1px solid var(--border); border-radius:14px; overflow:hidden;
|
|
78
|
+
background:var(--panel); margin-bottom:22px; }
|
|
79
|
+
.msc-hero-band { padding:18px 22px; background:var(--sweep), var(--panel-2);
|
|
80
|
+
border-bottom:1px solid var(--border); display:flex; align-items:baseline; gap:12px; flex-wrap:wrap; }
|
|
81
|
+
.msc-hero-state { font-size:19px; font-weight:800; letter-spacing:.02em; }
|
|
82
|
+
.msc-hero-state.ok { color:var(--good); }
|
|
83
|
+
.msc-hero-state.warn { color:var(--warn); }
|
|
84
|
+
.msc-hero-state.bad { color:var(--bad); }
|
|
85
|
+
.msc-hero-sub { color:var(--muted); font-size:12.5px; }
|
|
86
|
+
.msc-stats { display:flex; flex-wrap:wrap; gap:10px; padding:16px 22px; }
|
|
87
|
+
.msc-stat { flex:1; min-width:120px; background:var(--panel-2); border:1px solid var(--border);
|
|
88
|
+
border-radius:10px; padding:12px 14px; }
|
|
89
|
+
.msc-stat .n { font-size:20px; font-weight:800; }
|
|
90
|
+
.msc-stat .l { color:var(--muted); font-size:11.5px; margin-top:2px; letter-spacing:.02em; }
|
|
91
|
+
.msc-stat.warn .n { color:var(--warn); }
|
|
92
|
+
.msc-stat.bad .n { color:var(--bad); }
|
|
93
|
+
|
|
94
|
+
.msc-card { background:var(--panel); border:1px solid var(--border); border-left:3px solid var(--border);
|
|
95
|
+
border-radius:12px; padding:18px 20px; margin-bottom:14px; }
|
|
96
|
+
.msc-card.is-cleaned { border-left-color:var(--good); }
|
|
97
|
+
.msc-card.is-skipped { border-left-color:var(--warn); }
|
|
98
|
+
.msc-card.is-error { border-left-color:var(--bad); }
|
|
99
|
+
.msc-card.is-unsupported { border-left-color:var(--muted); }
|
|
100
|
+
.msc-file { font-family:ui-monospace,"SF Mono",SFMono-Regular,Menlo,monospace; font-size:13px;
|
|
101
|
+
word-break:break-all; }
|
|
102
|
+
.msc-meta { color:var(--muted); font-size:12px; }
|
|
103
|
+
|
|
104
|
+
.msc-pill { display:inline-block; font-size:10.5px; font-weight:800; padding:2px 9px;
|
|
105
|
+
border-radius:999px; text-transform:uppercase; letter-spacing:.05em; }
|
|
106
|
+
.msc-pill.cleaned{ background:color-mix(in srgb,var(--good) 18%,transparent); color:var(--good); }
|
|
107
|
+
.msc-pill.skipped{ background:color-mix(in srgb,var(--warn) 18%,transparent); color:var(--warn); }
|
|
108
|
+
.msc-pill.error{ background:color-mix(in srgb,var(--bad) 18%,transparent); color:var(--bad); }
|
|
109
|
+
.msc-pill.unsupported{ background:color-mix(in srgb,var(--muted) 18%,transparent); color:var(--muted); }
|
|
110
|
+
|
|
111
|
+
table.msc-diff { width:100%; border-collapse:collapse; margin-top:10px; font-size:12.5px; }
|
|
112
|
+
table.msc-diff th, table.msc-diff td { text-align:left; padding:6px 10px;
|
|
113
|
+
border-bottom:1px solid var(--border); vertical-align:top; }
|
|
114
|
+
table.msc-diff th { color:var(--muted); font-weight:600; }
|
|
115
|
+
table.msc-diff td.was { font-family:ui-monospace,SFMono-Regular,Menlo,monospace; color:var(--muted);
|
|
116
|
+
text-decoration:line-through; text-decoration-color:color-mix(in srgb,var(--bad) 60%,transparent);
|
|
117
|
+
word-break:break-word; }
|
|
118
|
+
.msc-kept { color:var(--good); font-size:12px; margin-top:10px; }
|
|
119
|
+
.msc-residual { color:var(--bad); font-size:12px; margin-top:10px; font-weight:600; }
|
|
120
|
+
.msc-empty { color:var(--muted); font-size:12.5px; margin-top:8px; }
|
|
121
|
+
|
|
122
|
+
.msc-btn { background:var(--accent); color:var(--accent-ink); border:0; border-radius:9px;
|
|
123
|
+
padding:12px 22px; font-weight:800; font-size:14px; cursor:pointer; }
|
|
124
|
+
.msc-btn:disabled { opacity:.6; cursor:wait; }
|
|
125
|
+
.msc-actions a { margin-right:16px; font-weight:600; font-size:13px; }
|
|
126
|
+
"""
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def full_css() -> str:
|
|
130
|
+
return TOKENS_CSS + BASE_CSS
|
metacls/api/__init__.py
ADDED
metacls/api/app.py
ADDED
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import io
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import secrets
|
|
7
|
+
import shutil
|
|
8
|
+
import tempfile
|
|
9
|
+
import time
|
|
10
|
+
import zipfile
|
|
11
|
+
from contextlib import asynccontextmanager
|
|
12
|
+
from datetime import datetime, timezone
|
|
13
|
+
|
|
14
|
+
from fastapi import Depends, FastAPI, File, Form, HTTPException, Request, UploadFile
|
|
15
|
+
from fastapi.responses import HTMLResponse, Response, StreamingResponse
|
|
16
|
+
from werkzeug.utils import secure_filename
|
|
17
|
+
|
|
18
|
+
from .. import __version__
|
|
19
|
+
from ..cleaner import clean_file_list
|
|
20
|
+
from ..config import CONTAINER_EXTENSIONS, MEDIA_EXTENSIONS, CleanConfig
|
|
21
|
+
from ..engines import format_support
|
|
22
|
+
from ..models import BatchReport
|
|
23
|
+
from .jobs import Job, JobQueueFull, JobStore
|
|
24
|
+
from .schemas import (
|
|
25
|
+
FormatsResponse,
|
|
26
|
+
HealthResponse,
|
|
27
|
+
JobCreated,
|
|
28
|
+
JobLogResponse,
|
|
29
|
+
JobStatusResponse,
|
|
30
|
+
JobSummary,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
_CHUNK = 1024 * 1024
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _rm(path: str) -> None:
|
|
37
|
+
try:
|
|
38
|
+
os.remove(path)
|
|
39
|
+
except OSError:
|
|
40
|
+
pass
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _summary(job: Job) -> JobSummary | None:
|
|
44
|
+
return JobSummary(**job.summary) if job.summary else None
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _links(request: Request, job_id: str) -> dict[str, str]:
|
|
48
|
+
return {
|
|
49
|
+
"self": str(request.url_for("get_job", job_id=job_id)),
|
|
50
|
+
"log": str(request.url_for("get_job_log", job_id=job_id)),
|
|
51
|
+
"report_json": str(request.url_for("get_job_report_json", job_id=job_id)),
|
|
52
|
+
"report_html": str(request.url_for("get_job_report_html", job_id=job_id)),
|
|
53
|
+
"download": str(request.url_for("get_job_download", job_id=job_id)),
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _status(job: Job, request: Request) -> JobStatusResponse:
|
|
58
|
+
return JobStatusResponse(
|
|
59
|
+
job_id=job.job_id, status=job.status, run_id=job.run_id, file_count=job.file_count,
|
|
60
|
+
created_at=job.created_at.isoformat(),
|
|
61
|
+
started_at=job.started_at.isoformat() if job.started_at else None,
|
|
62
|
+
finished_at=job.finished_at.isoformat() if job.finished_at else None,
|
|
63
|
+
error=job.error,
|
|
64
|
+
summary=_summary(job),
|
|
65
|
+
links=_links(request, job.job_id),
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _require(store: JobStore, job_id: str) -> Job:
|
|
70
|
+
job = store.get(job_id)
|
|
71
|
+
if job is None:
|
|
72
|
+
raise HTTPException(404, f"No job {job_id!r}. Jobs are in-memory and don't survive a restart.")
|
|
73
|
+
return job
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _require_done(store: JobStore, job_id: str) -> Job:
|
|
77
|
+
job = _require(store, job_id)
|
|
78
|
+
if job.status != "done":
|
|
79
|
+
raise HTTPException(409, {"job_id": job.job_id, "status": job.status, "error": job.error,
|
|
80
|
+
"message": "Not ready — poll GET /v1/clean/{job_id} until status is 'done'."})
|
|
81
|
+
return job
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def create_app(*, output_dir: str = "./metacls_cleaned", max_workers: int = 2,
|
|
85
|
+
max_pending: int = 50, api_key: str | None = None,
|
|
86
|
+
max_upload_mb: int = 200, max_files: int = 50,
|
|
87
|
+
run_ttl_days: int = 0, log_json: bool = False) -> FastAPI:
|
|
88
|
+
"""MetaCLS REST API — POST files, poll the job, pull the cleaned
|
|
89
|
+
files + report as a zip. Every job runs in a bounded background thread
|
|
90
|
+
pool so POST returns immediately; max_pending caps queued+running jobs.
|
|
91
|
+
|
|
92
|
+
`api_key`, when set, is required on every `/v1/*` route except the
|
|
93
|
+
open capability routes `/v1/health` and `/v1/formats` — as
|
|
94
|
+
`X-API-Key: <key>` or `Authorization: Bearer <key>`. `max_upload_mb` /
|
|
95
|
+
`max_files` bound a single request.
|
|
96
|
+
"""
|
|
97
|
+
os.makedirs(output_dir, exist_ok=True)
|
|
98
|
+
store = JobStore(output_dir=output_dir, max_workers=max_workers, max_pending=max_pending,
|
|
99
|
+
run_ttl_days=run_ttl_days)
|
|
100
|
+
max_upload_bytes = max_upload_mb * 1024 * 1024
|
|
101
|
+
|
|
102
|
+
def require_key(request: Request) -> None:
|
|
103
|
+
if not api_key:
|
|
104
|
+
return
|
|
105
|
+
supplied = request.headers.get("x-api-key")
|
|
106
|
+
if not supplied:
|
|
107
|
+
auth = request.headers.get("authorization", "")
|
|
108
|
+
if auth.lower().startswith("bearer "):
|
|
109
|
+
supplied = auth[7:]
|
|
110
|
+
if not (supplied and secrets.compare_digest(supplied, api_key)):
|
|
111
|
+
raise HTTPException(401, "missing or invalid API key")
|
|
112
|
+
|
|
113
|
+
guard = [Depends(require_key)]
|
|
114
|
+
|
|
115
|
+
@asynccontextmanager
|
|
116
|
+
async def lifespan(_app: FastAPI):
|
|
117
|
+
yield
|
|
118
|
+
store.shutdown()
|
|
119
|
+
|
|
120
|
+
app = FastAPI(
|
|
121
|
+
title="MetaCLS API",
|
|
122
|
+
version=__version__,
|
|
123
|
+
description="Job-based bulk metadata scrubbing for PDF/Office/image files. "
|
|
124
|
+
"Set an API key before exposing this — see the README.",
|
|
125
|
+
lifespan=lifespan,
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
if log_json:
|
|
129
|
+
@app.middleware("http")
|
|
130
|
+
async def _access_log(request: Request, call_next):
|
|
131
|
+
t0 = time.monotonic()
|
|
132
|
+
response = await call_next(request)
|
|
133
|
+
print(json.dumps({
|
|
134
|
+
"ts": datetime.now(timezone.utc).isoformat(),
|
|
135
|
+
"method": request.method, "path": request.url.path,
|
|
136
|
+
"status": response.status_code,
|
|
137
|
+
"ms": round((time.monotonic() - t0) * 1000, 1),
|
|
138
|
+
"client": request.client.host if request.client else None,
|
|
139
|
+
}), flush=True)
|
|
140
|
+
return response
|
|
141
|
+
|
|
142
|
+
@app.get("/v1/health", response_model=HealthResponse, tags=["meta"])
|
|
143
|
+
def health() -> HealthResponse:
|
|
144
|
+
active = sum(1 for j in store.list() if j.status in ("queued", "running"))
|
|
145
|
+
return HealthResponse(status="ok", version=__version__, active_jobs=active)
|
|
146
|
+
|
|
147
|
+
@app.get("/v1/formats", response_model=FormatsResponse, tags=["meta"])
|
|
148
|
+
def formats() -> FormatsResponse:
|
|
149
|
+
"""What this server can scrub: every supported extension, the
|
|
150
|
+
extensions each engine handles, and which optional pieces
|
|
151
|
+
(exiftool, LibreOffice, mutagen, py7zr, …) are installed here."""
|
|
152
|
+
return FormatsResponse(**format_support())
|
|
153
|
+
|
|
154
|
+
@app.post("/v1/clean", response_model=JobCreated, status_code=202, tags=["clean"],
|
|
155
|
+
dependencies=guard)
|
|
156
|
+
async def create_clean(
|
|
157
|
+
request: Request,
|
|
158
|
+
files: list[UploadFile] = File(..., description="Files to scrub."),
|
|
159
|
+
keep_title: bool = Form(False),
|
|
160
|
+
keep_color_profile: bool = Form(True),
|
|
161
|
+
recurse: bool = Form(False, description="Descend into .zip/.tar/.7z/.eml members."),
|
|
162
|
+
media: bool = Form(False, description="Also scrub uploaded audio/video files."),
|
|
163
|
+
in_place: bool = Form(False, description="Ignored — the API always returns cleaned copies."),
|
|
164
|
+
report_lang: str = Form("en"),
|
|
165
|
+
) -> JobCreated:
|
|
166
|
+
uploads = [f for f in files if f.filename]
|
|
167
|
+
if not uploads:
|
|
168
|
+
raise HTTPException(422, "No files provided.")
|
|
169
|
+
if len(uploads) > max_files:
|
|
170
|
+
raise HTTPException(413, f"too many files ({len(uploads)} > {max_files})")
|
|
171
|
+
|
|
172
|
+
# Stream each upload straight to a temp file (never the whole batch
|
|
173
|
+
# in memory), enforcing the size cap as we go. The worker later
|
|
174
|
+
# moves them into the job's run dir.
|
|
175
|
+
staged: list[tuple[str, str]] = [] # (original name, temp path)
|
|
176
|
+
total = 0
|
|
177
|
+
try:
|
|
178
|
+
for f in uploads:
|
|
179
|
+
tmp = tempfile.NamedTemporaryFile(prefix="metacls-up-", delete=False)
|
|
180
|
+
try:
|
|
181
|
+
while chunk := await f.read(_CHUNK):
|
|
182
|
+
total += len(chunk)
|
|
183
|
+
if total > max_upload_bytes:
|
|
184
|
+
raise HTTPException(413, f"upload exceeds {max_upload_mb} MB")
|
|
185
|
+
tmp.write(chunk)
|
|
186
|
+
finally:
|
|
187
|
+
tmp.close()
|
|
188
|
+
staged.append((secure_filename(f.filename or "") or "file", tmp.name))
|
|
189
|
+
except Exception:
|
|
190
|
+
for _n, p in staged:
|
|
191
|
+
_rm(p)
|
|
192
|
+
raise
|
|
193
|
+
|
|
194
|
+
def work(log, run_path: str) -> BatchReport:
|
|
195
|
+
up_dir = os.path.join(run_path, "uploads")
|
|
196
|
+
os.makedirs(up_dir, exist_ok=True)
|
|
197
|
+
paths: list[str] = []
|
|
198
|
+
for name, tmp_path in staged:
|
|
199
|
+
dest = os.path.join(up_dir, name)
|
|
200
|
+
stem, ext = os.path.splitext(dest)
|
|
201
|
+
n = 1
|
|
202
|
+
while dest in paths or os.path.exists(dest):
|
|
203
|
+
dest = f"{stem}({n}){ext}"
|
|
204
|
+
n += 1
|
|
205
|
+
shutil.move(tmp_path, dest)
|
|
206
|
+
paths.append(dest)
|
|
207
|
+
filetypes = list(CleanConfig().filetypes)
|
|
208
|
+
if media:
|
|
209
|
+
filetypes = sorted(set(filetypes) | MEDIA_EXTENSIONS)
|
|
210
|
+
if recurse:
|
|
211
|
+
filetypes = sorted(set(filetypes) | CONTAINER_EXTENSIONS)
|
|
212
|
+
cfg = CleanConfig(
|
|
213
|
+
filetypes=filetypes,
|
|
214
|
+
output_dir=os.path.join(run_path, "cleaned"),
|
|
215
|
+
keep_fields=["Title"] if keep_title else [],
|
|
216
|
+
keep_color_profile=keep_color_profile,
|
|
217
|
+
recurse=recurse,
|
|
218
|
+
)
|
|
219
|
+
return clean_file_list(paths, cfg, base_dir=up_dir, log=log)
|
|
220
|
+
|
|
221
|
+
try:
|
|
222
|
+
job = store.submit(report_lang=report_lang if report_lang in ("en", "tr") else "en",
|
|
223
|
+
file_count=len(staged), work=work)
|
|
224
|
+
except JobQueueFull as exc:
|
|
225
|
+
for _n, p in staged:
|
|
226
|
+
_rm(p)
|
|
227
|
+
raise HTTPException(429, str(exc)) from exc
|
|
228
|
+
|
|
229
|
+
return JobCreated(job_id=job.job_id, status="queued", run_id=job.run_id,
|
|
230
|
+
created_at=job.created_at.isoformat(), links=_links(request, job.job_id))
|
|
231
|
+
|
|
232
|
+
@app.get("/v1/clean", response_model=list[JobStatusResponse], tags=["clean"], dependencies=guard)
|
|
233
|
+
def list_jobs(request: Request) -> list[JobStatusResponse]:
|
|
234
|
+
return [_status(j, request) for j in store.list()]
|
|
235
|
+
|
|
236
|
+
@app.get("/v1/clean/{job_id}", response_model=JobStatusResponse, tags=["clean"], name="get_job", dependencies=guard)
|
|
237
|
+
def get_job(job_id: str, request: Request) -> JobStatusResponse:
|
|
238
|
+
return _status(_require(store, job_id), request)
|
|
239
|
+
|
|
240
|
+
@app.get("/v1/clean/{job_id}/log", response_model=JobLogResponse, tags=["clean"], name="get_job_log", dependencies=guard)
|
|
241
|
+
def get_job_log(job_id: str) -> JobLogResponse:
|
|
242
|
+
job = _require(store, job_id)
|
|
243
|
+
return JobLogResponse(job_id=job.job_id, status=job.status, lines=list(job.log_lines))
|
|
244
|
+
|
|
245
|
+
@app.get("/v1/clean/{job_id}/report.json", tags=["clean"], name="get_job_report_json", dependencies=guard)
|
|
246
|
+
def get_job_report_json(job_id: str) -> Response:
|
|
247
|
+
job = _require_done(store, job_id)
|
|
248
|
+
with open(os.path.join(job.run_path, "report.json"), encoding="utf-8") as fh:
|
|
249
|
+
return Response(fh.read(), media_type="application/json")
|
|
250
|
+
|
|
251
|
+
@app.get("/v1/clean/{job_id}/report.html", response_class=HTMLResponse, tags=["clean"],
|
|
252
|
+
name="get_job_report_html", dependencies=guard)
|
|
253
|
+
def get_job_report_html(job_id: str) -> str:
|
|
254
|
+
job = _require_done(store, job_id)
|
|
255
|
+
with open(os.path.join(job.run_path, "report.html"), encoding="utf-8") as fh:
|
|
256
|
+
return fh.read()
|
|
257
|
+
|
|
258
|
+
@app.get("/v1/clean/{job_id}/download", tags=["clean"], name="get_job_download", dependencies=guard)
|
|
259
|
+
def get_job_download(job_id: str) -> StreamingResponse:
|
|
260
|
+
job = _require_done(store, job_id)
|
|
261
|
+
buf = io.BytesIO()
|
|
262
|
+
with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
|
|
263
|
+
cleaned = os.path.join(job.run_path, "cleaned")
|
|
264
|
+
for root, _, names in os.walk(cleaned):
|
|
265
|
+
for name in names:
|
|
266
|
+
full = os.path.join(root, name)
|
|
267
|
+
zf.write(full, os.path.join("cleaned", os.path.relpath(full, cleaned)))
|
|
268
|
+
for meta in ("report.json", "report.html"):
|
|
269
|
+
p = os.path.join(job.run_path, meta)
|
|
270
|
+
if os.path.isfile(p):
|
|
271
|
+
zf.write(p, meta)
|
|
272
|
+
buf.seek(0)
|
|
273
|
+
headers = {"Content-Disposition": f'attachment; filename="metacls-{job.run_id}.zip"'}
|
|
274
|
+
return StreamingResponse(buf, media_type="application/zip", headers=headers)
|
|
275
|
+
|
|
276
|
+
return app
|
metacls/api/jobs.py
ADDED
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import shutil
|
|
6
|
+
import sqlite3
|
|
7
|
+
import threading
|
|
8
|
+
import uuid
|
|
9
|
+
from collections.abc import Callable
|
|
10
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
11
|
+
from datetime import datetime, timedelta, timezone
|
|
12
|
+
|
|
13
|
+
from ..models import BatchReport
|
|
14
|
+
from ..report import render_html_report, render_json_report
|
|
15
|
+
|
|
16
|
+
# work(log, run_path) -> BatchReport. Takes run_path explicitly (assigned
|
|
17
|
+
# before the job starts) so the worker never races on an attribute the
|
|
18
|
+
# route handler sets only after submit() returns.
|
|
19
|
+
JobWork = Callable[[Callable[[str], None], str], BatchReport]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class JobQueueFull(RuntimeError):
|
|
23
|
+
"""Raised by JobStore.submit() when max_pending jobs are already
|
|
24
|
+
queued or running — a bounded 429 instead of unbounded growth.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class Job:
|
|
29
|
+
__slots__ = ("job_id", "run_id", "output_dir", "report_lang", "file_count", "status",
|
|
30
|
+
"created_at", "started_at", "finished_at", "error", "log_lines", "summary")
|
|
31
|
+
|
|
32
|
+
def __init__(self, *, job_id: str, run_id: str, output_dir: str, report_lang: str,
|
|
33
|
+
file_count: int = 0, status: str = "queued",
|
|
34
|
+
created_at: datetime | None = None, started_at: datetime | None = None,
|
|
35
|
+
finished_at: datetime | None = None, error: str | None = None,
|
|
36
|
+
summary: dict | None = None) -> None:
|
|
37
|
+
self.job_id = job_id
|
|
38
|
+
self.run_id = run_id
|
|
39
|
+
self.output_dir = output_dir
|
|
40
|
+
self.report_lang = report_lang
|
|
41
|
+
self.file_count = file_count
|
|
42
|
+
self.status = status
|
|
43
|
+
self.created_at = created_at or datetime.now(timezone.utc)
|
|
44
|
+
self.started_at = started_at
|
|
45
|
+
self.finished_at = finished_at
|
|
46
|
+
self.error = error
|
|
47
|
+
self.summary = summary # JobSummary dict, or None until done
|
|
48
|
+
self.log_lines: list[str] = [] # never persisted — process-local only
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def run_path(self) -> str:
|
|
52
|
+
return os.path.join(self.output_dir, self.run_id)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class JobStore:
|
|
56
|
+
"""Job registry backed by a bounded thread pool and a SQLite file
|
|
57
|
+
(`<output_dir>/jobs.db`) so job status survives a process restart.
|
|
58
|
+
Log lines stay in memory only; a job's report.json / report.html and
|
|
59
|
+
cleaned files are on disk under `<output_dir>/<run_id>/`.
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
_LOG_LIMIT = 1000
|
|
63
|
+
|
|
64
|
+
def __init__(self, *, output_dir: str, max_workers: int = 2, max_jobs_kept: int = 500,
|
|
65
|
+
max_pending: int = 50, run_ttl_days: int = 0) -> None:
|
|
66
|
+
self.output_dir = output_dir
|
|
67
|
+
os.makedirs(output_dir, exist_ok=True)
|
|
68
|
+
self._executor = ThreadPoolExecutor(max_workers=max_workers, thread_name_prefix="metacls-job")
|
|
69
|
+
self._jobs: dict[str, Job] = {}
|
|
70
|
+
self._lock = threading.Lock()
|
|
71
|
+
self._max_jobs_kept = max_jobs_kept
|
|
72
|
+
self._max_pending = max_pending
|
|
73
|
+
self._run_ttl_days = run_ttl_days
|
|
74
|
+
|
|
75
|
+
self._db = sqlite3.connect(os.path.join(output_dir, "jobs.db"), check_same_thread=False)
|
|
76
|
+
self._db.execute("""
|
|
77
|
+
CREATE TABLE IF NOT EXISTS jobs (
|
|
78
|
+
job_id TEXT PRIMARY KEY, run_id TEXT, report_lang TEXT, file_count INTEGER,
|
|
79
|
+
status TEXT, created_at TEXT, started_at TEXT, finished_at TEXT,
|
|
80
|
+
error TEXT, summary_json TEXT
|
|
81
|
+
)""")
|
|
82
|
+
self._db.commit()
|
|
83
|
+
self._recover()
|
|
84
|
+
if run_ttl_days > 0:
|
|
85
|
+
self._sweep_expired()
|
|
86
|
+
|
|
87
|
+
# ------------------------------------------------------------------ persistence
|
|
88
|
+
|
|
89
|
+
def _recover(self) -> None:
|
|
90
|
+
rows = self._db.execute(
|
|
91
|
+
"SELECT job_id, run_id, report_lang, file_count, status, created_at, started_at, "
|
|
92
|
+
"finished_at, error, summary_json FROM jobs"
|
|
93
|
+
).fetchall()
|
|
94
|
+
for (job_id, run_id, lang, fc, status, created, started, finished, error, summary_json) in rows:
|
|
95
|
+
if status in ("queued", "running"):
|
|
96
|
+
status, error = "error", (error or "interrupted by a service restart")
|
|
97
|
+
finished = finished or _now_iso()
|
|
98
|
+
self._db.execute("UPDATE jobs SET status=?, error=?, finished_at=? WHERE job_id=?",
|
|
99
|
+
(status, error, finished, job_id))
|
|
100
|
+
self._jobs[job_id] = Job(
|
|
101
|
+
job_id=job_id, run_id=run_id, output_dir=self.output_dir, report_lang=lang or "en",
|
|
102
|
+
file_count=fc or 0, status=status, created_at=_parse(created),
|
|
103
|
+
started_at=_parse(started), finished_at=_parse(finished), error=error,
|
|
104
|
+
summary=json.loads(summary_json) if summary_json else None,
|
|
105
|
+
)
|
|
106
|
+
self._db.commit()
|
|
107
|
+
|
|
108
|
+
def _persist(self, job: Job) -> None:
|
|
109
|
+
with self._lock:
|
|
110
|
+
self._db.execute(
|
|
111
|
+
"INSERT OR REPLACE INTO jobs VALUES (?,?,?,?,?,?,?,?,?,?)",
|
|
112
|
+
(job.job_id, job.run_id, job.report_lang, job.file_count, job.status,
|
|
113
|
+
_iso(job.created_at), _iso(job.started_at), _iso(job.finished_at), job.error,
|
|
114
|
+
json.dumps(job.summary) if job.summary else None),
|
|
115
|
+
)
|
|
116
|
+
self._db.commit()
|
|
117
|
+
|
|
118
|
+
def _sweep_expired(self) -> None:
|
|
119
|
+
cutoff = datetime.now(timezone.utc) - timedelta(days=self._run_ttl_days)
|
|
120
|
+
for job in list(self._jobs.values()):
|
|
121
|
+
if job.status in ("done", "error") and job.created_at < cutoff:
|
|
122
|
+
shutil.rmtree(job.run_path, ignore_errors=True)
|
|
123
|
+
with self._lock:
|
|
124
|
+
self._jobs.pop(job.job_id, None)
|
|
125
|
+
self._db.execute("DELETE FROM jobs WHERE job_id=?", (job.job_id,))
|
|
126
|
+
self._db.commit()
|
|
127
|
+
|
|
128
|
+
# ------------------------------------------------------------------ api
|
|
129
|
+
|
|
130
|
+
def submit(self, *, report_lang: str, file_count: int, work: JobWork) -> Job:
|
|
131
|
+
with self._lock:
|
|
132
|
+
pending = sum(1 for j in self._jobs.values() if j.status in ("queued", "running"))
|
|
133
|
+
if pending >= self._max_pending:
|
|
134
|
+
raise JobQueueFull(
|
|
135
|
+
f"{pending} job(s) already queued or running (limit {self._max_pending})."
|
|
136
|
+
)
|
|
137
|
+
job_id = uuid.uuid4().hex[:12]
|
|
138
|
+
run_id = f"api-{datetime.now().strftime('%Y%m%d-%H%M%S')}-{job_id[:8]}"
|
|
139
|
+
job = Job(job_id=job_id, run_id=run_id, output_dir=self.output_dir,
|
|
140
|
+
report_lang=report_lang, file_count=file_count)
|
|
141
|
+
self._jobs[job_id] = job
|
|
142
|
+
self._evict_locked()
|
|
143
|
+
self._persist(job)
|
|
144
|
+
self._executor.submit(self._run, job, work)
|
|
145
|
+
return job
|
|
146
|
+
|
|
147
|
+
def get(self, job_id: str) -> Job | None:
|
|
148
|
+
with self._lock:
|
|
149
|
+
return self._jobs.get(job_id)
|
|
150
|
+
|
|
151
|
+
def list(self) -> list[Job]:
|
|
152
|
+
with self._lock:
|
|
153
|
+
return sorted(self._jobs.values(), key=lambda j: j.created_at, reverse=True)
|
|
154
|
+
|
|
155
|
+
def shutdown(self) -> None:
|
|
156
|
+
self._executor.shutdown(wait=False, cancel_futures=True)
|
|
157
|
+
try:
|
|
158
|
+
self._db.close()
|
|
159
|
+
except sqlite3.Error:
|
|
160
|
+
pass
|
|
161
|
+
|
|
162
|
+
# ------------------------------------------------------------------ internals
|
|
163
|
+
|
|
164
|
+
def _evict_locked(self) -> None:
|
|
165
|
+
if len(self._jobs) <= self._max_jobs_kept:
|
|
166
|
+
return
|
|
167
|
+
finished = sorted(
|
|
168
|
+
(j for j in self._jobs.values() if j.status in ("done", "error")),
|
|
169
|
+
key=lambda j: j.created_at,
|
|
170
|
+
)
|
|
171
|
+
for j in finished:
|
|
172
|
+
if len(self._jobs) <= self._max_jobs_kept:
|
|
173
|
+
break
|
|
174
|
+
self._jobs.pop(j.job_id, None)
|
|
175
|
+
self._db.execute("DELETE FROM jobs WHERE job_id=?", (j.job_id,))
|
|
176
|
+
self._db.commit()
|
|
177
|
+
|
|
178
|
+
def _push_log(self, job: Job, message: str) -> None:
|
|
179
|
+
with self._lock:
|
|
180
|
+
job.log_lines.append(message)
|
|
181
|
+
if len(job.log_lines) > self._LOG_LIMIT:
|
|
182
|
+
job.log_lines = job.log_lines[-self._LOG_LIMIT:]
|
|
183
|
+
|
|
184
|
+
def _run(self, job: Job, work: JobWork) -> None:
|
|
185
|
+
job.status = "running"
|
|
186
|
+
job.started_at = datetime.now(timezone.utc)
|
|
187
|
+
self._persist(job)
|
|
188
|
+
try:
|
|
189
|
+
report = work(lambda m: self._push_log(job, m), job.run_path)
|
|
190
|
+
os.makedirs(job.run_path, exist_ok=True)
|
|
191
|
+
with open(os.path.join(job.run_path, "report.json"), "w", encoding="utf-8") as fh:
|
|
192
|
+
fh.write(render_json_report(report))
|
|
193
|
+
with open(os.path.join(job.run_path, "report.html"), "w", encoding="utf-8") as fh:
|
|
194
|
+
fh.write(render_html_report(report, lang=job.report_lang))
|
|
195
|
+
job.summary = _summary_dict(report)
|
|
196
|
+
job.status = "done"
|
|
197
|
+
except Exception as exc: # noqa: BLE001
|
|
198
|
+
job.error = str(exc)
|
|
199
|
+
job.status = "error"
|
|
200
|
+
self._push_log(job, f"! job failed: {exc}")
|
|
201
|
+
finally:
|
|
202
|
+
job.finished_at = datetime.now(timezone.utc)
|
|
203
|
+
self._persist(job)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _summary_dict(report: BatchReport) -> dict:
|
|
207
|
+
return {
|
|
208
|
+
"files": len(report.results),
|
|
209
|
+
"by_status": report.counts,
|
|
210
|
+
"fields_removed": report.fields_removed,
|
|
211
|
+
"files_with_residual": len(report.files_with_residual),
|
|
212
|
+
"files_errored": len(report.errored),
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _now_iso() -> str:
|
|
217
|
+
return datetime.now(timezone.utc).isoformat()
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _iso(dt: datetime | None) -> str | None:
|
|
221
|
+
return dt.isoformat() if dt else None
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _parse(s: str | None) -> datetime | None:
|
|
225
|
+
if not s:
|
|
226
|
+
return None
|
|
227
|
+
try:
|
|
228
|
+
return datetime.fromisoformat(s)
|
|
229
|
+
except ValueError:
|
|
230
|
+
return None
|