meltify 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- meltify/__init__.py +6 -0
- meltify/__main__.py +5 -0
- meltify/cli.py +99 -0
- meltify/commands/__init__.py +0 -0
- meltify/commands/brief.py +181 -0
- meltify/commands/check.py +305 -0
- meltify/commands/doctor.py +255 -0
- meltify/commands/hidden.py +71 -0
- meltify/commands/media.py +230 -0
- meltify/commands/ocr.py +206 -0
- meltify/commands/read.py +134 -0
- meltify/commands/submit.py +344 -0
- meltify/config.py +140 -0
- meltify/converters/__init__.py +82 -0
- meltify/converters/mail.py +49 -0
- meltify/converters/office.py +17 -0
- meltify/converters/pdf.py +32 -0
- meltify/converters/sheet.py +53 -0
- meltify/converters/text.py +42 -0
- meltify/data/__init__.py +0 -0
- meltify/data/brief_en.md +21 -0
- meltify/data/brief_ko.md +21 -0
- meltify/data/defaults.toml +61 -0
- meltify/engines/__init__.py +0 -0
- meltify/engines/asr.py +148 -0
- meltify/engines/consensus.py +60 -0
- meltify/engines/ocr.py +177 -0
- meltify/evidence.py +114 -0
- meltify/ffmpeg.py +114 -0
- meltify/files.py +90 -0
- meltify/forensics/__init__.py +0 -0
- meltify/forensics/pdf.py +120 -0
- meltify/imaging.py +94 -0
- meltify/lang.py +36 -0
- meltify/llm.py +117 -0
- meltify/output.py +90 -0
- meltify/safe.py +56 -0
- meltify-0.1.0.dist-info/METADATA +142 -0
- meltify-0.1.0.dist-info/RECORD +42 -0
- meltify-0.1.0.dist-info/WHEEL +4 -0
- meltify-0.1.0.dist-info/entry_points.txt +3 -0
- meltify-0.1.0.dist-info/licenses/LICENSE +21 -0
meltify/__init__.py
ADDED
meltify/__main__.py
ADDED
meltify/cli.py
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import importlib
|
|
5
|
+
import sys
|
|
6
|
+
from collections.abc import Sequence
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from types import ModuleType
|
|
9
|
+
|
|
10
|
+
from meltify import __version__, config
|
|
11
|
+
from meltify.evidence import MISSING, USAGE, Envelope
|
|
12
|
+
from meltify.output import emit
|
|
13
|
+
from meltify.safe import MissingTool
|
|
14
|
+
|
|
15
|
+
# Each module exposes NAME, HELP, COLUMNS, add_arguments and run.
|
|
16
|
+
# Heavy imports live inside run, so --help and --version start fast
|
|
17
|
+
COMMANDS: list[str] = ["read", "ocr", "hidden", "media", "submit", "check", "brief", "doctor"]
|
|
18
|
+
|
|
19
|
+
USAGE_ERROR = 2
|
|
20
|
+
|
|
21
|
+
EXIT_CODES = """exit codes:
|
|
22
|
+
0 success
|
|
23
|
+
1 violations, failed checks or other errors
|
|
24
|
+
2 usage error or bad input
|
|
25
|
+
3 a needed engine, key, binary or base module is missing"""
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _modules() -> list[ModuleType]:
|
|
29
|
+
return [importlib.import_module(f"meltify.commands.{name}") for name in COMMANDS]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _limit(raw: str) -> int:
|
|
33
|
+
n = int(raw)
|
|
34
|
+
if n < 0:
|
|
35
|
+
raise argparse.ArgumentTypeError(f"must be 0 or more, got {n}")
|
|
36
|
+
return n
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def build_parser(modules: Sequence[ModuleType]) -> argparse.ArgumentParser:
|
|
40
|
+
common = argparse.ArgumentParser(add_help=False)
|
|
41
|
+
common.add_argument("--json", action="store_true", help="print the full result envelope")
|
|
42
|
+
common.add_argument("--out", type=Path, help="output directory for artifacts")
|
|
43
|
+
common.add_argument("--config", type=Path, help="project config file instead of meltify.toml")
|
|
44
|
+
common.add_argument(
|
|
45
|
+
"--limit", type=_limit, default=50, help="rows shown in the table, 0 for all"
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
parser = argparse.ArgumentParser(
|
|
49
|
+
prog="meltify",
|
|
50
|
+
description="Melt files of any format into prompt-ready text, with a citation per item",
|
|
51
|
+
epilog=EXIT_CODES,
|
|
52
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
53
|
+
)
|
|
54
|
+
parser.add_argument("--version", action="version", version=f"meltify {__version__}")
|
|
55
|
+
sub = parser.add_subparsers(dest="command", metavar="COMMAND")
|
|
56
|
+
for m in modules:
|
|
57
|
+
p = sub.add_parser(
|
|
58
|
+
m.NAME,
|
|
59
|
+
help=m.HELP,
|
|
60
|
+
description=m.HELP,
|
|
61
|
+
epilog=EXIT_CODES,
|
|
62
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
63
|
+
parents=[common],
|
|
64
|
+
)
|
|
65
|
+
m.add_arguments(p)
|
|
66
|
+
p.set_defaults(_module=m)
|
|
67
|
+
return parser
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
71
|
+
modules = _modules()
|
|
72
|
+
parser = build_parser(modules)
|
|
73
|
+
args = parser.parse_args(argv)
|
|
74
|
+
if not getattr(args, "_module", None):
|
|
75
|
+
parser.print_help(sys.stderr)
|
|
76
|
+
return USAGE_ERROR
|
|
77
|
+
module = args._module
|
|
78
|
+
|
|
79
|
+
overrides = {"out_dir": str(args.out)} if args.out is not None else None
|
|
80
|
+
try:
|
|
81
|
+
settings = config.load(overrides, project=args.config)
|
|
82
|
+
except config.ConfigError as e:
|
|
83
|
+
# Report it in the envelope, so --json callers still get valid JSON
|
|
84
|
+
env = Envelope(command=module.NAME, version=__version__)
|
|
85
|
+
env.error(USAGE, f"config: {e}")
|
|
86
|
+
emit(env, as_json=args.json, columns=module.COLUMNS, limit=args.limit)
|
|
87
|
+
return env.exit_code
|
|
88
|
+
|
|
89
|
+
try:
|
|
90
|
+
env = module.run(args, settings)
|
|
91
|
+
except MissingTool as e:
|
|
92
|
+
env = Envelope(command=module.NAME, version=__version__)
|
|
93
|
+
env.error(MISSING, str(e), e.hint)
|
|
94
|
+
except (FileNotFoundError, IsADirectoryError, ValueError) as e:
|
|
95
|
+
# Report bad input in the envelope, so --json callers still get valid JSON
|
|
96
|
+
env = Envelope(command=module.NAME, version=__version__)
|
|
97
|
+
env.error(USAGE, f"{type(e).__name__}: {e}")
|
|
98
|
+
emit(env, as_json=args.json, columns=env.columns or module.COLUMNS, limit=args.limit)
|
|
99
|
+
return env.exit_code
|
|
File without changes
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import json
|
|
5
|
+
import re
|
|
6
|
+
import time
|
|
7
|
+
from importlib.resources import files
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from meltify import __version__
|
|
12
|
+
from meltify.evidence import USAGE, Envelope, Src, finding
|
|
13
|
+
|
|
14
|
+
NAME = "brief"
|
|
15
|
+
HELP = "quote a problem statement's conditions with line numbers, and track several problems"
|
|
16
|
+
COLUMNS = ["kinds", "quote", "cite"]
|
|
17
|
+
BOARD_COLUMNS = ["problem", "state", "points", "checks", "idle_min", "answer"]
|
|
18
|
+
|
|
19
|
+
# Words that mark a sentence a solver shouldn't paraphrase, in English and Korean
|
|
20
|
+
KINDS: dict[str, re.Pattern[str]] = {
|
|
21
|
+
"count": re.compile(
|
|
22
|
+
r"\d+\s*(연승|연속|회|번|times?|in a row|consecutive|wins?|attempts?|items?|개|명|건)"
|
|
23
|
+
r"|at least|at most|exactly|최소|최대|이상|이하|미만|초과",
|
|
24
|
+
re.I,
|
|
25
|
+
),
|
|
26
|
+
"date": re.compile(r"\d{4}[-./]\d{1,2}[-./]\d{1,2}|기준일|as of|deadline|마감|기한|due", re.I),
|
|
27
|
+
"timezone": re.compile(r"\b(KST|UTC|GMT|PST|EST)\b|[+-]\d{2}:\d{2}|시간대|time ?zone", re.I),
|
|
28
|
+
"format": re.compile(
|
|
29
|
+
r"JSON|CSV|대문자|소문자|upper ?case|lower ?case|숫자만|digits? only|단어|words?\b"
|
|
30
|
+
r"|형식|format|쉼표|comma|소수점|decimal|템플릿|template|띄어쓰기|spaces?\b",
|
|
31
|
+
re.I,
|
|
32
|
+
),
|
|
33
|
+
"priority": re.compile(
|
|
34
|
+
r"우선|priority|lowest|highest|가장 낮은|가장 높은|먼저|first"
|
|
35
|
+
r"|여러 개|multiple|if more than",
|
|
36
|
+
re.I,
|
|
37
|
+
),
|
|
38
|
+
"limit": re.compile(
|
|
39
|
+
r"per (minute|second|hour)|분당|초당|\d+\s*분에|rate limit|제한"
|
|
40
|
+
r"|한 번만|only once|duplicate|중복",
|
|
41
|
+
re.I,
|
|
42
|
+
),
|
|
43
|
+
"prohibition": re.compile(
|
|
44
|
+
r"금지|must not|do not|don't|never|않|없이|추측|guess|assume|불가", re.I
|
|
45
|
+
),
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def add_arguments(p: argparse.ArgumentParser) -> None:
|
|
50
|
+
p.add_argument("action", choices=["extract", "new", "board", "set"])
|
|
51
|
+
p.add_argument("target", nargs="?", help="statement file for extract, number for new and set")
|
|
52
|
+
p.add_argument("title", nargs="?", help="title for new")
|
|
53
|
+
p.add_argument("--from", dest="source", type=Path, help="statement file for new")
|
|
54
|
+
p.add_argument("--state", help="state for set, like todo, doing, done or skipped")
|
|
55
|
+
p.add_argument("--points", type=float, help="points for set")
|
|
56
|
+
p.add_argument("--answer", help="answer note for set")
|
|
57
|
+
p.add_argument("--force", action="store_true", help="rewrite an existing problem's NOTES.md")
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def extract(text: str, origin: str) -> list[dict[str, Any]]:
|
|
61
|
+
rows = []
|
|
62
|
+
for n, line in enumerate(text.splitlines(), start=1):
|
|
63
|
+
quote = line.strip()
|
|
64
|
+
if not quote:
|
|
65
|
+
continue
|
|
66
|
+
kinds = [k for k, rx in KINDS.items() if rx.search(quote)]
|
|
67
|
+
if kinds:
|
|
68
|
+
rows.append(finding(Src(origin, line=n), kinds=kinds, quote=quote))
|
|
69
|
+
return rows
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _slug(title: str) -> str:
|
|
73
|
+
return re.sub(r"[^\w]+", "-", title).strip("-")[:40] or "problem"
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _problem_dirs(root: Path, number: str) -> list[Path]:
|
|
77
|
+
prefix = f"{int(number):02d}-"
|
|
78
|
+
return [d for d in sorted(root.glob(f"{prefix}*")) if d.is_dir()]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _new(d: Path, number: str, title: str, source: Path | None, body: str) -> None:
|
|
82
|
+
statement = source.read_text("utf-8") if source else ""
|
|
83
|
+
conditions = extract(statement, str(source)) if source else []
|
|
84
|
+
listed = "\n".join(f"- [ ] {r['quote']} `{r['cite']}`" for r in conditions)
|
|
85
|
+
notes = body.format(
|
|
86
|
+
number=number, title=title, statement=statement.strip(), conditions=listed or "-"
|
|
87
|
+
)
|
|
88
|
+
(d / "data").mkdir(parents=True, exist_ok=True)
|
|
89
|
+
(d / "NOTES.md").write_text(notes, encoding="utf-8")
|
|
90
|
+
status = d / "status.json"
|
|
91
|
+
if not status.exists():
|
|
92
|
+
status.write_text(json.dumps({"state": "todo", "points": 0, "answer": None}), "utf-8")
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _board(dirs: list[Path], now: float) -> list[dict[str, Any]]:
|
|
96
|
+
rows = []
|
|
97
|
+
for d in dirs:
|
|
98
|
+
status = json.loads((d / "status.json").read_text("utf-8"))
|
|
99
|
+
notes = (d / "NOTES.md").read_text("utf-8")
|
|
100
|
+
done, todo = notes.count("- [x]"), notes.count("- [ ]")
|
|
101
|
+
newest = max(f.stat().st_mtime for f in d.rglob("*") if f.is_file())
|
|
102
|
+
rows.append(
|
|
103
|
+
finding(
|
|
104
|
+
Src(str(d / "NOTES.md")),
|
|
105
|
+
problem=d.name,
|
|
106
|
+
state=status.get("state"),
|
|
107
|
+
points=status.get("points"),
|
|
108
|
+
checks=f"{done}/{done + todo}",
|
|
109
|
+
idle_min=int((now - newest) / 60),
|
|
110
|
+
answer=status.get("answer"),
|
|
111
|
+
)
|
|
112
|
+
)
|
|
113
|
+
return rows
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def run(args: argparse.Namespace, settings: dict[str, Any]) -> Envelope:
|
|
117
|
+
env = Envelope(command=NAME, version=__version__)
|
|
118
|
+
conf = settings.get("brief", {})
|
|
119
|
+
root = Path(conf.get("root", "problems"))
|
|
120
|
+
|
|
121
|
+
if args.action == "extract":
|
|
122
|
+
if not args.target:
|
|
123
|
+
env.error(USAGE, "extract needs a statement file")
|
|
124
|
+
return env
|
|
125
|
+
path = Path(args.target)
|
|
126
|
+
env.inputs.append({"path": str(path)})
|
|
127
|
+
env.results = extract(path.read_text("utf-8"), str(path))
|
|
128
|
+
env.summary = (
|
|
129
|
+
f"{len(env.results)} condition lines. Quote them as written and check each one"
|
|
130
|
+
)
|
|
131
|
+
return env
|
|
132
|
+
|
|
133
|
+
if args.action == "new":
|
|
134
|
+
if not (args.target and args.title):
|
|
135
|
+
env.error(USAGE, "new needs a number and a title")
|
|
136
|
+
return env
|
|
137
|
+
existing = _problem_dirs(root, args.target)
|
|
138
|
+
if existing and not args.force:
|
|
139
|
+
env.error(
|
|
140
|
+
USAGE,
|
|
141
|
+
f"problem {args.target} already exists at {existing[0]}, add --force to rewrite it",
|
|
142
|
+
)
|
|
143
|
+
return env
|
|
144
|
+
if len(existing) > 1:
|
|
145
|
+
env.error(USAGE, f"problem {args.target} matches {len(existing)} folders under {root}")
|
|
146
|
+
return env
|
|
147
|
+
template = files("meltify.data").joinpath(f"brief_{conf.get('template', 'en')}.md")
|
|
148
|
+
if not template.is_file():
|
|
149
|
+
env.error(USAGE, f"no brief template {conf.get('template')!r}, use en or ko")
|
|
150
|
+
return env
|
|
151
|
+
d = existing[0] if existing else root / f"{int(args.target):02d}-{_slug(args.title)}"
|
|
152
|
+
_new(d, args.target, args.title, args.source, template.read_text("utf-8"))
|
|
153
|
+
env.artifact(str(d / "NOTES.md"), "notes")
|
|
154
|
+
env.summary = f"{'rewrote' if existing else 'created'} {d}"
|
|
155
|
+
return env
|
|
156
|
+
|
|
157
|
+
if args.action == "set":
|
|
158
|
+
found = _problem_dirs(root, args.target) if args.target else []
|
|
159
|
+
if len(found) != 1:
|
|
160
|
+
what = "no problem" if not found else f"{len(found)} folders for problem"
|
|
161
|
+
env.error(USAGE, f"{what} {args.target} under {root}")
|
|
162
|
+
return env
|
|
163
|
+
status_path = found[0] / "status.json"
|
|
164
|
+
status = json.loads(status_path.read_text("utf-8"))
|
|
165
|
+
for key in ("state", "points", "answer"):
|
|
166
|
+
if (value := getattr(args, key)) is not None:
|
|
167
|
+
status[key] = value
|
|
168
|
+
status_path.write_text(json.dumps(status, ensure_ascii=False), "utf-8")
|
|
169
|
+
|
|
170
|
+
dirs = sorted(p for p in root.iterdir() if p.is_dir()) if root.exists() else []
|
|
171
|
+
complete = []
|
|
172
|
+
for d in dirs:
|
|
173
|
+
missing = [n for n in ("status.json", "NOTES.md") if not (d / n).is_file()]
|
|
174
|
+
if missing:
|
|
175
|
+
env.warnings.append(f"skipped {d}: no {' and no '.join(missing)}")
|
|
176
|
+
else:
|
|
177
|
+
complete.append(d)
|
|
178
|
+
env.results = _board(complete, time.time())
|
|
179
|
+
env.columns = BOARD_COLUMNS
|
|
180
|
+
env.summary = f"{len(env.results)} problems under {root}"
|
|
181
|
+
return env
|
|
@@ -0,0 +1,305 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import csv
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
import unicodedata
|
|
8
|
+
from collections.abc import Iterator
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from meltify import __version__
|
|
14
|
+
from meltify.evidence import USAGE, Envelope, Src, finding
|
|
15
|
+
|
|
16
|
+
NAME = "check"
|
|
17
|
+
HELP = "check an answer file or string against its required output format before you hand it in"
|
|
18
|
+
COLUMNS = ["rule", "message", "value", "cite"]
|
|
19
|
+
VIOLATION = "violation"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def add_arguments(p: argparse.ArgumentParser) -> None:
|
|
23
|
+
p.add_argument("file", nargs="?", type=Path, help="JSON, JSONL or CSV answer file")
|
|
24
|
+
p.add_argument("--text", help="check one string instead of a file")
|
|
25
|
+
p.add_argument("--schema", type=Path, help="JSON Schema file")
|
|
26
|
+
p.add_argument("--count", type=int, help="exact number of top-level items")
|
|
27
|
+
p.add_argument("--unique", action="append", help="key that can't repeat across items")
|
|
28
|
+
p.add_argument(
|
|
29
|
+
"--field",
|
|
30
|
+
action="append",
|
|
31
|
+
default=[],
|
|
32
|
+
metavar="PATH",
|
|
33
|
+
help="JSONPath the text rules apply to, like $[*].id (repeatable)",
|
|
34
|
+
)
|
|
35
|
+
p.add_argument("--pattern", help="regex the whole value must match")
|
|
36
|
+
p.add_argument("--words", type=int, help="exact word count")
|
|
37
|
+
p.add_argument("--max-words", type=int, help="maximum word count")
|
|
38
|
+
p.add_argument("--upper", action="store_true", help="no lowercase letters")
|
|
39
|
+
p.add_argument("--digits", action="store_true", help="ASCII digits only")
|
|
40
|
+
p.add_argument("--include", action="append", help="phrase the value must contain")
|
|
41
|
+
p.add_argument("--no-hygiene", action="store_true", help="skip whitespace and width checks")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass(frozen=True)
|
|
45
|
+
class Rule:
|
|
46
|
+
"""Text format rules for the values at one JSONPath"""
|
|
47
|
+
|
|
48
|
+
path: str = "$"
|
|
49
|
+
pattern: str | None = None
|
|
50
|
+
words: int | None = None
|
|
51
|
+
max_words: int | None = None
|
|
52
|
+
upper: bool = False
|
|
53
|
+
digits: bool = False
|
|
54
|
+
include: tuple[str, ...] = ()
|
|
55
|
+
|
|
56
|
+
@classmethod
|
|
57
|
+
def from_config(cls, raw: dict[str, Any]) -> Rule:
|
|
58
|
+
return cls(
|
|
59
|
+
path=raw.get("path", "$"),
|
|
60
|
+
pattern=raw.get("pattern"),
|
|
61
|
+
words=raw.get("words"),
|
|
62
|
+
max_words=raw.get("max_words"),
|
|
63
|
+
upper=bool(raw.get("upper", False)),
|
|
64
|
+
digits=bool(raw.get("digits", False)),
|
|
65
|
+
include=tuple(raw.get("must_include", raw.get("include", ()))),
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
def problems(self, value: Any) -> Iterator[tuple[str, str]]:
|
|
69
|
+
if not isinstance(value, str):
|
|
70
|
+
yield "type", f"expected a string, got {type(value).__name__}"
|
|
71
|
+
return
|
|
72
|
+
# \d also matches full-width digits, so spell out the ASCII class
|
|
73
|
+
if self.digits and not re.fullmatch(r"[0-9]+", value):
|
|
74
|
+
yield "digits", "ASCII digits only"
|
|
75
|
+
if self.upper and value != value.upper():
|
|
76
|
+
yield "upper", "contains lowercase letters"
|
|
77
|
+
if self.pattern and not re.fullmatch(self.pattern, value):
|
|
78
|
+
yield "pattern", f"doesn't fully match {self.pattern}"
|
|
79
|
+
n = len(value.split())
|
|
80
|
+
if self.words is not None and n != self.words:
|
|
81
|
+
yield "words", f"{n} words, expected {self.words}"
|
|
82
|
+
if self.max_words is not None and n > self.max_words:
|
|
83
|
+
yield "max_words", f"{n} words, at most {self.max_words}"
|
|
84
|
+
for phrase in self.include:
|
|
85
|
+
if phrase not in value:
|
|
86
|
+
yield "include", f"missing {phrase!r}"
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def hygiene(value: str) -> Iterator[tuple[str, str]]:
|
|
90
|
+
if value != value.strip():
|
|
91
|
+
yield "hygiene", "leading or trailing whitespace"
|
|
92
|
+
if " " in value:
|
|
93
|
+
yield "hygiene", "double space"
|
|
94
|
+
if unicodedata.normalize("NFKC", value) != value:
|
|
95
|
+
yield "hygiene", "full-width or compatibility characters"
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
PATH_TOKEN = re.compile(r"\.([^.\[\]]+)|\[(\*|-?\d+)\]")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def path_tokens(path: str) -> list[tuple[str, str]]:
|
|
102
|
+
"""Minimal JSONPath: dot keys, [n] and [*]"""
|
|
103
|
+
body = path.removeprefix("$")
|
|
104
|
+
found = list(PATH_TOKEN.finditer(body))
|
|
105
|
+
# Reject syntax like filters, which would otherwise select nothing and look like missing data
|
|
106
|
+
if not path.startswith("$") or "".join(m.group(0) for m in found) != body:
|
|
107
|
+
raise ValueError(path)
|
|
108
|
+
return [(m.group(1) or "", m.group(2) or "") for m in found]
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def select(data: Any, path: str) -> Iterator[tuple[str, Any]]:
|
|
112
|
+
nodes: list[tuple[str, Any]] = [("$", data)]
|
|
113
|
+
for key, index in path_tokens(path):
|
|
114
|
+
nxt = []
|
|
115
|
+
for p, node in nodes:
|
|
116
|
+
if key and isinstance(node, dict) and key in node:
|
|
117
|
+
nxt.append((f"{p}.{key}", node[key]))
|
|
118
|
+
elif index == "*" and isinstance(node, list):
|
|
119
|
+
nxt.extend((f"{p}[{i}]", v) for i, v in enumerate(node))
|
|
120
|
+
elif index == "*" and isinstance(node, dict):
|
|
121
|
+
nxt.extend((f"{p}.{k}", v) for k, v in node.items())
|
|
122
|
+
elif (
|
|
123
|
+
index
|
|
124
|
+
and index != "*"
|
|
125
|
+
and isinstance(node, list)
|
|
126
|
+
and -len(node) <= int(index) < len(node)
|
|
127
|
+
):
|
|
128
|
+
nxt.append((f"{p}[{index}]", node[int(index)]))
|
|
129
|
+
nodes = nxt
|
|
130
|
+
return iter(nodes)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def strings(data: Any, path: str = "$") -> Iterator[tuple[str, str]]:
|
|
134
|
+
if isinstance(data, str):
|
|
135
|
+
yield path, data
|
|
136
|
+
elif isinstance(data, dict):
|
|
137
|
+
for k, v in data.items():
|
|
138
|
+
yield from strings(v, f"{path}.{k}")
|
|
139
|
+
elif isinstance(data, list):
|
|
140
|
+
for i, v in enumerate(data):
|
|
141
|
+
yield from strings(v, f"{path}[{i}]")
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def load(path: Path) -> Any:
|
|
145
|
+
suffix = path.suffix.lower()
|
|
146
|
+
text = path.read_text("utf-8-sig")
|
|
147
|
+
if suffix == ".csv":
|
|
148
|
+
return list(csv.DictReader(text.splitlines()))
|
|
149
|
+
if suffix == ".jsonl":
|
|
150
|
+
return [json.loads(line) for line in text.splitlines() if line.strip()]
|
|
151
|
+
return json.loads(text)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
@dataclass
|
|
155
|
+
class Checker:
|
|
156
|
+
source: str
|
|
157
|
+
rules: list[Rule] = field(default_factory=list)
|
|
158
|
+
schema: dict[str, Any] | None = None
|
|
159
|
+
count: int | None = None
|
|
160
|
+
unique: list[str] = field(default_factory=list)
|
|
161
|
+
check_hygiene: bool = True
|
|
162
|
+
|
|
163
|
+
def run(self, data: Any) -> list[dict[str, Any]]:
|
|
164
|
+
found: list[dict[str, Any]] = []
|
|
165
|
+
|
|
166
|
+
def add(rule: str, message: str, jpath: str, value: Any = None) -> None:
|
|
167
|
+
found.append(
|
|
168
|
+
finding(Src(self.source, jpath=jpath), rule=rule, message=message, value=value)
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
if self.schema is not None:
|
|
172
|
+
from jsonschema import Draft202012Validator
|
|
173
|
+
|
|
174
|
+
for e in Draft202012Validator(self.schema).iter_errors(data):
|
|
175
|
+
add("schema", e.message, e.json_path, e.instance if _scalar(e.instance) else None)
|
|
176
|
+
if not isinstance(data, list):
|
|
177
|
+
# count and unique describe a list of items, so any other shape fails them
|
|
178
|
+
got = type(data).__name__
|
|
179
|
+
if self.count is not None:
|
|
180
|
+
add("count", f"expected a list of items, got {got}", "$")
|
|
181
|
+
if self.unique:
|
|
182
|
+
add("unique", f"expected a list of items, got {got}", "$")
|
|
183
|
+
elif self.count is not None and len(data) != self.count:
|
|
184
|
+
add("count", f"{len(data)} items, expected {self.count}", "$", len(data))
|
|
185
|
+
for key in self.unique:
|
|
186
|
+
# Include the type in the key, since 1 == True in Python
|
|
187
|
+
seen: dict[tuple[type, Any], int] = {}
|
|
188
|
+
for i, item in enumerate(data if isinstance(data, list) else []):
|
|
189
|
+
if not isinstance(item, dict) or key not in item or not _scalar(item[key]):
|
|
190
|
+
continue
|
|
191
|
+
v = item[key]
|
|
192
|
+
if (type(v), v) in seen:
|
|
193
|
+
add("unique", f"{key} repeats item {seen[type(v), v]}", f"$[{i}].{key}", v)
|
|
194
|
+
continue
|
|
195
|
+
seen[type(v), v] = i
|
|
196
|
+
if self.check_hygiene:
|
|
197
|
+
for jpath, s in strings(data):
|
|
198
|
+
for rule, message in hygiene(s):
|
|
199
|
+
add(rule, message, jpath, s)
|
|
200
|
+
for r in self.rules:
|
|
201
|
+
matched = list(select(data, r.path))
|
|
202
|
+
if not matched:
|
|
203
|
+
add("path", f"nothing at {r.path}", r.path)
|
|
204
|
+
for jpath, value in matched:
|
|
205
|
+
for rule, message in r.problems(value):
|
|
206
|
+
add(rule, message, jpath, value if _scalar(value) else None)
|
|
207
|
+
return found
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _scalar(v: Any) -> bool:
|
|
211
|
+
return isinstance(v, str | int | float | bool) or v is None
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _has_text_rule(args: argparse.Namespace) -> bool:
|
|
215
|
+
return any(
|
|
216
|
+
[
|
|
217
|
+
args.pattern,
|
|
218
|
+
args.words is not None,
|
|
219
|
+
args.max_words is not None,
|
|
220
|
+
args.upper,
|
|
221
|
+
args.digits,
|
|
222
|
+
args.include,
|
|
223
|
+
]
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _cli_rule(args: argparse.Namespace, path: str) -> Rule:
|
|
228
|
+
return Rule(
|
|
229
|
+
path=path,
|
|
230
|
+
pattern=args.pattern,
|
|
231
|
+
words=args.words,
|
|
232
|
+
max_words=args.max_words,
|
|
233
|
+
upper=args.upper,
|
|
234
|
+
digits=args.digits,
|
|
235
|
+
include=tuple(args.include or ()),
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def run(args: argparse.Namespace, settings: dict[str, Any]) -> Envelope:
|
|
240
|
+
env = Envelope(command=NAME, version=__version__)
|
|
241
|
+
conf = settings.get("check", {})
|
|
242
|
+
if (args.file is None) == (args.text is None):
|
|
243
|
+
env.error(USAGE, "give exactly one of FILE or --text")
|
|
244
|
+
return env
|
|
245
|
+
|
|
246
|
+
hygiene_on = conf.get("hygiene", True) and not args.no_hygiene
|
|
247
|
+
if args.field and not _has_text_rule(args):
|
|
248
|
+
env.error(USAGE, "--field needs a text rule, like --pattern or --upper")
|
|
249
|
+
return env
|
|
250
|
+
if args.text is not None:
|
|
251
|
+
ignored = [
|
|
252
|
+
flag
|
|
253
|
+
for flag, value in (
|
|
254
|
+
("--schema", args.schema),
|
|
255
|
+
("--count", args.count),
|
|
256
|
+
("--unique", args.unique),
|
|
257
|
+
("--field", args.field),
|
|
258
|
+
)
|
|
259
|
+
if value not in (None, [])
|
|
260
|
+
]
|
|
261
|
+
if ignored:
|
|
262
|
+
env.error(USAGE, f"{', '.join(ignored)} apply to a FILE, not --text")
|
|
263
|
+
return env
|
|
264
|
+
# Project rules describe an answer file, so a single string uses only its own flags
|
|
265
|
+
rules = [_cli_rule(args, "$")] if _has_text_rule(args) else []
|
|
266
|
+
else:
|
|
267
|
+
if _has_text_rule(args) and not args.field:
|
|
268
|
+
env.error(USAGE, "text rules on a file need --field, like $[*].id or $.q2")
|
|
269
|
+
return env
|
|
270
|
+
rules = [Rule.from_config(r) for r in conf.get("field", [])]
|
|
271
|
+
if _has_text_rule(args):
|
|
272
|
+
rules += [_cli_rule(args, p) for p in args.field]
|
|
273
|
+
for r in rules:
|
|
274
|
+
try:
|
|
275
|
+
path_tokens(r.path)
|
|
276
|
+
except ValueError:
|
|
277
|
+
env.error(USAGE, f"unsupported path {r.path}, use $ with .key, [n], [-1] and [*]")
|
|
278
|
+
return env
|
|
279
|
+
try:
|
|
280
|
+
re.compile(r.pattern or "")
|
|
281
|
+
except re.error as e:
|
|
282
|
+
env.error(USAGE, f"bad pattern {r.pattern!r}: {e}")
|
|
283
|
+
return env
|
|
284
|
+
if args.text is not None:
|
|
285
|
+
checker = Checker(source="<text>", rules=rules, check_hygiene=hygiene_on)
|
|
286
|
+
data: Any = args.text
|
|
287
|
+
else:
|
|
288
|
+
schema_path = args.schema or (Path(conf["schema"]) if conf.get("schema") else None)
|
|
289
|
+
checker = Checker(
|
|
290
|
+
source=str(args.file),
|
|
291
|
+
rules=rules,
|
|
292
|
+
schema=json.loads(schema_path.read_text("utf-8")) if schema_path else None,
|
|
293
|
+
count=args.count if args.count is not None else conf.get("count"),
|
|
294
|
+
unique=args.unique or list(conf.get("unique", [])),
|
|
295
|
+
check_hygiene=hygiene_on,
|
|
296
|
+
)
|
|
297
|
+
env.inputs.append({"path": str(args.file)})
|
|
298
|
+
data = load(args.file)
|
|
299
|
+
source = checker.source
|
|
300
|
+
env.results = checker.run(data)
|
|
301
|
+
if env.results:
|
|
302
|
+
env.error(VIOLATION, f"{len(env.results)} violations in {source}")
|
|
303
|
+
else:
|
|
304
|
+
env.summary = f"no violations in {source}"
|
|
305
|
+
return env
|