jevkit-lint 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- jevkit_lint/__init__.py +14 -0
- jevkit_lint/cli.py +140 -0
- jevkit_lint/diagnostic.py +57 -0
- jevkit_lint/linter.py +92 -0
- jevkit_lint/rules.py +595 -0
- jevkit_lint-0.1.0.dist-info/METADATA +135 -0
- jevkit_lint-0.1.0.dist-info/RECORD +9 -0
- jevkit_lint-0.1.0.dist-info/WHEEL +4 -0
- jevkit_lint-0.1.0.dist-info/entry_points.txt +2 -0
jevkit_lint/__init__.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Static linter for TypeSafe Jev questions.
|
|
2
|
+
|
|
3
|
+
Catches the failure modes TypeSafe documents for jev-1.13 before you spend a
|
|
4
|
+
token on them. Needs no API key.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from .diagnostic import Diagnostic, Severity
|
|
8
|
+
from .linter import LintResult, lint
|
|
9
|
+
from .rules import RULES, Rule, all_codes
|
|
10
|
+
|
|
11
|
+
__version__ = "0.1.0"
|
|
12
|
+
|
|
13
|
+
__all__ = ["lint", "LintResult", "Diagnostic", "Severity", "Rule", "RULES",
|
|
14
|
+
"all_codes", "__version__"]
|
jevkit_lint/cli.py
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""``jevkit-lint`` command line interface."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import sys
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from jevkit_core import RecordFormatError, read_records
|
|
11
|
+
|
|
12
|
+
from .diagnostic import Severity
|
|
13
|
+
from .linter import lint
|
|
14
|
+
from .rules import RULES
|
|
15
|
+
|
|
16
|
+
EXIT_OK = 0
|
|
17
|
+
EXIT_FINDINGS = 1
|
|
18
|
+
EXIT_USAGE = 2
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _load_payload(path: str) -> list[tuple[str, Any, dict[str, Any]]]:
|
|
22
|
+
"""Return (label, state, questions) triples from a file.
|
|
23
|
+
|
|
24
|
+
Accepts a `.jevl` record file, a request object with `state` and
|
|
25
|
+
`questions`, or a bare questions object.
|
|
26
|
+
"""
|
|
27
|
+
if path == "-":
|
|
28
|
+
text = sys.stdin.read()
|
|
29
|
+
source = "<stdin>"
|
|
30
|
+
else:
|
|
31
|
+
with open(path, "r", encoding="utf-8") as fh:
|
|
32
|
+
text = fh.read()
|
|
33
|
+
source = path
|
|
34
|
+
|
|
35
|
+
if path.endswith(".jevl"):
|
|
36
|
+
import io
|
|
37
|
+
out = []
|
|
38
|
+
for i, record in enumerate(read_records(io.StringIO(text))):
|
|
39
|
+
out.append((f"{source}#{i}", record.state, record.questions))
|
|
40
|
+
return out
|
|
41
|
+
|
|
42
|
+
data = json.loads(text)
|
|
43
|
+
if not isinstance(data, dict):
|
|
44
|
+
raise ValueError(f"{source}: expected a JSON object at the top level")
|
|
45
|
+
|
|
46
|
+
if "questions" in data and isinstance(data["questions"], dict):
|
|
47
|
+
return [(source, data.get("state", ""), data["questions"])]
|
|
48
|
+
return [(source, "", data)]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _print_rules() -> None:
|
|
52
|
+
width = max(len(r.name) for r in RULES)
|
|
53
|
+
print("code name" + " " * (width - 4) + " documented failure mode")
|
|
54
|
+
print("-" * (8 + width + 2 + 40))
|
|
55
|
+
for rule in RULES:
|
|
56
|
+
print(f"{rule.code} {rule.name:<{width}} {rule.mode}")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def main(argv: list[str] | None = None) -> int:
|
|
60
|
+
parser = argparse.ArgumentParser(
|
|
61
|
+
prog="jevkit-lint",
|
|
62
|
+
description="Statically lint TypeSafe Jev questions against the documented "
|
|
63
|
+
"jev-1.13 failure modes. Never calls the API.",
|
|
64
|
+
)
|
|
65
|
+
parser.add_argument("files", nargs="*",
|
|
66
|
+
help="JSON request files, .jevl record files, or - for stdin")
|
|
67
|
+
parser.add_argument("--select", metavar="CODES",
|
|
68
|
+
help="comma-separated rule codes to run exclusively")
|
|
69
|
+
parser.add_argument("--ignore", metavar="CODES",
|
|
70
|
+
help="comma-separated rule codes to skip")
|
|
71
|
+
parser.add_argument("--format", choices=("text", "json"), default="text")
|
|
72
|
+
parser.add_argument("--strict", action="store_true",
|
|
73
|
+
help="exit non-zero on warnings too, not just errors")
|
|
74
|
+
parser.add_argument("--list-rules", action="store_true", help="list rules and exit")
|
|
75
|
+
parser.add_argument("--no-color", action="store_true")
|
|
76
|
+
args = parser.parse_args(argv)
|
|
77
|
+
|
|
78
|
+
if args.list_rules:
|
|
79
|
+
_print_rules()
|
|
80
|
+
return EXIT_OK
|
|
81
|
+
|
|
82
|
+
if not args.files:
|
|
83
|
+
parser.error("no input files (use - to read stdin, or --list-rules)")
|
|
84
|
+
|
|
85
|
+
select = args.select.split(",") if args.select else None
|
|
86
|
+
ignore = args.ignore.split(",") if args.ignore else None
|
|
87
|
+
color = sys.stdout.isatty() and not args.no_color
|
|
88
|
+
|
|
89
|
+
payloads: list[tuple[str, Any, dict[str, Any]]] = []
|
|
90
|
+
for path in args.files:
|
|
91
|
+
try:
|
|
92
|
+
payloads.extend(_load_payload(path))
|
|
93
|
+
except (OSError, ValueError, RecordFormatError) as exc:
|
|
94
|
+
print(f"jevkit-lint: {exc}", file=sys.stderr)
|
|
95
|
+
return EXIT_USAGE
|
|
96
|
+
|
|
97
|
+
reports = []
|
|
98
|
+
worst = Severity.INFO
|
|
99
|
+
any_error = any_warning = False
|
|
100
|
+
|
|
101
|
+
for label, state, questions in payloads:
|
|
102
|
+
try:
|
|
103
|
+
result = lint(questions, state, select=select, ignore=ignore)
|
|
104
|
+
except ValueError as exc:
|
|
105
|
+
print(f"jevkit-lint: {label}: {exc}", file=sys.stderr)
|
|
106
|
+
return EXIT_USAGE
|
|
107
|
+
any_error = any_error or bool(result.errors)
|
|
108
|
+
any_warning = any_warning or bool(result.warnings)
|
|
109
|
+
reports.append((label, result))
|
|
110
|
+
|
|
111
|
+
if args.format == "json":
|
|
112
|
+
print(json.dumps(
|
|
113
|
+
{"results": [{"source": label, **result.to_dict()} for label, result in reports]},
|
|
114
|
+
indent=2,
|
|
115
|
+
))
|
|
116
|
+
else:
|
|
117
|
+
for label, result in reports:
|
|
118
|
+
if len(payloads) > 1:
|
|
119
|
+
print(f"== {label}")
|
|
120
|
+
print(result.format(color=color))
|
|
121
|
+
if len(payloads) > 1:
|
|
122
|
+
print()
|
|
123
|
+
total = {"error": 0, "warning": 0, "info": 0}
|
|
124
|
+
for _, result in reports:
|
|
125
|
+
for key, value in result.counts().items():
|
|
126
|
+
total[key] += value
|
|
127
|
+
if sum(total.values()):
|
|
128
|
+
print(f"\n{total['error']} error(s), {total['warning']} warning(s), "
|
|
129
|
+
f"{total['info']} info")
|
|
130
|
+
|
|
131
|
+
del worst
|
|
132
|
+
if any_error:
|
|
133
|
+
return EXIT_FINDINGS
|
|
134
|
+
if args.strict and any_warning:
|
|
135
|
+
return EXIT_FINDINGS
|
|
136
|
+
return EXIT_OK
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
if __name__ == "__main__": # pragma: no cover
|
|
140
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""Diagnostics produced by the linter."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from enum import Enum
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
__all__ = ["Severity", "Diagnostic"]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Severity(str, Enum):
|
|
13
|
+
ERROR = "error" # Will very likely produce wrong answers or be rejected.
|
|
14
|
+
WARNING = "warning" # A documented failure mode is in play.
|
|
15
|
+
INFO = "info" # Worth a look; may be intentional.
|
|
16
|
+
|
|
17
|
+
def __str__(self) -> str: # pragma: no cover - display only
|
|
18
|
+
return self.value
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
_ORDER = {Severity.ERROR: 0, Severity.WARNING: 1, Severity.INFO: 2}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class Diagnostic:
|
|
26
|
+
code: str
|
|
27
|
+
severity: Severity
|
|
28
|
+
question_id: str | None
|
|
29
|
+
message: str
|
|
30
|
+
hint: str
|
|
31
|
+
evidence: str = ""
|
|
32
|
+
|
|
33
|
+
@property
|
|
34
|
+
def sort_key(self) -> tuple[int, str, str]:
|
|
35
|
+
return (_ORDER[self.severity], self.code, self.question_id or "")
|
|
36
|
+
|
|
37
|
+
def format(self, *, color: bool = False) -> str:
|
|
38
|
+
loc = self.question_id or "<request>"
|
|
39
|
+
head = f"{loc}: {self.severity.value} [{self.code}] {self.message}"
|
|
40
|
+
if color:
|
|
41
|
+
tint = {"error": "\033[31m", "warning": "\033[33m", "info": "\033[36m"}
|
|
42
|
+
head = f"{tint[self.severity.value]}{head}\033[0m"
|
|
43
|
+
lines = [head]
|
|
44
|
+
if self.evidence:
|
|
45
|
+
lines.append(f" found: {self.evidence}")
|
|
46
|
+
lines.append(f" hint: {self.hint}")
|
|
47
|
+
return "\n".join(lines)
|
|
48
|
+
|
|
49
|
+
def to_dict(self) -> dict[str, Any]:
|
|
50
|
+
return {
|
|
51
|
+
"code": self.code,
|
|
52
|
+
"severity": self.severity.value,
|
|
53
|
+
"question_id": self.question_id,
|
|
54
|
+
"message": self.message,
|
|
55
|
+
"hint": self.hint,
|
|
56
|
+
"evidence": self.evidence,
|
|
57
|
+
}
|
jevkit_lint/linter.py
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""The lint entry point."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, Iterable
|
|
6
|
+
|
|
7
|
+
from jevkit_core import normalize_questions
|
|
8
|
+
|
|
9
|
+
from .diagnostic import Diagnostic, Severity
|
|
10
|
+
from .rules import LintContext, rules_for
|
|
11
|
+
|
|
12
|
+
__all__ = ["lint", "LintResult"]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class LintResult:
|
|
16
|
+
"""Diagnostics for one request, ordered most severe first."""
|
|
17
|
+
|
|
18
|
+
def __init__(self, diagnostics: list[Diagnostic]) -> None:
|
|
19
|
+
self.diagnostics = sorted(diagnostics, key=lambda d: d.sort_key)
|
|
20
|
+
|
|
21
|
+
def __iter__(self):
|
|
22
|
+
return iter(self.diagnostics)
|
|
23
|
+
|
|
24
|
+
def __len__(self) -> int:
|
|
25
|
+
return len(self.diagnostics)
|
|
26
|
+
|
|
27
|
+
def __bool__(self) -> bool:
|
|
28
|
+
return bool(self.diagnostics)
|
|
29
|
+
|
|
30
|
+
def by_severity(self, severity: Severity) -> list[Diagnostic]:
|
|
31
|
+
return [d for d in self.diagnostics if d.severity is severity]
|
|
32
|
+
|
|
33
|
+
@property
|
|
34
|
+
def errors(self) -> list[Diagnostic]:
|
|
35
|
+
return self.by_severity(Severity.ERROR)
|
|
36
|
+
|
|
37
|
+
@property
|
|
38
|
+
def warnings(self) -> list[Diagnostic]:
|
|
39
|
+
return self.by_severity(Severity.WARNING)
|
|
40
|
+
|
|
41
|
+
@property
|
|
42
|
+
def infos(self) -> list[Diagnostic]:
|
|
43
|
+
return self.by_severity(Severity.INFO)
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def ok(self) -> bool:
|
|
47
|
+
"""True when nothing rose to an error."""
|
|
48
|
+
return not self.errors
|
|
49
|
+
|
|
50
|
+
def counts(self) -> dict[str, int]:
|
|
51
|
+
return {
|
|
52
|
+
"error": len(self.errors),
|
|
53
|
+
"warning": len(self.warnings),
|
|
54
|
+
"info": len(self.infos),
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
def to_dict(self) -> dict[str, Any]:
|
|
58
|
+
return {
|
|
59
|
+
"ok": self.ok,
|
|
60
|
+
"counts": self.counts(),
|
|
61
|
+
"diagnostics": [d.to_dict() for d in self.diagnostics],
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
def format(self, *, color: bool = False) -> str:
|
|
65
|
+
if not self.diagnostics:
|
|
66
|
+
return "No problems found."
|
|
67
|
+
return "\n".join(d.format(color=color) for d in self.diagnostics)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def lint(
|
|
71
|
+
questions: dict[str, Any],
|
|
72
|
+
state: Any = "",
|
|
73
|
+
*,
|
|
74
|
+
select: Iterable[str] | None = None,
|
|
75
|
+
ignore: Iterable[str] | None = None,
|
|
76
|
+
) -> LintResult:
|
|
77
|
+
"""Lint a jev request without calling the API.
|
|
78
|
+
|
|
79
|
+
``questions`` accepts SDK objects or plain dicts. ``state`` is optional: omit
|
|
80
|
+
it to check the questions alone, though the budget rules can only report
|
|
81
|
+
meaningfully when the real state is supplied.
|
|
82
|
+
"""
|
|
83
|
+
normalized = normalize_questions(questions)
|
|
84
|
+
ctx = LintContext.build(state, normalized)
|
|
85
|
+
ignored = {c.upper() for c in (ignore or ())}
|
|
86
|
+
|
|
87
|
+
found: list[Diagnostic] = []
|
|
88
|
+
for rule in rules_for(select):
|
|
89
|
+
if rule.code in ignored:
|
|
90
|
+
continue
|
|
91
|
+
found.extend(rule.check(ctx))
|
|
92
|
+
return LintResult(found)
|
jevkit_lint/rules.py
ADDED
|
@@ -0,0 +1,595 @@
|
|
|
1
|
+
"""Lint rules.
|
|
2
|
+
|
|
3
|
+
Every rule here maps to a failure mode TypeSafe documents for `jev-1.13`
|
|
4
|
+
(https://docs.typesafe.ai/model-jaggedness/jev-1.13) or to a structural
|
|
5
|
+
mistake that makes an answer unusable. The `mode` field on each rule names
|
|
6
|
+
the documented failure mode so the CLI can point at the source.
|
|
7
|
+
|
|
8
|
+
Rules are static. They read your question definitions and never call the API,
|
|
9
|
+
which is why jevkit-lint works without an API key.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import re
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
from typing import Any, Callable, Iterable
|
|
17
|
+
|
|
18
|
+
from jevkit_core import BudgetReport, Question, check_budget
|
|
19
|
+
|
|
20
|
+
from .diagnostic import Diagnostic, Severity
|
|
21
|
+
|
|
22
|
+
__all__ = ["LintContext", "Rule", "RULES", "rules_for", "all_codes"]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class LintContext:
|
|
27
|
+
questions: list[Question]
|
|
28
|
+
state: Any
|
|
29
|
+
budget: BudgetReport
|
|
30
|
+
|
|
31
|
+
@classmethod
|
|
32
|
+
def build(cls, state: Any, questions: list[Question]) -> "LintContext":
|
|
33
|
+
# Budget estimation canonicalizes its input, so questions are reduced to
|
|
34
|
+
# plain JSON here. SDK objects are not serializable and the raw form is
|
|
35
|
+
# only ever needed by rules that read the normalized view anyway.
|
|
36
|
+
plain = {
|
|
37
|
+
q.id: {"type": q.type, "instructions": q.instructions, "criteria": q.criteria}
|
|
38
|
+
for q in questions
|
|
39
|
+
}
|
|
40
|
+
return cls(questions=questions, state=state, budget=check_budget(state, plain))
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass(frozen=True)
|
|
44
|
+
class Rule:
|
|
45
|
+
code: str
|
|
46
|
+
name: str
|
|
47
|
+
mode: str
|
|
48
|
+
check: Callable[[LintContext], list[Diagnostic]]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _words(text: str) -> str:
|
|
52
|
+
return text.lower()
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _find(patterns: Iterable[str], text: str) -> str | None:
|
|
56
|
+
for pattern in patterns:
|
|
57
|
+
match = re.search(pattern, text, flags=re.IGNORECASE)
|
|
58
|
+
if match:
|
|
59
|
+
return match.group(0).strip()
|
|
60
|
+
return None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
# --------------------------------------------------------------------------
|
|
64
|
+
# JEV001 — Math and Numbers
|
|
65
|
+
# --------------------------------------------------------------------------
|
|
66
|
+
|
|
67
|
+
_COUNT_PATTERNS = [
|
|
68
|
+
r"\bhow many\b", r"\bnumber of\b", r"\bcount (?:the|of|how)\b", r"\btally\b",
|
|
69
|
+
r"\btotal (?:number|count|of)\b", r"\bsum of\b", r"\baverage\b", r"\bmean of\b",
|
|
70
|
+
r"\bcalculate\b", r"\bcompute the\b", r"\bpercentage of\b", r"\bhow much (?:is|does)\b",
|
|
71
|
+
r"\bmultipl(?:y|ied)\b", r"\bdivide[d]?\b", r"\bsubtract\b",
|
|
72
|
+
]
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _check_math(ctx: LintContext) -> list[Diagnostic]:
|
|
76
|
+
out = []
|
|
77
|
+
for q in ctx.questions:
|
|
78
|
+
hit = _find(_COUNT_PATTERNS, q.text)
|
|
79
|
+
if hit:
|
|
80
|
+
out.append(Diagnostic(
|
|
81
|
+
code="JEV001", severity=Severity.ERROR, question_id=q.id,
|
|
82
|
+
message="Question asks the model to count or do arithmetic.",
|
|
83
|
+
evidence=hit,
|
|
84
|
+
hint="jev-1.13 is not a calculator and does not count reliably. "
|
|
85
|
+
"Iterate the candidates in code, ask one Noul per item, and sum "
|
|
86
|
+
"the answers yourself.",
|
|
87
|
+
))
|
|
88
|
+
return out
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
# --------------------------------------------------------------------------
|
|
92
|
+
# JEV002 — Date and time comparison
|
|
93
|
+
# --------------------------------------------------------------------------
|
|
94
|
+
|
|
95
|
+
_DATE_COMPARE_PATTERNS = [
|
|
96
|
+
r"\b(?:before|after|earlier than|later than|prior to)\b[^.?]{0,40}\b(?:date|day|month|year|deadline|timestamp)\b",
|
|
97
|
+
r"\b(?:date|day|month|year|deadline|timestamp)\b[^.?]{0,40}\b(?:before|after|earlier|later)\b",
|
|
98
|
+
r"\bhow (?:long|many days|many months|many years)\b",
|
|
99
|
+
r"\bwithin \d+ (?:day|week|month|year)s?\b",
|
|
100
|
+
r"\b(?:days|weeks|months|years) (?:between|apart|since|until)\b",
|
|
101
|
+
r"\bmost recent\b", r"\bchronologic(?:al|ally)\b",
|
|
102
|
+
r"\b(?:which|what)\b[^.?]{0,30}\b(?:came|comes|happened|occurred|was|is)\s+"
|
|
103
|
+
r"(?:first|last|earliest|latest|more recent)\b",
|
|
104
|
+
r"\b(?:earliest|latest|oldest|newest)\b[^.?]{0,20}"
|
|
105
|
+
r"\b(?:date|day|month|year|deadline|timestamp|event|entry)\b",
|
|
106
|
+
r"\b(?:date|day|month|year|deadline|timestamp)\b[^.?]{0,20}"
|
|
107
|
+
r"\b(?:first|last|earliest|latest)\b",
|
|
108
|
+
r"\bexpired?\b", r"\boverdue\b", r"\bin the (?:past|last|next) \d+\b",
|
|
109
|
+
]
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _check_dates(ctx: LintContext) -> list[Diagnostic]:
|
|
113
|
+
out = []
|
|
114
|
+
for q in ctx.questions:
|
|
115
|
+
hit = _find(_DATE_COMPARE_PATTERNS, q.text)
|
|
116
|
+
if hit:
|
|
117
|
+
out.append(Diagnostic(
|
|
118
|
+
code="JEV002", severity=Severity.ERROR, question_id=q.id,
|
|
119
|
+
message="Question compares or measures dates.",
|
|
120
|
+
evidence=hit,
|
|
121
|
+
hint="jev-1.13 reads dates as text, not ordered quantities. Extract the "
|
|
122
|
+
"parts as Choices over closed sets (12 months, 31 days, a bounded "
|
|
123
|
+
"year range, plus an explicit 'not stated'), then order and subtract "
|
|
124
|
+
"in code.",
|
|
125
|
+
))
|
|
126
|
+
return out
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
# --------------------------------------------------------------------------
|
|
130
|
+
# JEV003 — Generation
|
|
131
|
+
# --------------------------------------------------------------------------
|
|
132
|
+
|
|
133
|
+
_GENERATION_PATTERNS = [
|
|
134
|
+
r"\b(?:write|draft|compose|author) (?:a|an|the|some)\b",
|
|
135
|
+
r"\b(?:generate|produce|create) (?:a|an|the)\s+(?:summary|response|reply|message|list|description|explanation|text|paragraph)\b",
|
|
136
|
+
r"\bsummari[sz]e\b", r"\bparaphrase\b", r"\brephrase\b", r"\brewrite\b",
|
|
137
|
+
r"\bexplain (?:why|how|what)\b", r"\bdescribe (?:in|the|what|how)\b",
|
|
138
|
+
r"\bin your own words\b", r"\bprovide (?:a|an) (?:summary|explanation|rationale|reason)\b",
|
|
139
|
+
r"\bgive (?:a|an) (?:reason|explanation|rationale)\b", r"\btranslate\b",
|
|
140
|
+
]
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _check_generation(ctx: LintContext) -> list[Diagnostic]:
|
|
144
|
+
out = []
|
|
145
|
+
for q in ctx.questions:
|
|
146
|
+
hit = _find(_GENERATION_PATTERNS, q.instructions_text)
|
|
147
|
+
if hit:
|
|
148
|
+
out.append(Diagnostic(
|
|
149
|
+
code="JEV003", severity=Severity.ERROR, question_id=q.id,
|
|
150
|
+
message="Question asks the model to generate text.",
|
|
151
|
+
evidence=hit,
|
|
152
|
+
hint="Jev returns typed answers and probabilities, never generated text. "
|
|
153
|
+
"Use a generative model for this, or restate it as a selection over "
|
|
154
|
+
"candidates your code already has.",
|
|
155
|
+
))
|
|
156
|
+
return out
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
# --------------------------------------------------------------------------
|
|
160
|
+
# JEV004 — Literal reading: negation
|
|
161
|
+
# --------------------------------------------------------------------------
|
|
162
|
+
|
|
163
|
+
_NEGATIONS = r"\b(?:not|never|no|none|neither|nor|without|except|unless|excluding|absent|lacks?|fails? to|cannot|can't|doesn't|does not|isn't|is not|aren't|won't)\b"
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _check_negation(ctx: LintContext) -> list[Diagnostic]:
|
|
167
|
+
out = []
|
|
168
|
+
for q in ctx.questions:
|
|
169
|
+
text = q.instructions_text
|
|
170
|
+
hits = re.findall(_NEGATIONS, text, flags=re.IGNORECASE)
|
|
171
|
+
if len(hits) >= 2:
|
|
172
|
+
out.append(Diagnostic(
|
|
173
|
+
code="JEV004", severity=Severity.WARNING, question_id=q.id,
|
|
174
|
+
message=f"Instructions contain {len(hits)} negations, which compounds into "
|
|
175
|
+
"a double negative.",
|
|
176
|
+
evidence=", ".join(sorted({h.lower() for h in hits})),
|
|
177
|
+
hint="jev-1.13 reads negations literally and loses accuracy on double "
|
|
178
|
+
"negatives. Restate positively, or split into two literal questions "
|
|
179
|
+
"and combine them in code.",
|
|
180
|
+
))
|
|
181
|
+
elif hits and q.type == "noul":
|
|
182
|
+
out.append(Diagnostic(
|
|
183
|
+
code="JEV004", severity=Severity.INFO, question_id=q.id,
|
|
184
|
+
message="Noul instruction is phrased negatively.",
|
|
185
|
+
evidence=hits[0].lower(),
|
|
186
|
+
hint="A Noul returns the probability the statement is true. A negative "
|
|
187
|
+
"statement inverts the reading of every threshold downstream. Prefer "
|
|
188
|
+
"the positive form and invert in code if you need it.",
|
|
189
|
+
))
|
|
190
|
+
return out
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
# --------------------------------------------------------------------------
|
|
194
|
+
# JEV005 — Literal reading: vague scoping
|
|
195
|
+
# --------------------------------------------------------------------------
|
|
196
|
+
|
|
197
|
+
_VAGUE = [
|
|
198
|
+
r"\brelevant\b", r"\bappropriate\b", r"\bimportant\b", r"\bsignificant\b",
|
|
199
|
+
r"\bsuitable\b", r"\bproper\b", r"\bgood\b", r"\bbad\b", r"\bbetter\b",
|
|
200
|
+
r"\breasonable\b", r"\bacceptable\b", r"\bsufficient\b", r"\badequate\b",
|
|
201
|
+
r"\bmeaningful\b", r"\bnoteworthy\b", r"\bproblematic\b",
|
|
202
|
+
]
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _check_vague(ctx: LintContext) -> list[Diagnostic]:
|
|
206
|
+
out = []
|
|
207
|
+
for q in ctx.questions:
|
|
208
|
+
found = sorted({
|
|
209
|
+
m.group(0).lower()
|
|
210
|
+
for pattern in _VAGUE
|
|
211
|
+
for m in re.finditer(pattern, q.instructions_text, flags=re.IGNORECASE)
|
|
212
|
+
})
|
|
213
|
+
if found and not q.criteria_text.strip():
|
|
214
|
+
out.append(Diagnostic(
|
|
215
|
+
code="JEV005", severity=Severity.WARNING, question_id=q.id,
|
|
216
|
+
message="Instructions lean on an undefined evaluative word and give no "
|
|
217
|
+
"criteria to pin it down.",
|
|
218
|
+
evidence=", ".join(found),
|
|
219
|
+
hint="jev-1.13 answers the question you wrote, not the one you meant. "
|
|
220
|
+
"Define what the word means here in the criteria, including the "
|
|
221
|
+
"boundary cases.",
|
|
222
|
+
))
|
|
223
|
+
elif len(found) >= 2:
|
|
224
|
+
out.append(Diagnostic(
|
|
225
|
+
code="JEV005", severity=Severity.INFO, question_id=q.id,
|
|
226
|
+
message="Instructions stack several undefined evaluative words.",
|
|
227
|
+
evidence=", ".join(found),
|
|
228
|
+
hint="Each one is a separate judgment the model has to guess at. Name the "
|
|
229
|
+
"exact condition, or split into separate questions.",
|
|
230
|
+
))
|
|
231
|
+
return out
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
# --------------------------------------------------------------------------
|
|
235
|
+
# JEV006 — Indirection
|
|
236
|
+
# --------------------------------------------------------------------------
|
|
237
|
+
|
|
238
|
+
_INDIRECTION_PATTERNS = [
|
|
239
|
+
r"\bthe \w+ of the \w+ of the\b",
|
|
240
|
+
r"\bwhoever\b[^.?]{0,40}\bwhose\b",
|
|
241
|
+
r"\bif .{0,60}\bthen\b.{0,60}\bif\b",
|
|
242
|
+
r"\bwould have (?:been|had)\b",
|
|
243
|
+
r"\bimplies? that\b[^.?]{0,40}\bwhich\b",
|
|
244
|
+
r"\bindirectly\b", r"\btransitively\b",
|
|
245
|
+
]
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _check_indirection(ctx: LintContext) -> list[Diagnostic]:
|
|
249
|
+
out = []
|
|
250
|
+
for q in ctx.questions:
|
|
251
|
+
hit = _find(_INDIRECTION_PATTERNS, q.instructions_text)
|
|
252
|
+
if hit:
|
|
253
|
+
out.append(Diagnostic(
|
|
254
|
+
code="JEV006", severity=Severity.WARNING, question_id=q.id,
|
|
255
|
+
message="Instructions require multiple hops of reasoning.",
|
|
256
|
+
evidence=hit,
|
|
257
|
+
hint="Every hop costs accuracy. Resolve the intermediate step in code and "
|
|
258
|
+
"name the resulting state field directly in the instruction.",
|
|
259
|
+
))
|
|
260
|
+
return out
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
# --------------------------------------------------------------------------
|
|
264
|
+
# JEV007 — Contradictory instructions and criteria (Noul polarity)
|
|
265
|
+
# --------------------------------------------------------------------------
|
|
266
|
+
|
|
267
|
+
_FALSEY = {"no", "false", "absent", "none", "negative", "not present", "does not", "fails"}
|
|
268
|
+
_TRUTHY = {"yes", "true", "present", "positive", "does", "passes"}
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def _check_noul_polarity(ctx: LintContext) -> list[Diagnostic]:
|
|
272
|
+
out = []
|
|
273
|
+
for q in ctx.questions:
|
|
274
|
+
if q.type != "noul" or not isinstance(q.criteria, dict):
|
|
275
|
+
continue
|
|
276
|
+
true_side = _words(str(q.criteria.get("true", "")))
|
|
277
|
+
false_side = _words(str(q.criteria.get("false", "")))
|
|
278
|
+
if not true_side and not false_side:
|
|
279
|
+
continue
|
|
280
|
+
inverted = (
|
|
281
|
+
any(tok in true_side for tok in _FALSEY)
|
|
282
|
+
and any(tok in false_side for tok in _TRUTHY)
|
|
283
|
+
)
|
|
284
|
+
if inverted:
|
|
285
|
+
out.append(Diagnostic(
|
|
286
|
+
code="JEV007", severity=Severity.ERROR, question_id=q.id,
|
|
287
|
+
message="Noul criteria invert polarity: 'true' describes a no and 'false' "
|
|
288
|
+
"describes a yes.",
|
|
289
|
+
evidence=f"true={true_side!r} false={false_side!r}",
|
|
290
|
+
hint="TypeSafe documents this exact shape as a performance loss. Swap the "
|
|
291
|
+
"two descriptions and rewrite the instruction so 'true' means yes.",
|
|
292
|
+
))
|
|
293
|
+
return out
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
# --------------------------------------------------------------------------
|
|
297
|
+
# JEV008 — Choice structure
|
|
298
|
+
# --------------------------------------------------------------------------
|
|
299
|
+
|
|
300
|
+
_NO_MATCH = {
|
|
301
|
+
"none", "none_of_the_above", "no_match", "nomatch", "other", "unknown",
|
|
302
|
+
"unclear", "not_stated", "not_applicable", "n_a", "na", "neither",
|
|
303
|
+
"cannot_tell", "insufficient", "uncertain", "ambiguous", "not_specified",
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _check_choice_no_match(ctx: LintContext) -> list[Diagnostic]:
|
|
308
|
+
out = []
|
|
309
|
+
for q in ctx.questions:
|
|
310
|
+
if q.type != "choice":
|
|
311
|
+
continue
|
|
312
|
+
keys = {k.lower().replace("-", "_").replace(" ", "_") for k in q.options}
|
|
313
|
+
if keys and not (keys & _NO_MATCH):
|
|
314
|
+
out.append(Diagnostic(
|
|
315
|
+
code="JEV008", severity=Severity.WARNING, question_id=q.id,
|
|
316
|
+
message="Choice has no no-match option.",
|
|
317
|
+
evidence=", ".join(sorted(keys)),
|
|
318
|
+
hint="A Choice always returns one of its options. With nothing meaning "
|
|
319
|
+
"'none of these', a state that fits no option still produces a "
|
|
320
|
+
"confident-looking answer. Add an explicit none/unknown option, or "
|
|
321
|
+
"gate on a separate presence Noul.",
|
|
322
|
+
))
|
|
323
|
+
return out
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def _check_choice_arity(ctx: LintContext) -> list[Diagnostic]:
|
|
327
|
+
out = []
|
|
328
|
+
for q in ctx.questions:
|
|
329
|
+
if q.type != "choice":
|
|
330
|
+
continue
|
|
331
|
+
n = len(q.options)
|
|
332
|
+
if n == 0:
|
|
333
|
+
out.append(Diagnostic(
|
|
334
|
+
code="JEV009", severity=Severity.ERROR, question_id=q.id,
|
|
335
|
+
message="Choice defines no options.",
|
|
336
|
+
hint="Give the Choice a criteria object mapping each option key to a "
|
|
337
|
+
"description of when it applies.",
|
|
338
|
+
))
|
|
339
|
+
elif n == 1:
|
|
340
|
+
out.append(Diagnostic(
|
|
341
|
+
code="JEV009", severity=Severity.ERROR, question_id=q.id,
|
|
342
|
+
message="Choice defines a single option, so the answer is predetermined.",
|
|
343
|
+
evidence=q.options[0],
|
|
344
|
+
hint="Use a Noul if the question is really yes/no, or add the competing "
|
|
345
|
+
"options.",
|
|
346
|
+
))
|
|
347
|
+
return out
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _check_empty_descriptions(ctx: LintContext) -> list[Diagnostic]:
|
|
351
|
+
out = []
|
|
352
|
+
for q in ctx.questions:
|
|
353
|
+
if q.type != "choice":
|
|
354
|
+
continue
|
|
355
|
+
blank = [key for key, desc in q.option_descriptions() if not desc.strip()]
|
|
356
|
+
if blank:
|
|
357
|
+
out.append(Diagnostic(
|
|
358
|
+
code="JEV010", severity=Severity.WARNING, question_id=q.id,
|
|
359
|
+
message="Choice options have no description.",
|
|
360
|
+
evidence=", ".join(blank),
|
|
361
|
+
hint="The option key alone is all the model gets. Describe when each "
|
|
362
|
+
"option applies, especially the boundary against its nearest rival.",
|
|
363
|
+
))
|
|
364
|
+
return out
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
# --------------------------------------------------------------------------
|
|
368
|
+
# JEV011 — Score structure
|
|
369
|
+
# --------------------------------------------------------------------------
|
|
370
|
+
|
|
371
|
+
def _check_score_levels(ctx: LintContext) -> list[Diagnostic]:
|
|
372
|
+
out = []
|
|
373
|
+
for q in ctx.questions:
|
|
374
|
+
if q.type != "score":
|
|
375
|
+
continue
|
|
376
|
+
levels = q.options
|
|
377
|
+
if len(levels) < 2:
|
|
378
|
+
out.append(Diagnostic(
|
|
379
|
+
code="JEV011", severity=Severity.ERROR, question_id=q.id,
|
|
380
|
+
message=f"Score defines {len(levels)} level(s); at least 2 are needed to "
|
|
381
|
+
"form a scale.",
|
|
382
|
+
hint="Give each level a concrete description of the situation it covers.",
|
|
383
|
+
))
|
|
384
|
+
continue
|
|
385
|
+
terse = [lv for lv in levels if len(lv.split()) < 2]
|
|
386
|
+
if terse:
|
|
387
|
+
out.append(Diagnostic(
|
|
388
|
+
code="JEV011", severity=Severity.WARNING, question_id=q.id,
|
|
389
|
+
message="Score levels are bare labels rather than descriptions.",
|
|
390
|
+
evidence=", ".join(terse),
|
|
391
|
+
hint="Levels must describe concrete situations and stand on their own. "
|
|
392
|
+
"'Frustrated but civil' works; 'medium' does not.",
|
|
393
|
+
))
|
|
394
|
+
return out
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
# --------------------------------------------------------------------------
|
|
398
|
+
# JEV012 — Instructions present and substantive
|
|
399
|
+
# --------------------------------------------------------------------------
|
|
400
|
+
|
|
401
|
+
def _check_instructions(ctx: LintContext) -> list[Diagnostic]:
|
|
402
|
+
out = []
|
|
403
|
+
for q in ctx.questions:
|
|
404
|
+
text = q.instructions_text.strip()
|
|
405
|
+
if not text:
|
|
406
|
+
out.append(Diagnostic(
|
|
407
|
+
code="JEV012", severity=Severity.ERROR, question_id=q.id,
|
|
408
|
+
message="Question has no instructions.",
|
|
409
|
+
hint="The instruction carries the judgment. Without it the model only has "
|
|
410
|
+
"the criteria to go on.",
|
|
411
|
+
))
|
|
412
|
+
elif len(text.split()) < 3:
|
|
413
|
+
out.append(Diagnostic(
|
|
414
|
+
code="JEV012", severity=Severity.WARNING, question_id=q.id,
|
|
415
|
+
message="Instructions are too terse to state a condition.",
|
|
416
|
+
evidence=text,
|
|
417
|
+
hint="State the exact condition being judged, in language an average "
|
|
418
|
+
"reader would resolve the same way you do.",
|
|
419
|
+
))
|
|
420
|
+
return out
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
# --------------------------------------------------------------------------
|
|
424
|
+
# JEV013 — Question ids are not sent to the model
|
|
425
|
+
# --------------------------------------------------------------------------
|
|
426
|
+
|
|
427
|
+
def _check_id_reference(ctx: LintContext) -> list[Diagnostic]:
|
|
428
|
+
out = []
|
|
429
|
+
ids = {q.id for q in ctx.questions}
|
|
430
|
+
for q in ctx.questions:
|
|
431
|
+
text = q.instructions_text
|
|
432
|
+
referenced = sorted({
|
|
433
|
+
other for other in ids
|
|
434
|
+
if other != q.id
|
|
435
|
+
and re.search(rf"(?<![\w.`]){re.escape(other)}(?![\w`])", text)
|
|
436
|
+
})
|
|
437
|
+
if referenced:
|
|
438
|
+
out.append(Diagnostic(
|
|
439
|
+
code="JEV013", severity=Severity.WARNING, question_id=q.id,
|
|
440
|
+
message="Instructions refer to another question by its id.",
|
|
441
|
+
evidence=", ".join(referenced),
|
|
442
|
+
hint="Question ids are for your code and are not sent to the model. "
|
|
443
|
+
"Questions in one request are answered in parallel and cannot see "
|
|
444
|
+
"each other. State the premise explicitly, or split into a second "
|
|
445
|
+
"request.",
|
|
446
|
+
))
|
|
447
|
+
return out
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
# --------------------------------------------------------------------------
|
|
451
|
+
# JEV014 / JEV015 — Budgets
|
|
452
|
+
# --------------------------------------------------------------------------
|
|
453
|
+
|
|
454
|
+
def _check_state_budget(ctx: LintContext) -> list[Diagnostic]:
|
|
455
|
+
budget = ctx.budget
|
|
456
|
+
if not budget.over_state:
|
|
457
|
+
return []
|
|
458
|
+
return [Diagnostic(
|
|
459
|
+
code="JEV014", severity=Severity.ERROR, question_id=None,
|
|
460
|
+
message=f"state plus the longest question is about {budget.longest_pair:,} "
|
|
461
|
+
f"tokens, over the 32,000 limit.",
|
|
462
|
+
evidence=f"longest question: {budget.longest_question_id}",
|
|
463
|
+
hint="Retrieve and filter in code so the state carries only the fields this "
|
|
464
|
+
"judgment needs. Estimates are approximate; leave headroom.",
|
|
465
|
+
)]
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def _check_total_budget(ctx: LintContext) -> list[Diagnostic]:
|
|
469
|
+
budget = ctx.budget
|
|
470
|
+
if not budget.over_total:
|
|
471
|
+
return []
|
|
472
|
+
return [Diagnostic(
|
|
473
|
+
code="JEV015", severity=Severity.ERROR, question_id=None,
|
|
474
|
+
message=f"state plus all {len(ctx.questions)} questions is about "
|
|
475
|
+
f"{budget.total:,} tokens, over the 64,000 limit.",
|
|
476
|
+
hint="Split into several requests, or drop speculative questions that this "
|
|
477
|
+
"state can never make relevant.",
|
|
478
|
+
)]
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
# --------------------------------------------------------------------------
|
|
482
|
+
# JEV016 — Large state of mostly irrelevant detail
|
|
483
|
+
# --------------------------------------------------------------------------
|
|
484
|
+
|
|
485
|
+
_LARGE_STATE_RATIO = 8.0
|
|
486
|
+
_LARGE_STATE_FLOOR = 4_000
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
def _check_state_size(ctx: LintContext) -> list[Diagnostic]:
|
|
490
|
+
budget = ctx.budget
|
|
491
|
+
if budget.over_state or budget.over_total:
|
|
492
|
+
return [] # Already reported as an error; don't double up.
|
|
493
|
+
question_tokens = sum(budget.question_tokens.values()) or 1
|
|
494
|
+
ratio = budget.state_tokens / question_tokens
|
|
495
|
+
if budget.state_tokens >= _LARGE_STATE_FLOOR and ratio >= _LARGE_STATE_RATIO:
|
|
496
|
+
return [Diagnostic(
|
|
497
|
+
code="JEV016", severity=Severity.INFO, question_id=None,
|
|
498
|
+
message=f"state is about {budget.state_tokens:,} tokens against "
|
|
499
|
+
f"{question_tokens:,} tokens of questions ({ratio:.0f}x).",
|
|
500
|
+
hint="Accuracy falls as the state grows with content unrelated to the "
|
|
501
|
+
"decision, and a large state makes a wrong answer hard to attribute. "
|
|
502
|
+
"Filter in code first, or use a Noul to screen for relevance.",
|
|
503
|
+
)]
|
|
504
|
+
return []
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
# --------------------------------------------------------------------------
|
|
508
|
+
# JEV017 — Numeric representations
|
|
509
|
+
# --------------------------------------------------------------------------
|
|
510
|
+
|
|
511
|
+
_NUMERIC_REPR = [
|
|
512
|
+
r"#[0-9a-fA-F]{6}\b", r"\brgba?\s*\(", r"\bhex(?:adecimal)? (?:value|code|colou?r)\b",
|
|
513
|
+
r"\bbinary (?:value|encoding|representation)\b", r"\bbase64\b",
|
|
514
|
+
r"\bassembly (?:instruction|opcode)\b", r"\bopcode\b", r"\bbytecode\b",
|
|
515
|
+
]
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
def _check_numeric_repr(ctx: LintContext) -> list[Diagnostic]:
|
|
519
|
+
out = []
|
|
520
|
+
for q in ctx.questions:
|
|
521
|
+
hit = _find(_NUMERIC_REPR, q.text)
|
|
522
|
+
if hit:
|
|
523
|
+
out.append(Diagnostic(
|
|
524
|
+
code="JEV017", severity=Severity.WARNING, question_id=q.id,
|
|
525
|
+
message="Question reasons over a machine-oriented numeric representation.",
|
|
526
|
+
evidence=hit,
|
|
527
|
+
hint="jev-1.13 does better on semantic representations than numeric ones: "
|
|
528
|
+
"colour names beat hex, high-level code beats bytecode. Convert in "
|
|
529
|
+
"code and pass the name or a named bucket.",
|
|
530
|
+
))
|
|
531
|
+
return out
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
# --------------------------------------------------------------------------
|
|
535
|
+
# JEV018 — Duplicate judgments
|
|
536
|
+
# --------------------------------------------------------------------------
|
|
537
|
+
|
|
538
|
+
def _check_duplicates(ctx: LintContext) -> list[Diagnostic]:
|
|
539
|
+
seen: dict[str, str] = {}
|
|
540
|
+
out = []
|
|
541
|
+
for q in ctx.questions:
|
|
542
|
+
key = re.sub(r"\W+", " ", q.text.lower()).strip()
|
|
543
|
+
if not key:
|
|
544
|
+
continue
|
|
545
|
+
if key in seen:
|
|
546
|
+
out.append(Diagnostic(
|
|
547
|
+
code="JEV018", severity=Severity.INFO, question_id=q.id,
|
|
548
|
+
message=f"Question is textually identical to {seen[key]!r}.",
|
|
549
|
+
hint="Identical questions in one request cost tokens twice for the same "
|
|
550
|
+
"answer. If you meant to measure self-consistency, note that they "
|
|
551
|
+
"are evaluated in one pass and are not independent samples.",
|
|
552
|
+
))
|
|
553
|
+
else:
|
|
554
|
+
seen[key] = q.id
|
|
555
|
+
return out
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
RULES: list[Rule] = [
|
|
559
|
+
Rule("JEV001", "math-and-counting", "Math and Numbers", _check_math),
|
|
560
|
+
Rule("JEV002", "date-comparison", "Date and time comparison", _check_dates),
|
|
561
|
+
Rule("JEV003", "generation-request", "Generation", _check_generation),
|
|
562
|
+
Rule("JEV004", "negation", "Literal reading", _check_negation),
|
|
563
|
+
Rule("JEV005", "vague-scoping", "Literal reading", _check_vague),
|
|
564
|
+
Rule("JEV006", "indirection", "Indirection", _check_indirection),
|
|
565
|
+
Rule("JEV007", "noul-polarity", "Contradictory instructions and criteria",
|
|
566
|
+
_check_noul_polarity),
|
|
567
|
+
Rule("JEV008", "choice-no-match", "Common-sense structural invariants",
|
|
568
|
+
_check_choice_no_match),
|
|
569
|
+
Rule("JEV009", "choice-arity", "Common-sense structural invariants",
|
|
570
|
+
_check_choice_arity),
|
|
571
|
+
Rule("JEV010", "option-descriptions", "Literal reading", _check_empty_descriptions),
|
|
572
|
+
Rule("JEV011", "score-levels", "Literal reading", _check_score_levels),
|
|
573
|
+
Rule("JEV012", "instructions-present", "Literal reading", _check_instructions),
|
|
574
|
+
Rule("JEV013", "question-id-reference", "Indirection", _check_id_reference),
|
|
575
|
+
Rule("JEV014", "state-budget", "Large state full of irrelevant detail",
|
|
576
|
+
_check_state_budget),
|
|
577
|
+
Rule("JEV015", "total-budget", "Large state full of irrelevant detail",
|
|
578
|
+
_check_total_budget),
|
|
579
|
+
Rule("JEV016", "state-noise-ratio", "Large state full of irrelevant detail",
|
|
580
|
+
_check_state_size),
|
|
581
|
+
Rule("JEV017", "numeric-representation", "Math and Numbers", _check_numeric_repr),
|
|
582
|
+
Rule("JEV018", "duplicate-questions", "Common-sense structural invariants",
|
|
583
|
+
_check_duplicates),
|
|
584
|
+
]
|
|
585
|
+
|
|
586
|
+
|
|
587
|
+
def rules_for(codes: Iterable[str] | None = None) -> list[Rule]:
|
|
588
|
+
if codes is None:
|
|
589
|
+
return list(RULES)
|
|
590
|
+
wanted = {c.upper() for c in codes}
|
|
591
|
+
return [r for r in RULES if r.code in wanted]
|
|
592
|
+
|
|
593
|
+
|
|
594
|
+
def all_codes() -> list[str]:
|
|
595
|
+
return [r.code for r in RULES]
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: jevkit-lint
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Static linter for TypeSafe Jev questions. Catches the documented jev-1.13 failure modes before you spend a token. No API key required.
|
|
5
|
+
Project-URL: Homepage, https://github.com/pjdurden/jevkit-py
|
|
6
|
+
Project-URL: Issues, https://github.com/pjdurden/jevkit-py/issues
|
|
7
|
+
Author: Prajjwal Chittori
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
Keywords: jev,lint,linter,static-analysis,system-one,typesafe
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
17
|
+
Requires-Python: >=3.10
|
|
18
|
+
Requires-Dist: jevkit-core>=0.1.0
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
|
|
21
|
+
# jevkit-lint
|
|
22
|
+
|
|
23
|
+
Static linter for [TypeSafe](https://typesafe.ai) Jev questions.
|
|
24
|
+
|
|
25
|
+
TypeSafe publishes a [list of failure modes](https://docs.typesafe.ai/model-jaggedness/jev-1.13)
|
|
26
|
+
for `jev-1.13`: it reads instructions literally, it cannot count, it reads dates
|
|
27
|
+
as text rather than ordered quantities, it loses accuracy on indirection. Most
|
|
28
|
+
of those are visible in your question definitions before you send anything.
|
|
29
|
+
|
|
30
|
+
`jevkit-lint` reads the definitions and tells you. It never calls the API, so it
|
|
31
|
+
needs no key and costs nothing to run in CI.
|
|
32
|
+
|
|
33
|
+
> Unofficial and unaffiliated with TypeSafe.
|
|
34
|
+
|
|
35
|
+
## Install
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
pip install jevkit-lint
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Use
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from jevkit_lint import lint
|
|
45
|
+
|
|
46
|
+
result = lint({
|
|
47
|
+
"urgency": {"type": "noul", "instructions": "How many days has the customer waited?"},
|
|
48
|
+
"team": {"type": "choice", "instructions": "Which team should handle this",
|
|
49
|
+
"criteria": {"billing": "Payment issues", "technical": "Bugs"}},
|
|
50
|
+
})
|
|
51
|
+
|
|
52
|
+
print(result.format())
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
```
|
|
56
|
+
urgency: error [JEV001] Question asks the model to count or do arithmetic.
|
|
57
|
+
found: How many
|
|
58
|
+
hint: jev-1.13 is not a calculator and does not count reliably. Iterate the
|
|
59
|
+
candidates in code, ask one Noul per item, and sum the answers yourself.
|
|
60
|
+
team: warning [JEV008] Choice has no no-match option.
|
|
61
|
+
found: billing, technical
|
|
62
|
+
hint: A Choice always returns one of its options. With nothing meaning 'none of
|
|
63
|
+
these', a state that fits no option still produces a confident-looking
|
|
64
|
+
answer. Add an explicit none/unknown option, or gate on a separate
|
|
65
|
+
presence Noul.
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
`result.ok` is `True` when nothing rose to an error, so it drops straight into a
|
|
69
|
+
guard:
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
if not lint(questions, state).ok:
|
|
73
|
+
raise ValueError("refusing to send a request that will not answer what we meant")
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
## CLI
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
jevkit-lint request.json # a {state, questions} object, or bare questions
|
|
80
|
+
jevkit-lint cassette.jevl # lint every recorded request
|
|
81
|
+
jevkit-lint - < request.json # stdin
|
|
82
|
+
jevkit-lint request.json --strict # exit non-zero on warnings too
|
|
83
|
+
jevkit-lint request.json --format json
|
|
84
|
+
jevkit-lint --list-rules
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Exit codes: `0` clean, `1` findings, `2` bad usage.
|
|
88
|
+
|
|
89
|
+
## Rules
|
|
90
|
+
|
|
91
|
+
Each rule names the documented failure mode it comes from.
|
|
92
|
+
|
|
93
|
+
| Code | Rule | Failure mode |
|
|
94
|
+
| --- | --- | --- |
|
|
95
|
+
| JEV001 | math-and-counting | Math and Numbers |
|
|
96
|
+
| JEV002 | date-comparison | Date and time comparison |
|
|
97
|
+
| JEV003 | generation-request | Generation |
|
|
98
|
+
| JEV004 | negation | Literal reading |
|
|
99
|
+
| JEV005 | vague-scoping | Literal reading |
|
|
100
|
+
| JEV006 | indirection | Indirection |
|
|
101
|
+
| JEV007 | noul-polarity | Contradictory instructions and criteria |
|
|
102
|
+
| JEV008 | choice-no-match | Common-sense structural invariants |
|
|
103
|
+
| JEV009 | choice-arity | Common-sense structural invariants |
|
|
104
|
+
| JEV010 | option-descriptions | Literal reading |
|
|
105
|
+
| JEV011 | score-levels | Literal reading |
|
|
106
|
+
| JEV012 | instructions-present | Literal reading |
|
|
107
|
+
| JEV013 | question-id-reference | Indirection |
|
|
108
|
+
| JEV014 | state-budget | Large state full of irrelevant detail |
|
|
109
|
+
| JEV015 | total-budget | Large state full of irrelevant detail |
|
|
110
|
+
| JEV016 | state-noise-ratio | Large state full of irrelevant detail |
|
|
111
|
+
| JEV017 | numeric-representation | Math and Numbers |
|
|
112
|
+
| JEV018 | duplicate-questions | Common-sense structural invariants |
|
|
113
|
+
|
|
114
|
+
Select or suppress by code:
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
lint(questions, select=["JEV001", "JEV002"])
|
|
118
|
+
lint(questions, ignore=["JEV008"])
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
## What it cannot do
|
|
122
|
+
|
|
123
|
+
Adversarial content is a documented failure mode and is **not** linted. Whether
|
|
124
|
+
a state is hostile depends on the content at runtime, not on the question
|
|
125
|
+
definition, so a static rule would be theatre. Screen untrusted state at
|
|
126
|
+
request time instead.
|
|
127
|
+
|
|
128
|
+
Token counts are estimates. jevkit deliberately does not bundle a tokenizer:
|
|
129
|
+
TypeSafe does not publish which one Jev uses, and a confidently wrong count is
|
|
130
|
+
worse than an honest approximation. The estimator errs conservative, so leave
|
|
131
|
+
headroom near the limits.
|
|
132
|
+
|
|
133
|
+
## License
|
|
134
|
+
|
|
135
|
+
MIT
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
jevkit_lint/__init__.py,sha256=6w0GwccE5aCu271PYDkk21epxexHh4ujDPBh5quwnZQ,424
|
|
2
|
+
jevkit_lint/cli.py,sha256=sJfmI3B9ierXp1Udrp7Q3YW6OJHRNCz796DXVd1Io7U,4768
|
|
3
|
+
jevkit_lint/diagnostic.py,sha256=zKnNH28gnzbmkf0ohLMhPgp13wTi9J9jaS_HYYeu_7A,1718
|
|
4
|
+
jevkit_lint/linter.py,sha256=31qvvM_VL1YtdbazkE5uzSTryWUFJnKqzcZZj_lU0hI,2631
|
|
5
|
+
jevkit_lint/rules.py,sha256=91npICn0UYogQlD4fT1ynJPLSYU-SHuwW5H9ohdrWmU,25196
|
|
6
|
+
jevkit_lint-0.1.0.dist-info/METADATA,sha256=kJJhDF1H5F1AbUtVvBiUFh1H32uGBbIdvfk1pu8JGLQ,4920
|
|
7
|
+
jevkit_lint-0.1.0.dist-info/WHEEL,sha256=THafob7ofN-NsuMN7Mg4qZyHaQI7KkD-QlcQatYhXPo,87
|
|
8
|
+
jevkit_lint-0.1.0.dist-info/entry_points.txt,sha256=qoSHRa-WSIxrt_qNimmZOCFPZKNYio1Bkp92ybhT3oM,53
|
|
9
|
+
jevkit_lint-0.1.0.dist-info/RECORD,,
|