technical-answer-validator 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_server.py +46 -0
- tav_api.py +224 -0
- tav_core.py +206 -0
- technical_answer_validator-0.1.0.dist-info/METADATA +92 -0
- technical_answer_validator-0.1.0.dist-info/RECORD +8 -0
- technical_answer_validator-0.1.0.dist-info/WHEEL +4 -0
- technical_answer_validator-0.1.0.dist-info/entry_points.txt +3 -0
- technical_answer_validator-0.1.0.dist-info/licenses/LICENSE +9 -0
mcp_server.py
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""MCP stdio server exposing the answer evaluator as an agent tool."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from mcp.server import MCPServer
|
|
10
|
+
|
|
11
|
+
from tav_core import RequestError, evaluate
|
|
12
|
+
|
|
13
|
+
logging.basicConfig(level=logging.WARNING, format="%(levelname)s %(message)s")
|
|
14
|
+
mcp = MCPServer(
|
|
15
|
+
"Technical Answer Validator",
|
|
16
|
+
instructions=(
|
|
17
|
+
"Evaluate a user's technical answer against a rubric supplied for this call. "
|
|
18
|
+
"Always pass the caller's rubric and answer. Explain that results are keyword-based "
|
|
19
|
+
"assistive feedback, never official exam grades; preserve review_required in your response."
|
|
20
|
+
),
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@mcp.tool()
|
|
25
|
+
def evaluate_answer(rubric: dict[str, Any], answer: str) -> dict[str, Any]:
|
|
26
|
+
"""Check an answer against caller-provided concepts, synonyms, and numeric requirements.
|
|
27
|
+
|
|
28
|
+
Rubric fields: required_concepts (1-50 strings), optional accepted_synonyms keyed by
|
|
29
|
+
concept, optional numeric_requirements [{value, unit?, tolerance?}], and optional
|
|
30
|
+
required_count. No question bank is bundled. Review the returned result; it is not
|
|
31
|
+
an official exam grade.
|
|
32
|
+
"""
|
|
33
|
+
try:
|
|
34
|
+
return evaluate({"rubric": rubric, "answer": answer})
|
|
35
|
+
except RequestError as exc:
|
|
36
|
+
# Return a structured error result so the agent can repair its inputs.
|
|
37
|
+
return {"status": "invalid_request", "error": str(exc), "review_required": True}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def main() -> None:
|
|
41
|
+
# FastMCP stdio transport uses stdout for protocol messages; keep diagnostics on stderr.
|
|
42
|
+
mcp.run(transport="stdio")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
if __name__ == "__main__":
|
|
46
|
+
main()
|
tav_api.py
ADDED
|
@@ -0,0 +1,224 @@
|
|
|
1
|
+
"""Minimal standard-library HTTP adapter for the TAV MVP."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections import defaultdict, deque
|
|
6
|
+
from contextlib import closing
|
|
7
|
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
8
|
+
import hashlib
|
|
9
|
+
import hmac
|
|
10
|
+
import json
|
|
11
|
+
import re
|
|
12
|
+
import os
|
|
13
|
+
import sqlite3
|
|
14
|
+
import time
|
|
15
|
+
from datetime import datetime, timedelta, timezone
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from tav_core import RequestError, evaluate
|
|
19
|
+
|
|
20
|
+
MAX_BODY = 64 * 1024
|
|
21
|
+
RATE_LIMIT = 60
|
|
22
|
+
RATE_WINDOW = 60
|
|
23
|
+
_requests: dict[str, deque[float]] = defaultdict(deque)
|
|
24
|
+
MAX_RATE_CLIENTS = 10000
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _usage_db() -> Path:
|
|
28
|
+
return Path(os.environ.get("TAV_USAGE_DB", "usage.sqlite3"))
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _api_key_id(supplied_header: str) -> str | None:
|
|
32
|
+
"""Resolve bearer credentials against SHA-256 digests in TAV_API_KEYS."""
|
|
33
|
+
if not supplied_header.startswith("Bearer "):
|
|
34
|
+
return None
|
|
35
|
+
token = supplied_header[7:]
|
|
36
|
+
configured = os.environ.get("TAV_API_KEYS", "")
|
|
37
|
+
if configured:
|
|
38
|
+
try:
|
|
39
|
+
key_map = json.loads(configured)
|
|
40
|
+
except json.JSONDecodeError as exc:
|
|
41
|
+
raise RuntimeError("TAV_API_KEYS must be a JSON object of client_id to SHA-256 hex digest") from exc
|
|
42
|
+
if not isinstance(key_map, dict):
|
|
43
|
+
raise RuntimeError("TAV_API_KEYS must be a JSON object")
|
|
44
|
+
digest = hashlib.sha256(token.encode("utf-8")).hexdigest()
|
|
45
|
+
for client_id, expected_digest in key_map.items():
|
|
46
|
+
if not isinstance(client_id, str) or not re.fullmatch(r"[A-Za-z0-9_-]{1,48}", client_id):
|
|
47
|
+
continue
|
|
48
|
+
if isinstance(expected_digest, str) and hmac.compare_digest(digest, expected_digest.lower()):
|
|
49
|
+
return client_id
|
|
50
|
+
return None
|
|
51
|
+
# Single-key mode is for local development and backwards-compatible self-hosting.
|
|
52
|
+
expected = os.environ.get("TAV_API_KEY", "")
|
|
53
|
+
return "default" if expected and hmac.compare_digest(token, expected) else None
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _record_usage(client_id: str, kind: str) -> None:
|
|
57
|
+
"""Persist aggregate counters only; never store request content or answers."""
|
|
58
|
+
path = _usage_db()
|
|
59
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
60
|
+
today = datetime.now(timezone.utc).date().isoformat()
|
|
61
|
+
with closing(sqlite3.connect(path, timeout=5)) as connection:
|
|
62
|
+
with connection:
|
|
63
|
+
connection.execute("""CREATE TABLE IF NOT EXISTS daily_usage (
|
|
64
|
+
day TEXT NOT NULL, client_id TEXT NOT NULL, requests INTEGER NOT NULL DEFAULT 0,
|
|
65
|
+
successful INTEGER NOT NULL DEFAULT 0, client_errors INTEGER NOT NULL DEFAULT 0,
|
|
66
|
+
rate_limited INTEGER NOT NULL DEFAULT 0, PRIMARY KEY(day, client_id))""")
|
|
67
|
+
columns = {"success": "successful", "client_error": "client_errors", "rate_limited": "rate_limited"}
|
|
68
|
+
if kind not in {"success", "client_error", "rate_limited", "server_error"}:
|
|
69
|
+
raise ValueError("invalid usage event")
|
|
70
|
+
connection.execute("DELETE FROM daily_usage WHERE day < ?", ((datetime.now(timezone.utc).date() - timedelta(days=89)).isoformat(),))
|
|
71
|
+
connection.execute("INSERT OR IGNORE INTO daily_usage(day, client_id) VALUES(?, ?)", (today, client_id))
|
|
72
|
+
column = columns.get(kind)
|
|
73
|
+
if column:
|
|
74
|
+
connection.execute(f"UPDATE daily_usage SET requests=requests+1, {column}={column}+1 WHERE day=? AND client_id=?", (today, client_id))
|
|
75
|
+
else:
|
|
76
|
+
connection.execute("UPDATE daily_usage SET requests=requests+1 WHERE day=? AND client_id=?", (today, client_id))
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _read_usage(client_id: str) -> dict[str, int | str]:
|
|
80
|
+
path = _usage_db()
|
|
81
|
+
today = datetime.now(timezone.utc).date().isoformat()
|
|
82
|
+
if not path.exists():
|
|
83
|
+
return {"date_utc": today, "requests": 0, "successful": 0, "client_errors": 0, "rate_limited": 0}
|
|
84
|
+
with closing(sqlite3.connect(path, timeout=5)) as connection:
|
|
85
|
+
row = connection.execute(
|
|
86
|
+
"SELECT day, requests, successful, client_errors, rate_limited FROM daily_usage WHERE day=? AND client_id=?",
|
|
87
|
+
(today, client_id),
|
|
88
|
+
).fetchone()
|
|
89
|
+
if not row:
|
|
90
|
+
return {"date_utc": today, "requests": 0, "successful": 0, "client_errors": 0, "rate_limited": 0}
|
|
91
|
+
if row[0] != today:
|
|
92
|
+
return {"date_utc": today, "requests": 0, "successful": 0, "client_errors": 0, "rate_limited": 0}
|
|
93
|
+
return {"date_utc": row[0], "requests": row[1], "successful": row[2], "client_errors": row[3], "rate_limited": row[4]}
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class Handler(BaseHTTPRequestHandler):
|
|
97
|
+
server_version = "TAV/0.1"
|
|
98
|
+
|
|
99
|
+
def log_message(self, fmt: str, *args: object) -> None:
|
|
100
|
+
# Avoid default logging of client address and request details.
|
|
101
|
+
return
|
|
102
|
+
|
|
103
|
+
def _send(self, status: int, data: dict) -> None:
|
|
104
|
+
body = json.dumps(data, ensure_ascii=False, separators=(",", ":")).encode("utf-8")
|
|
105
|
+
self.send_response(status)
|
|
106
|
+
self.send_header("Content-Type", "application/json; charset=utf-8")
|
|
107
|
+
self.send_header("Content-Length", str(len(body)))
|
|
108
|
+
self.send_header("Cache-Control", "no-store")
|
|
109
|
+
self.end_headers()
|
|
110
|
+
self.wfile.write(body)
|
|
111
|
+
|
|
112
|
+
def _error(self, status: int, code: str, message: str) -> None:
|
|
113
|
+
self._send(status, {"error": {"code": code, "message": message}})
|
|
114
|
+
|
|
115
|
+
def do_GET(self) -> None:
|
|
116
|
+
if self.path == "/healthz":
|
|
117
|
+
self._send(200, {"status": "ok"})
|
|
118
|
+
return
|
|
119
|
+
if self.path == "/v1/usage":
|
|
120
|
+
client_id = _api_key_id(self.headers.get("Authorization", ""))
|
|
121
|
+
if client_id is None:
|
|
122
|
+
self._error(401, "unauthorized", "A valid bearer API key is required")
|
|
123
|
+
return
|
|
124
|
+
self._send(200, _read_usage(client_id))
|
|
125
|
+
return
|
|
126
|
+
self._error(404, "not_found", "Route not found")
|
|
127
|
+
|
|
128
|
+
def do_POST(self) -> None:
|
|
129
|
+
if self.path != "/v1/evaluate":
|
|
130
|
+
self._error(404, "not_found", "Route not found")
|
|
131
|
+
return
|
|
132
|
+
client_id = _api_key_id(self.headers.get("Authorization", ""))
|
|
133
|
+
if client_id is None:
|
|
134
|
+
self._error(401, "unauthorized", "A valid bearer API key is required")
|
|
135
|
+
return
|
|
136
|
+
|
|
137
|
+
now = time.monotonic()
|
|
138
|
+
if client_id not in _requests and len(_requests) >= MAX_RATE_CLIENTS:
|
|
139
|
+
_requests.pop(next(iter(_requests)))
|
|
140
|
+
history = _requests[client_id]
|
|
141
|
+
while history and now - history[0] >= RATE_WINDOW:
|
|
142
|
+
history.popleft()
|
|
143
|
+
if len(history) >= RATE_LIMIT:
|
|
144
|
+
_record_usage(client_id, "rate_limited")
|
|
145
|
+
self._error(429, "rate_limited", "Request limit exceeded")
|
|
146
|
+
return
|
|
147
|
+
history.append(now)
|
|
148
|
+
|
|
149
|
+
try:
|
|
150
|
+
length = int(self.headers.get("Content-Length", "-1"))
|
|
151
|
+
except ValueError:
|
|
152
|
+
_record_usage(client_id, "client_error")
|
|
153
|
+
self._error(400, "invalid_content_length", "Content-Length must be an integer")
|
|
154
|
+
return
|
|
155
|
+
if length < 0:
|
|
156
|
+
_record_usage(client_id, "client_error")
|
|
157
|
+
self._error(400, "content_length_required", "Content-Length is required")
|
|
158
|
+
return
|
|
159
|
+
if length > MAX_BODY:
|
|
160
|
+
_record_usage(client_id, "client_error")
|
|
161
|
+
self._error(413, "body_too_large", "Request body exceeds 64 KiB")
|
|
162
|
+
return
|
|
163
|
+
content_type = self.headers.get("Content-Type", "").split(";", 1)[0].strip().lower()
|
|
164
|
+
if content_type != "application/json":
|
|
165
|
+
_record_usage(client_id, "client_error")
|
|
166
|
+
self._error(415, "unsupported_media_type", "Content-Type must be application/json")
|
|
167
|
+
return
|
|
168
|
+
raw = self.rfile.read(length)
|
|
169
|
+
try:
|
|
170
|
+
payload = json.loads(raw.decode("utf-8"))
|
|
171
|
+
except (UnicodeDecodeError, json.JSONDecodeError):
|
|
172
|
+
_record_usage(client_id, "client_error")
|
|
173
|
+
self._error(400, "invalid_json", "Body must be valid UTF-8 JSON")
|
|
174
|
+
return
|
|
175
|
+
try:
|
|
176
|
+
result = evaluate(payload)
|
|
177
|
+
except RequestError as exc:
|
|
178
|
+
_record_usage(client_id, "client_error")
|
|
179
|
+
self._error(400, "invalid_request", str(exc))
|
|
180
|
+
return
|
|
181
|
+
_record_usage(client_id, "success")
|
|
182
|
+
self._send(200, result)
|
|
183
|
+
|
|
184
|
+
def do_PUT(self) -> None:
|
|
185
|
+
self._error(405, "method_not_allowed", "Method not allowed")
|
|
186
|
+
|
|
187
|
+
do_PATCH = do_PUT
|
|
188
|
+
do_DELETE = do_PUT
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def main() -> None:
|
|
192
|
+
key = os.environ.get("TAV_API_KEY", "")
|
|
193
|
+
key_map = os.environ.get("TAV_API_KEYS", "")
|
|
194
|
+
if key_map:
|
|
195
|
+
try:
|
|
196
|
+
parsed_keys = json.loads(key_map)
|
|
197
|
+
except json.JSONDecodeError:
|
|
198
|
+
raise SystemExit("TAV_API_KEYS must be JSON mapping client IDs to SHA-256 digests") from None
|
|
199
|
+
if not isinstance(parsed_keys, dict) or not parsed_keys:
|
|
200
|
+
raise SystemExit("TAV_API_KEYS must contain at least one client key")
|
|
201
|
+
if any(
|
|
202
|
+
not isinstance(client_id, str)
|
|
203
|
+
or not re.fullmatch(r"[A-Za-z0-9_-]{1,48}", client_id)
|
|
204
|
+
or not isinstance(digest, str)
|
|
205
|
+
or not re.fullmatch(r"[a-fA-F0-9]{64}", digest)
|
|
206
|
+
for client_id, digest in parsed_keys.items()
|
|
207
|
+
):
|
|
208
|
+
raise SystemExit("TAV_API_KEYS entries must use valid client IDs and 64-character SHA-256 hex digests")
|
|
209
|
+
elif len(key) < 24:
|
|
210
|
+
raise SystemExit("Set TAV_API_KEYS or TAV_API_KEY (at least 24 characters for single-key mode)")
|
|
211
|
+
host = os.environ.get("TAV_HOST", "127.0.0.1")
|
|
212
|
+
port = int(os.environ.get("TAV_PORT", "8080"))
|
|
213
|
+
server = ThreadingHTTPServer((host, port), Handler)
|
|
214
|
+
print(f"TAV listening on {host}:{port}; request bodies are not logged")
|
|
215
|
+
try:
|
|
216
|
+
server.serve_forever()
|
|
217
|
+
except KeyboardInterrupt:
|
|
218
|
+
pass
|
|
219
|
+
finally:
|
|
220
|
+
server.server_close()
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
if __name__ == "__main__":
|
|
224
|
+
main()
|
tav_core.py
ADDED
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"""Deterministic grading against caller-supplied rubric data only."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from difflib import SequenceMatcher
|
|
8
|
+
from decimal import Decimal, InvalidOperation
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class RequestError(ValueError):
|
|
13
|
+
"""Input does not satisfy the API contract."""
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
_NUMBER_RE = re.compile(r"\d{1,3}(?:,\d{3})+(?:\.\d+)?|\d+(?:[.,]\d+)?")
|
|
17
|
+
_ALLOWED_RUBRIC = {"required_concepts", "accepted_synonyms", "numeric_requirements", "required_count"}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _normalize(value: str) -> str:
|
|
21
|
+
return re.sub(r"[\W_]+", "", value.casefold(), flags=re.UNICODE)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _ensure_utf8(value: Any) -> None:
|
|
25
|
+
if isinstance(value, str):
|
|
26
|
+
try:
|
|
27
|
+
value.encode("utf-8")
|
|
28
|
+
except UnicodeEncodeError:
|
|
29
|
+
raise RequestError("all input strings must be valid Unicode") from None
|
|
30
|
+
elif isinstance(value, dict):
|
|
31
|
+
for key, item in value.items():
|
|
32
|
+
_ensure_utf8(key)
|
|
33
|
+
_ensure_utf8(item)
|
|
34
|
+
elif isinstance(value, list):
|
|
35
|
+
for item in value:
|
|
36
|
+
_ensure_utf8(item)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _match_concept(concept: str, synonyms: list[str], answer: str) -> bool:
|
|
40
|
+
candidates = [concept, *synonyms]
|
|
41
|
+
norm_answer = _normalize(answer)
|
|
42
|
+
for candidate in candidates:
|
|
43
|
+
norm = _normalize(candidate)
|
|
44
|
+
if norm and norm in norm_answer:
|
|
45
|
+
return True
|
|
46
|
+
|
|
47
|
+
# Conservative typo tolerance: only for candidates of 5+ characters,
|
|
48
|
+
# using sliding windows to avoid comparing against the whole paragraph.
|
|
49
|
+
for candidate in candidates:
|
|
50
|
+
norm = _normalize(candidate)
|
|
51
|
+
if len(norm) < 5:
|
|
52
|
+
continue
|
|
53
|
+
window_count = max(1, len(norm_answer) - len(norm) + 1)
|
|
54
|
+
stride = max(1, (window_count + 255) // 256)
|
|
55
|
+
for start in range(0, window_count, stride):
|
|
56
|
+
window = norm_answer[start : start + len(norm)]
|
|
57
|
+
if SequenceMatcher(None, norm, window).ratio() >= 0.88:
|
|
58
|
+
return True
|
|
59
|
+
return False
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _validate_rubric(rubric: Any) -> tuple[list[str], dict[str, list[str]], list[dict[str, str]], int]:
|
|
63
|
+
if not isinstance(rubric, dict):
|
|
64
|
+
raise RequestError("rubric must be an object")
|
|
65
|
+
extra = set(rubric) - _ALLOWED_RUBRIC
|
|
66
|
+
if extra:
|
|
67
|
+
raise RequestError(f"unknown rubric field: {sorted(extra)[0]}")
|
|
68
|
+
|
|
69
|
+
concepts = rubric.get("required_concepts")
|
|
70
|
+
if not isinstance(concepts, list) or not 1 <= len(concepts) <= 50:
|
|
71
|
+
raise RequestError("required_concepts must be an array with 1 to 50 items")
|
|
72
|
+
if any(not isinstance(item, str) or not item.strip() or len(item) > 120 for item in concepts):
|
|
73
|
+
raise RequestError("each concept must be a non-empty string of at most 120 characters")
|
|
74
|
+
if len({_normalize(x) for x in concepts}) != len(concepts):
|
|
75
|
+
raise RequestError("required_concepts must be unique after normalization")
|
|
76
|
+
|
|
77
|
+
if len(json.dumps(rubric, ensure_ascii=False).encode("utf-8")) > 8192:
|
|
78
|
+
raise RequestError("rubric must not exceed 8 KiB")
|
|
79
|
+
|
|
80
|
+
synonym_map = rubric.get("accepted_synonyms", {})
|
|
81
|
+
if not isinstance(synonym_map, dict) or set(synonym_map) - set(concepts):
|
|
82
|
+
raise RequestError("accepted_synonyms must map existing concepts to arrays")
|
|
83
|
+
synonyms: dict[str, list[str]] = {}
|
|
84
|
+
term_owners: dict[str, str] = {}
|
|
85
|
+
for concept in concepts:
|
|
86
|
+
term_owners[_normalize(concept)] = concept
|
|
87
|
+
for concept, values in synonym_map.items():
|
|
88
|
+
if not isinstance(values, list) or len(values) > 20:
|
|
89
|
+
raise RequestError("each synonym list must contain at most 20 strings")
|
|
90
|
+
if any(not isinstance(value, str) or not value.strip() or len(value) > 120 for value in values):
|
|
91
|
+
raise RequestError("synonyms must be non-empty strings of at most 120 characters")
|
|
92
|
+
synonyms[concept] = values
|
|
93
|
+
for value in values:
|
|
94
|
+
normalized = _normalize(value)
|
|
95
|
+
owner = term_owners.get(normalized)
|
|
96
|
+
if owner is not None and owner != concept:
|
|
97
|
+
raise RequestError("a concept or synonym cannot belong to multiple concepts")
|
|
98
|
+
term_owners[normalized] = concept
|
|
99
|
+
if sum(1 + len(synonyms.get(concept, [])) for concept in concepts) > 100:
|
|
100
|
+
raise RequestError("required concepts and synonyms may contain at most 100 total terms")
|
|
101
|
+
|
|
102
|
+
numeric = rubric.get("numeric_requirements", [])
|
|
103
|
+
if not isinstance(numeric, list) or len(numeric) > 20:
|
|
104
|
+
raise RequestError("numeric_requirements must be an array with at most 20 items")
|
|
105
|
+
normalized_numeric: list[dict[str, str]] = []
|
|
106
|
+
for item in numeric:
|
|
107
|
+
if not isinstance(item, dict) or set(item) - {"value", "unit", "tolerance"} or "value" not in item:
|
|
108
|
+
raise RequestError("each numeric requirement needs value and optional unit/tolerance")
|
|
109
|
+
try:
|
|
110
|
+
expected_decimal = Decimal(str(item["value"]))
|
|
111
|
+
tolerance_decimal = Decimal(str(item.get("tolerance", "0")))
|
|
112
|
+
if not expected_decimal.is_finite() or not tolerance_decimal.is_finite() or tolerance_decimal < 0:
|
|
113
|
+
raise InvalidOperation
|
|
114
|
+
value = str(expected_decimal)
|
|
115
|
+
tolerance = str(tolerance_decimal)
|
|
116
|
+
except (InvalidOperation, ValueError):
|
|
117
|
+
raise RequestError("numeric value and non-negative tolerance must be decimals") from None
|
|
118
|
+
unit = item.get("unit", "")
|
|
119
|
+
if not isinstance(unit, str) or len(unit) > 24:
|
|
120
|
+
raise RequestError("numeric unit must be a string of at most 24 characters")
|
|
121
|
+
normalized_numeric.append({"value": value, "unit": unit, "tolerance": tolerance})
|
|
122
|
+
|
|
123
|
+
required_count = rubric.get("required_count", len(concepts))
|
|
124
|
+
if not isinstance(required_count, int) or isinstance(required_count, bool) or not 1 <= required_count <= len(concepts):
|
|
125
|
+
raise RequestError("required_count must be an integer from 1 to the number of concepts")
|
|
126
|
+
return concepts, synonyms, normalized_numeric, required_count
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _parse_number(raw: str) -> Decimal | None:
|
|
130
|
+
if "," in raw:
|
|
131
|
+
if re.fullmatch(r"\d{1,3}(?:,\d{3})+(?:\.\d+)?", raw):
|
|
132
|
+
raw = raw.replace(",", "")
|
|
133
|
+
elif raw.count(",") == 1 and "." not in raw:
|
|
134
|
+
raw = raw.replace(",", ".")
|
|
135
|
+
else:
|
|
136
|
+
return None
|
|
137
|
+
try:
|
|
138
|
+
value = Decimal(raw)
|
|
139
|
+
except InvalidOperation:
|
|
140
|
+
return None
|
|
141
|
+
return value if value.is_finite() else None
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _check_numbers(answer: str, requirements: list[dict[str, str]]) -> list[dict[str, Any]]:
|
|
145
|
+
checks = []
|
|
146
|
+
for requirement in requirements:
|
|
147
|
+
expected = Decimal(requirement["value"])
|
|
148
|
+
tolerance = Decimal(requirement["tolerance"])
|
|
149
|
+
unit = requirement["unit"]
|
|
150
|
+
observed: list[Decimal] = []
|
|
151
|
+
candidates = list(_NUMBER_RE.finditer(answer))
|
|
152
|
+
for match in candidates:
|
|
153
|
+
suffix = answer[match.end() : match.end() + len(unit) + 8].lstrip() if unit else ""
|
|
154
|
+
if unit:
|
|
155
|
+
unit_match = re.match(re.escape(unit) + r"(?![\w])", suffix, flags=re.IGNORECASE | re.UNICODE)
|
|
156
|
+
if not unit_match:
|
|
157
|
+
continue
|
|
158
|
+
actual = _parse_number(match.group())
|
|
159
|
+
if actual is not None:
|
|
160
|
+
observed.append(actual)
|
|
161
|
+
passed = bool(observed) and all(abs(actual - expected) <= tolerance for actual in observed)
|
|
162
|
+
checks.append({
|
|
163
|
+
"expected": requirement["value"],
|
|
164
|
+
"unit": unit,
|
|
165
|
+
"tolerance": requirement["tolerance"],
|
|
166
|
+
"observed": [str(value) for value in observed],
|
|
167
|
+
"passed": passed,
|
|
168
|
+
})
|
|
169
|
+
return checks
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def evaluate(payload: Any) -> dict[str, Any]:
|
|
173
|
+
_ensure_utf8(payload)
|
|
174
|
+
if not isinstance(payload, dict) or set(payload) != {"rubric", "answer"}:
|
|
175
|
+
raise RequestError("request must contain exactly rubric and answer")
|
|
176
|
+
answer = payload["answer"]
|
|
177
|
+
if not isinstance(answer, str) or not answer.strip() or len(answer) > 8000:
|
|
178
|
+
raise RequestError("answer must be a string of at most 8000 characters")
|
|
179
|
+
rubric = payload["rubric"]
|
|
180
|
+
concepts, synonyms, numeric, required_count = _validate_rubric(rubric)
|
|
181
|
+
|
|
182
|
+
matched = [concept for concept in concepts if _match_concept(concept, synonyms.get(concept, []), answer)]
|
|
183
|
+
missing = [concept for concept in concepts if concept not in matched]
|
|
184
|
+
score = min(len(matched) / required_count, 1.0)
|
|
185
|
+
numeric_checks = _check_numbers(answer, numeric)
|
|
186
|
+
if any(not check["passed"] for check in numeric_checks):
|
|
187
|
+
score *= 0.5
|
|
188
|
+
score = round(score, 3)
|
|
189
|
+
verdict = "correct" if score >= 0.8 else "partial" if score >= 0.4 else "wrong"
|
|
190
|
+
notes = []
|
|
191
|
+
if len(matched) < required_count:
|
|
192
|
+
notes.append(f"required_count={required_count}; matched={len(matched)}")
|
|
193
|
+
if any(not check["passed"] for check in numeric_checks):
|
|
194
|
+
notes.append("One or more numeric requirements did not match.")
|
|
195
|
+
return {
|
|
196
|
+
"api_version": "v1",
|
|
197
|
+
"status": "graded",
|
|
198
|
+
"score": score,
|
|
199
|
+
"verdict": verdict,
|
|
200
|
+
"matched_concepts": matched,
|
|
201
|
+
"missing_concepts": missing,
|
|
202
|
+
"numeric_checks": numeric_checks,
|
|
203
|
+
"notes": notes,
|
|
204
|
+
"review_required": True,
|
|
205
|
+
"limitations": ["Keyword-based assistive review only; not an official exam grade."],
|
|
206
|
+
}
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: technical-answer-validator
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: MCP and REST tools for deterministic rubric-based technical answer review
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Requires-Python: >=3.10
|
|
8
|
+
Requires-Dist: mcp==2.2.0
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
|
|
11
|
+
# Technical Answer Validator
|
|
12
|
+
|
|
13
|
+
<!-- mcp-name: io.github.christofer566/technical-answer-validator -->
|
|
14
|
+
|
|
15
|
+
A tiny, deterministic answer review tool for AI agents, available over **MCP stdio** and as a REST API. Each caller supplies the concepts, accepted synonyms, numeric requirements, and answer text for a single evaluation. It does not include a question bank or answer corpus.
|
|
16
|
+
|
|
17
|
+
This is an **assistive practice tool**, not an official certification exam grader. Keyword matching can miss semantically correct paraphrases and can accept misleading surface matches. Users should review the supplied rubric and every result.
|
|
18
|
+
|
|
19
|
+
## Run locally
|
|
20
|
+
|
|
21
|
+
Python 3.10+; the REST server uses the standard library. The MCP adapter uses the official Python SDK 2.2.0.
|
|
22
|
+
|
|
23
|
+
```powershell
|
|
24
|
+
$env:TAV_API_KEY = "replace-with-a-long-random-secret-at-least-24-characters"
|
|
25
|
+
python -m tav_api
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
The service listens on `127.0.0.1:8080` by default. To change it, set `TAV_HOST` and `TAV_PORT`. Local single-key mode refuses to start without a `TAV_API_KEY` of at least 24 characters. For deployment, create a distinct key per caller with `python scripts/create_api_key.py CLIENT_ID` and configure `TAV_API_KEYS` as a JSON object mapping each client ID to the generated SHA-256 digest. Store the one-time raw key with that client; do not store or commit it in this repository. When `TAV_API_KEYS` is set, it takes precedence over `TAV_API_KEY`.
|
|
29
|
+
|
|
30
|
+
## MCP for AI agents
|
|
31
|
+
|
|
32
|
+
Install the pinned MCP SDK 2.2.0 in a virtual environment from the committed lockfile:
|
|
33
|
+
|
|
34
|
+
```powershell
|
|
35
|
+
uv sync --locked
|
|
36
|
+
.\.venv\Scripts\Activate.ps1
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
The stdio MCP server exposes one tool: `evaluate_answer(rubric, answer)`. Configure an MCP host with the absolute path to the environment's Python and `mcp_server.py`. Example Claude Desktop configuration (replace paths):
|
|
40
|
+
|
|
41
|
+
```json
|
|
42
|
+
{
|
|
43
|
+
"mcpServers": {
|
|
44
|
+
"technical-answer-validator": {
|
|
45
|
+
"command": "C:\\path\\to\\technical-answer-validator\\.venv\\Scripts\\python.exe",
|
|
46
|
+
"args": ["C:\\path\\to\\technical-answer-validator\\mcp_server.py"]
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
After publishing to PyPI, run with `uvx --from technical-answer-validator tav-mcp`. For local development use the `.venv` Python plus `mcp_server.py`. For Codex CLI or another MCP host, use its stdio server configuration with that command and script path. Restart the host, then ask it to list tools and call `evaluate_answer`. The stdio transport is local to the user's agent host and needs no internet endpoint or API key.
|
|
53
|
+
|
|
54
|
+
The Codex TOML template is `codex-mcp-config.example.toml`; the Claude Desktop JSON template is `claude-mcp-config.example.json`. Replace both placeholder paths with absolute paths. Merge the block into the host configuration; do not overwrite other MCP servers or global settings.
|
|
55
|
+
|
|
56
|
+
## Request
|
|
57
|
+
|
|
58
|
+
`POST /v1/evaluate`
|
|
59
|
+
|
|
60
|
+
```json
|
|
61
|
+
{
|
|
62
|
+
"rubric": {
|
|
63
|
+
"required_concepts": ["isolation", "lockout tag"],
|
|
64
|
+
"accepted_synonyms": {"isolation": ["energy isolation"]},
|
|
65
|
+
"numeric_requirements": [],
|
|
66
|
+
"required_count": 2
|
|
67
|
+
},
|
|
68
|
+
"answer": "Apply energy isolation and attach a lockout tag."
|
|
69
|
+
}
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
`accepted_synonyms` keys must exactly match a concept. `numeric_requirements` is an optional array such as `[{"value":"10","unit":"kN","tolerance":"0"}]`. Numbers in an answer are only checked when explicit requirements are provided. Concept score is matched concepts / required_count (defaults to the number of concepts), capped at 1.0; numeric failures apply a 50% score penalty. The verdict thresholds are `correct >= 0.8`, `partial >= 0.4`, otherwise `wrong`.
|
|
73
|
+
|
|
74
|
+
## Response and errors
|
|
75
|
+
|
|
76
|
+
Successful requests return `api_version`, `status`, `score`, `verdict`, matched/missing concepts, numeric check details, and `review_required: true`.
|
|
77
|
+
|
|
78
|
+
Errors use JSON `{ "error": { "code": "...", "message": "..." } }`. Statuses include 400 (invalid request), 401 (missing/invalid key), 404, 405, 413 (body over 64 KiB), and 429 (over 60 requests/minute per client ID). The rate limit is 60 requests/minute per authenticated client ID, in memory, and resets when the process restarts. Daily request/success/client-error/rate-limited counters are persisted in SQLite without answer text and expire after 90 days. `GET /v1/usage` returns only the caller's current UTC-day counts. Retain and back up the usage volume as desired; it contains client IDs and aggregates only.
|
|
79
|
+
|
|
80
|
+
## Privacy and deployment limits
|
|
81
|
+
|
|
82
|
+
The server does not log request bodies or answers. It stores daily counts keyed by client ID and request timestamps in process memory for REST rate limiting. The **stdio MCP option runs locally inside the agent host** and sends no requests to this HTTP server. The REST API is containerized and keeps usage counters in a persistent volume. Before public service, terminate TLS at a reverse proxy, set proxy-level rate/concurrency limits, deploy from a secret manager, monitor the host, and publish a data-retention/contact policy. The app-level per-client rate limit resets on restart and is not a substitute for edge controls.
|
|
83
|
+
|
|
84
|
+
## Verify
|
|
85
|
+
|
|
86
|
+
```powershell
|
|
87
|
+
python -m unittest discover -s tests -v
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
The OpenAPI contract is in `openapi.yaml`; the draft official MCP Registry descriptor is `server.json`, with publication steps in `PUBLISHING.md`. Run locally with Docker Compose after copying `.env.example` to `.env` and adding a private key; Compose publishes the service only on loopback, so configure an HTTPS reverse proxy separately. `compose.yaml` persists aggregate usage in a named volume and applies a read-only root filesystem, dropped Linux capabilities, and resource limits.
|
|
91
|
+
|
|
92
|
+
These tests check API and grading behavior; they do not establish professional exam accuracy.
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
mcp_server.py,sha256=ERHVEgOxppUkWT2lrJzABozeyOU0A5UIHu2vI78xkfM,1617
|
|
2
|
+
tav_api.py,sha256=x9_J_l2TElqynRT0pyjKvsK94t4sFVinws_-c76B2EU,9735
|
|
3
|
+
tav_core.py,sha256=iRidyFalhizO5bLYAhoDZ9NlEdak6HOXOBq_7rDaejw,9208
|
|
4
|
+
technical_answer_validator-0.1.0.dist-info/METADATA,sha256=orfG2Z25RmqTdvEjM6877rs-YPdarGh0cG9jzG4vRD0,5931
|
|
5
|
+
technical_answer_validator-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
6
|
+
technical_answer_validator-0.1.0.dist-info/entry_points.txt,sha256=JmuLu79ZYGv5TmvlG7JBOIwa8BN8bkSDTvIc8OfzPVA,67
|
|
7
|
+
technical_answer_validator-0.1.0.dist-info/licenses/LICENSE,sha256=cRMNPhwVrOILz60oLjerOJC23KvqHr1EEJFBKCMOenI,1062
|
|
8
|
+
technical_answer_validator-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 JAMUS
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the "Software"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions:
|
|
6
|
+
|
|
7
|
+
The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software.
|
|
8
|
+
|
|
9
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|