technical-answer-validator 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- technical_answer_validator-0.1.0/.dockerignore +8 -0
- technical_answer_validator-0.1.0/.env.example +6 -0
- technical_answer_validator-0.1.0/.github/workflows/ci.yml +18 -0
- technical_answer_validator-0.1.0/.github/workflows/publish-container.yml +39 -0
- technical_answer_validator-0.1.0/.github/workflows/publish-pypi.yml +23 -0
- technical_answer_validator-0.1.0/.gitignore +8 -0
- technical_answer_validator-0.1.0/Dockerfile +17 -0
- technical_answer_validator-0.1.0/LICENSE +9 -0
- technical_answer_validator-0.1.0/NOTICE.md +7 -0
- technical_answer_validator-0.1.0/PKG-INFO +92 -0
- technical_answer_validator-0.1.0/PUBLISHING.md +31 -0
- technical_answer_validator-0.1.0/README.md +82 -0
- technical_answer_validator-0.1.0/claude-mcp-config.example.json +8 -0
- technical_answer_validator-0.1.0/codex-mcp-config.example.toml +8 -0
- technical_answer_validator-0.1.0/compose.yaml +29 -0
- technical_answer_validator-0.1.0/mcp_server.py +46 -0
- technical_answer_validator-0.1.0/openapi.yaml +137 -0
- technical_answer_validator-0.1.0/pyproject.toml +37 -0
- technical_answer_validator-0.1.0/requirements.txt +1 -0
- technical_answer_validator-0.1.0/scripts/create_api_key.py +15 -0
- technical_answer_validator-0.1.0/server.json +20 -0
- technical_answer_validator-0.1.0/tav_api.py +224 -0
- technical_answer_validator-0.1.0/tav_core.py +206 -0
- technical_answer_validator-0.1.0/tests/test_api.py +103 -0
- technical_answer_validator-0.1.0/tests/test_mcp_server.py +51 -0
- technical_answer_validator-0.1.0/tests/test_tav.py +105 -0
- technical_answer_validator-0.1.0/uv.lock +917 -0
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
# Local single-key mode only. Generate a different secret with a password manager.
|
|
2
|
+
# For a hosted multi-caller deployment, configure TAV_API_KEYS in the secret manager instead.
|
|
3
|
+
TAV_API_KEY=replace-with-a-unique-random-secret-at-least-32-characters
|
|
4
|
+
TAV_HOST=127.0.0.1
|
|
5
|
+
TAV_PORT=8080
|
|
6
|
+
# TAV_API_KEYS={"client-id":"64-character-sha256-hex-digest"}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
pull_request:
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
test:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
steps:
|
|
11
|
+
- uses: actions/checkout@v4
|
|
12
|
+
- uses: astral-sh/setup-uv@v6
|
|
13
|
+
with:
|
|
14
|
+
enable-cache: true
|
|
15
|
+
- run: uv sync --locked
|
|
16
|
+
- run: uv run python -m unittest discover -s tests -v
|
|
17
|
+
- run: uv run python -m py_compile tav_core.py tav_api.py mcp_server.py scripts/create_api_key.py
|
|
18
|
+
- run: docker build --tag technical-answer-validator:ci .
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
name: Publish REST API image
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags:
|
|
6
|
+
- "v*"
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
jobs:
|
|
12
|
+
build-and-push:
|
|
13
|
+
runs-on: ubuntu-latest
|
|
14
|
+
permissions:
|
|
15
|
+
contents: read
|
|
16
|
+
packages: write
|
|
17
|
+
steps:
|
|
18
|
+
- uses: actions/checkout@v4
|
|
19
|
+
- uses: docker/setup-qemu-action@v3
|
|
20
|
+
- uses: docker/setup-buildx-action@v3
|
|
21
|
+
- uses: docker/login-action@v3
|
|
22
|
+
with:
|
|
23
|
+
registry: ghcr.io
|
|
24
|
+
username: ${{ github.actor }}
|
|
25
|
+
password: ${{ secrets.GITHUB_TOKEN }}
|
|
26
|
+
- uses: docker/metadata-action@v5
|
|
27
|
+
id: meta
|
|
28
|
+
with:
|
|
29
|
+
images: ghcr.io/${{ github.repository }}-api
|
|
30
|
+
tags: |
|
|
31
|
+
type=ref,event=tag
|
|
32
|
+
type=raw,value=latest
|
|
33
|
+
- uses: docker/build-push-action@v6
|
|
34
|
+
with:
|
|
35
|
+
context: .
|
|
36
|
+
push: true
|
|
37
|
+
platforms: linux/amd64,linux/arm64
|
|
38
|
+
tags: ${{ steps.meta.outputs.tags }}
|
|
39
|
+
labels: ${{ steps.meta.outputs.labels }}
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
name: Publish MCP package to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags:
|
|
6
|
+
- "v*"
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
jobs:
|
|
12
|
+
publish:
|
|
13
|
+
runs-on: ubuntu-latest
|
|
14
|
+
environment:
|
|
15
|
+
name: pypi
|
|
16
|
+
url: https://pypi.org/project/technical-answer-validator/
|
|
17
|
+
permissions:
|
|
18
|
+
id-token: write
|
|
19
|
+
steps:
|
|
20
|
+
- uses: actions/checkout@v4
|
|
21
|
+
- uses: astral-sh/setup-uv@v6
|
|
22
|
+
- run: uv build
|
|
23
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
FROM python:3.12-slim
|
|
2
|
+
|
|
3
|
+
ENV PYTHONDONTWRITEBYTECODE=1 \
|
|
4
|
+
PYTHONUNBUFFERED=1 \
|
|
5
|
+
TAV_HOST=0.0.0.0 \
|
|
6
|
+
TAV_PORT=8080 \
|
|
7
|
+
TAV_USAGE_DB=/data/usage.sqlite3
|
|
8
|
+
|
|
9
|
+
WORKDIR /app
|
|
10
|
+
COPY tav_api.py tav_core.py ./
|
|
11
|
+
RUN useradd --system --uid 10001 --no-create-home tav && mkdir /data && chown 10001:10001 /data
|
|
12
|
+
VOLUME ["/data"]
|
|
13
|
+
USER 10001:10001
|
|
14
|
+
EXPOSE 8080
|
|
15
|
+
HEALTHCHECK --interval=30s --timeout=3s --start-period=5s --retries=3 \
|
|
16
|
+
CMD python -c "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8080/healthz', timeout=2)"
|
|
17
|
+
CMD ["python", "-m", "tav_api"]
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 JAMUS
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the "Software"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions:
|
|
6
|
+
|
|
7
|
+
The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software.
|
|
8
|
+
|
|
9
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# Notice
|
|
2
|
+
|
|
3
|
+
Technical Answer Validator is an experimental, keyword-based assistive tool. It does not provide official exam grades, professional advice, or a guarantee of correctness. Operators and callers must review rubrics and outputs before relying on them.
|
|
4
|
+
|
|
5
|
+
No SafetyExam question bank, answer corpus, study database, or personal study history is included in this project. The caller supplies the rubric and answer for every request.
|
|
6
|
+
|
|
7
|
+
The project is licensed under the MIT License; see `LICENSE`. The release owner should still confirm that all included project code and assets are eligible for that license before public distribution.
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: technical-answer-validator
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: MCP and REST tools for deterministic rubric-based technical answer review
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Requires-Python: >=3.10
|
|
8
|
+
Requires-Dist: mcp==2.2.0
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
|
|
11
|
+
# Technical Answer Validator
|
|
12
|
+
|
|
13
|
+
<!-- mcp-name: io.github.christofer566/technical-answer-validator -->
|
|
14
|
+
|
|
15
|
+
A tiny, deterministic answer review tool for AI agents, available over **MCP stdio** and as a REST API. Each caller supplies the concepts, accepted synonyms, numeric requirements, and answer text for a single evaluation. It does not include a question bank or answer corpus.
|
|
16
|
+
|
|
17
|
+
This is an **assistive practice tool**, not an official certification exam grader. Keyword matching can miss semantically correct paraphrases and can accept misleading surface matches. Users should review the supplied rubric and every result.
|
|
18
|
+
|
|
19
|
+
## Run locally
|
|
20
|
+
|
|
21
|
+
Python 3.10+; the REST server uses the standard library. The MCP adapter uses the official Python SDK 2.2.0.
|
|
22
|
+
|
|
23
|
+
```powershell
|
|
24
|
+
$env:TAV_API_KEY = "replace-with-a-long-random-secret-at-least-24-characters"
|
|
25
|
+
python -m tav_api
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
The service listens on `127.0.0.1:8080` by default. To change it, set `TAV_HOST` and `TAV_PORT`. Local single-key mode refuses to start without a `TAV_API_KEY` of at least 24 characters. For deployment, create a distinct key per caller with `python scripts/create_api_key.py CLIENT_ID` and configure `TAV_API_KEYS` as a JSON object mapping each client ID to the generated SHA-256 digest. Store the one-time raw key with that client; do not store or commit it in this repository. When `TAV_API_KEYS` is set, it takes precedence over `TAV_API_KEY`.
|
|
29
|
+
|
|
30
|
+
## MCP for AI agents
|
|
31
|
+
|
|
32
|
+
Install the pinned MCP SDK 2.2.0 in a virtual environment from the committed lockfile:
|
|
33
|
+
|
|
34
|
+
```powershell
|
|
35
|
+
uv sync --locked
|
|
36
|
+
.\.venv\Scripts\Activate.ps1
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
The stdio MCP server exposes one tool: `evaluate_answer(rubric, answer)`. Configure an MCP host with the absolute path to the environment's Python and `mcp_server.py`. Example Claude Desktop configuration (replace paths):
|
|
40
|
+
|
|
41
|
+
```json
|
|
42
|
+
{
|
|
43
|
+
"mcpServers": {
|
|
44
|
+
"technical-answer-validator": {
|
|
45
|
+
"command": "C:\\path\\to\\technical-answer-validator\\.venv\\Scripts\\python.exe",
|
|
46
|
+
"args": ["C:\\path\\to\\technical-answer-validator\\mcp_server.py"]
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
After publishing to PyPI, run with `uvx --from technical-answer-validator tav-mcp`. For local development use the `.venv` Python plus `mcp_server.py`. For Codex CLI or another MCP host, use its stdio server configuration with that command and script path. Restart the host, then ask it to list tools and call `evaluate_answer`. The stdio transport is local to the user's agent host and needs no internet endpoint or API key.
|
|
53
|
+
|
|
54
|
+
The Codex TOML template is `codex-mcp-config.example.toml`; the Claude Desktop JSON template is `claude-mcp-config.example.json`. Replace both placeholder paths with absolute paths. Merge the block into the host configuration; do not overwrite other MCP servers or global settings.
|
|
55
|
+
|
|
56
|
+
## Request
|
|
57
|
+
|
|
58
|
+
`POST /v1/evaluate`
|
|
59
|
+
|
|
60
|
+
```json
|
|
61
|
+
{
|
|
62
|
+
"rubric": {
|
|
63
|
+
"required_concepts": ["isolation", "lockout tag"],
|
|
64
|
+
"accepted_synonyms": {"isolation": ["energy isolation"]},
|
|
65
|
+
"numeric_requirements": [],
|
|
66
|
+
"required_count": 2
|
|
67
|
+
},
|
|
68
|
+
"answer": "Apply energy isolation and attach a lockout tag."
|
|
69
|
+
}
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
`accepted_synonyms` keys must exactly match a concept. `numeric_requirements` is an optional array such as `[{"value":"10","unit":"kN","tolerance":"0"}]`. Numbers in an answer are only checked when explicit requirements are provided. Concept score is matched concepts / required_count (defaults to the number of concepts), capped at 1.0; numeric failures apply a 50% score penalty. The verdict thresholds are `correct >= 0.8`, `partial >= 0.4`, otherwise `wrong`.
|
|
73
|
+
|
|
74
|
+
## Response and errors
|
|
75
|
+
|
|
76
|
+
Successful requests return `api_version`, `status`, `score`, `verdict`, matched/missing concepts, numeric check details, and `review_required: true`.
|
|
77
|
+
|
|
78
|
+
Errors use JSON `{ "error": { "code": "...", "message": "..." } }`. Statuses include 400 (invalid request), 401 (missing/invalid key), 404, 405, 413 (body over 64 KiB), and 429 (over 60 requests/minute per client ID). The rate limit is 60 requests/minute per authenticated client ID, in memory, and resets when the process restarts. Daily request/success/client-error/rate-limited counters are persisted in SQLite without answer text and expire after 90 days. `GET /v1/usage` returns only the caller's current UTC-day counts. Retain and back up the usage volume as desired; it contains client IDs and aggregates only.
|
|
79
|
+
|
|
80
|
+
## Privacy and deployment limits
|
|
81
|
+
|
|
82
|
+
The server does not log request bodies or answers. It stores daily counts keyed by client ID and request timestamps in process memory for REST rate limiting. The **stdio MCP option runs locally inside the agent host** and sends no requests to this HTTP server. The REST API is containerized and keeps usage counters in a persistent volume. Before public service, terminate TLS at a reverse proxy, set proxy-level rate/concurrency limits, deploy from a secret manager, monitor the host, and publish a data-retention/contact policy. The app-level per-client rate limit resets on restart and is not a substitute for edge controls.
|
|
83
|
+
|
|
84
|
+
## Verify
|
|
85
|
+
|
|
86
|
+
```powershell
|
|
87
|
+
python -m unittest discover -s tests -v
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
The OpenAPI contract is in `openapi.yaml`; the draft official MCP Registry descriptor is `server.json`, with publication steps in `PUBLISHING.md`. Run locally with Docker Compose after copying `.env.example` to `.env` and adding a private key; Compose publishes the service only on loopback, so configure an HTTPS reverse proxy separately. `compose.yaml` persists aggregate usage in a named volume and applies a read-only root filesystem, dropped Linux capabilities, and resource limits.
|
|
91
|
+
|
|
92
|
+
These tests check API and grading behavior; they do not establish professional exam accuracy.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# Public release runbook
|
|
2
|
+
|
|
3
|
+
The package contains two delivery paths: local MCP stdio for an agent that can spawn a local process, and a REST API container for hosted agents. No public artifact has been uploaded and no service is deployed by this preparation.
|
|
4
|
+
|
|
5
|
+
## Local MCP package
|
|
6
|
+
|
|
7
|
+
1. MIT was selected and is included in `LICENSE` and the Python package metadata. Confirm that all included source/assets are eligible for MIT before public release.
|
|
8
|
+
2. The public GitHub repository exists at `https://github.com/Christofer566/technical-answer-validator`; source is on `main`. Keep `.venv`, `.uv-cache`, `.env`, usage databases, and generated state out of commits.
|
|
9
|
+
3. Build and inspect the wheel/sdist: `uv build`; test the wheel in a clean virtual environment; `uv run python -m unittest discover -s tests -v`.
|
|
10
|
+
4. PyPI JSON lookup returned HTTP 404 on 2026-10-02. In the PyPI account, register a pending GitHub Actions Trusted Publisher for project `technical-answer-validator`, owner `Christofer566`, repository `technical-answer-validator`, workflow `.github/workflows/publish-pypi.yml`, environment `pypi`. The GitHub `pypi` environment already exists. Publish version `0.1.0` by pushing a `v0.1.0` tag only after ownership/IP approval.
|
|
11
|
+
5. Confirm the built PyPI README contains the exact `mcp-name` marker and that package metadata, repository URL, license, and author are correct.
|
|
12
|
+
6. Validate `server.json` against the current MCP Registry schema and verify the GitHub/PyPI namespace, package name, and version. Then publish with the official MCP Registry CLI using the owner's GitHub identity.
|
|
13
|
+
7. Test installation from the published package in a clean environment with `uvx --from technical-answer-validator tav-mcp`; connect it in target agent hosts and call `evaluate_answer`.
|
|
14
|
+
|
|
15
|
+
## Hosted REST API
|
|
16
|
+
|
|
17
|
+
1. Choose the VPS/host, public domain, TLS reverse proxy, budget ceiling, operator contact, and retention policy. These account/domain choices are not present in the workspace.
|
|
18
|
+
2. Set `TAV_API_KEYS` in the host's secret manager to `{client_id: sha256_hex}` entries generated by `python scripts/create_api_key.py CLIENT_ID`. Deliver each raw key privately to its intended agent operator; only the digest belongs in deployment configuration.
|
|
19
|
+
3. Push an approved `v0.1.0` tag to build and publish `ghcr.io/<owner>/<repo>-api` via `.github/workflows/publish-container.yml`. Pull that image on the host or build directly from the reviewed repository; deploy with `compose.yaml` and retain the `/data` volume. Keep the app bound to loopback behind the HTTPS proxy. Add proxy-level connection/time/body limits and rate limiting; app-level rate buckets reset on restart.
|
|
20
|
+
4. Monitor `/healthz`, per-customer `/v1/usage`, host availability, disk space for SQLite aggregates, and spend. Define key rotation/revocation by removing the corresponding digest and restarting/redeploying.
|
|
21
|
+
5. Verify external HTTPS from a separate network with unauthenticated rejection, valid client isolation, 64 KiB limit, rate limits, no answer logging, and accurate OpenAPI docs before announcing a URL.
|
|
22
|
+
|
|
23
|
+
## Blocks that require owner-controlled decisions/actions
|
|
24
|
+
|
|
25
|
+
- Public source repository exists and `server.json` points to it.
|
|
26
|
+
- PyPI name was not registered at the 2026-10-02 lookup (HTTP 404). PyPI account authentication is required to register its pending publisher; no package has been uploaded.
|
|
27
|
+
- MIT is selected. The owner must still confirm project ownership and third-party rights for included files before publication.
|
|
28
|
+
- No production URL, host account, budget, or service operator contact has been supplied; therefore the container remains local and has not been deployed.
|
|
29
|
+
- The tool's keyword/numeric decisions are not validated against independent subject-matter gold data. Position it as assistive agent feedback, not a trustworthy final grader.
|
|
30
|
+
|
|
31
|
+
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
# Technical Answer Validator
|
|
2
|
+
|
|
3
|
+
<!-- mcp-name: io.github.christofer566/technical-answer-validator -->
|
|
4
|
+
|
|
5
|
+
A tiny, deterministic answer review tool for AI agents, available over **MCP stdio** and as a REST API. Each caller supplies the concepts, accepted synonyms, numeric requirements, and answer text for a single evaluation. It does not include a question bank or answer corpus.
|
|
6
|
+
|
|
7
|
+
This is an **assistive practice tool**, not an official certification exam grader. Keyword matching can miss semantically correct paraphrases and can accept misleading surface matches. Users should review the supplied rubric and every result.
|
|
8
|
+
|
|
9
|
+
## Run locally
|
|
10
|
+
|
|
11
|
+
Python 3.10+; the REST server uses the standard library. The MCP adapter uses the official Python SDK 2.2.0.
|
|
12
|
+
|
|
13
|
+
```powershell
|
|
14
|
+
$env:TAV_API_KEY = "replace-with-a-long-random-secret-at-least-24-characters"
|
|
15
|
+
python -m tav_api
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
The service listens on `127.0.0.1:8080` by default. To change it, set `TAV_HOST` and `TAV_PORT`. Local single-key mode refuses to start without a `TAV_API_KEY` of at least 24 characters. For deployment, create a distinct key per caller with `python scripts/create_api_key.py CLIENT_ID` and configure `TAV_API_KEYS` as a JSON object mapping each client ID to the generated SHA-256 digest. Store the one-time raw key with that client; do not store or commit it in this repository. When `TAV_API_KEYS` is set, it takes precedence over `TAV_API_KEY`.
|
|
19
|
+
|
|
20
|
+
## MCP for AI agents
|
|
21
|
+
|
|
22
|
+
Install the pinned MCP SDK 2.2.0 in a virtual environment from the committed lockfile:
|
|
23
|
+
|
|
24
|
+
```powershell
|
|
25
|
+
uv sync --locked
|
|
26
|
+
.\.venv\Scripts\Activate.ps1
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
The stdio MCP server exposes one tool: `evaluate_answer(rubric, answer)`. Configure an MCP host with the absolute path to the environment's Python and `mcp_server.py`. Example Claude Desktop configuration (replace paths):
|
|
30
|
+
|
|
31
|
+
```json
|
|
32
|
+
{
|
|
33
|
+
"mcpServers": {
|
|
34
|
+
"technical-answer-validator": {
|
|
35
|
+
"command": "C:\\path\\to\\technical-answer-validator\\.venv\\Scripts\\python.exe",
|
|
36
|
+
"args": ["C:\\path\\to\\technical-answer-validator\\mcp_server.py"]
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
After publishing to PyPI, run with `uvx --from technical-answer-validator tav-mcp`. For local development use the `.venv` Python plus `mcp_server.py`. For Codex CLI or another MCP host, use its stdio server configuration with that command and script path. Restart the host, then ask it to list tools and call `evaluate_answer`. The stdio transport is local to the user's agent host and needs no internet endpoint or API key.
|
|
43
|
+
|
|
44
|
+
The Codex TOML template is `codex-mcp-config.example.toml`; the Claude Desktop JSON template is `claude-mcp-config.example.json`. Replace both placeholder paths with absolute paths. Merge the block into the host configuration; do not overwrite other MCP servers or global settings.
|
|
45
|
+
|
|
46
|
+
## Request
|
|
47
|
+
|
|
48
|
+
`POST /v1/evaluate`
|
|
49
|
+
|
|
50
|
+
```json
|
|
51
|
+
{
|
|
52
|
+
"rubric": {
|
|
53
|
+
"required_concepts": ["isolation", "lockout tag"],
|
|
54
|
+
"accepted_synonyms": {"isolation": ["energy isolation"]},
|
|
55
|
+
"numeric_requirements": [],
|
|
56
|
+
"required_count": 2
|
|
57
|
+
},
|
|
58
|
+
"answer": "Apply energy isolation and attach a lockout tag."
|
|
59
|
+
}
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
`accepted_synonyms` keys must exactly match a concept. `numeric_requirements` is an optional array such as `[{"value":"10","unit":"kN","tolerance":"0"}]`. Numbers in an answer are only checked when explicit requirements are provided. Concept score is matched concepts / required_count (defaults to the number of concepts), capped at 1.0; numeric failures apply a 50% score penalty. The verdict thresholds are `correct >= 0.8`, `partial >= 0.4`, otherwise `wrong`.
|
|
63
|
+
|
|
64
|
+
## Response and errors
|
|
65
|
+
|
|
66
|
+
Successful requests return `api_version`, `status`, `score`, `verdict`, matched/missing concepts, numeric check details, and `review_required: true`.
|
|
67
|
+
|
|
68
|
+
Errors use JSON `{ "error": { "code": "...", "message": "..." } }`. Statuses include 400 (invalid request), 401 (missing/invalid key), 404, 405, 413 (body over 64 KiB), and 429 (over 60 requests/minute per client ID). The rate limit is 60 requests/minute per authenticated client ID, in memory, and resets when the process restarts. Daily request/success/client-error/rate-limited counters are persisted in SQLite without answer text and expire after 90 days. `GET /v1/usage` returns only the caller's current UTC-day counts. Retain and back up the usage volume as desired; it contains client IDs and aggregates only.
|
|
69
|
+
|
|
70
|
+
## Privacy and deployment limits
|
|
71
|
+
|
|
72
|
+
The server does not log request bodies or answers. It stores daily counts keyed by client ID and request timestamps in process memory for REST rate limiting. The **stdio MCP option runs locally inside the agent host** and sends no requests to this HTTP server. The REST API is containerized and keeps usage counters in a persistent volume. Before public service, terminate TLS at a reverse proxy, set proxy-level rate/concurrency limits, deploy from a secret manager, monitor the host, and publish a data-retention/contact policy. The app-level per-client rate limit resets on restart and is not a substitute for edge controls.
|
|
73
|
+
|
|
74
|
+
## Verify
|
|
75
|
+
|
|
76
|
+
```powershell
|
|
77
|
+
python -m unittest discover -s tests -v
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
The OpenAPI contract is in `openapi.yaml`; the draft official MCP Registry descriptor is `server.json`, with publication steps in `PUBLISHING.md`. Run locally with Docker Compose after copying `.env.example` to `.env` and adding a private key; Compose publishes the service only on loopback, so configure an HTTPS reverse proxy separately. `compose.yaml` persists aggregate usage in a named volume and applies a read-only root filesystem, dropped Linux capabilities, and resource limits.
|
|
81
|
+
|
|
82
|
+
These tests check API and grading behavior; they do not establish professional exam accuracy.
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# Merge this server block into the user's Codex MCP configuration.
|
|
2
|
+
# Replace the absolute path for the machine where the server is installed.
|
|
3
|
+
|
|
4
|
+
[mcp_servers.technical_answer_validator]
|
|
5
|
+
command = "C:/absolute/path/technical-answer-validator/.venv/Scripts/python.exe"
|
|
6
|
+
args = ["C:/absolute/path/technical-answer-validator/mcp_server.py"]
|
|
7
|
+
startup_timeout_sec = 30
|
|
8
|
+
tool_timeout_sec = 30
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
services:
|
|
2
|
+
tav-api:
|
|
3
|
+
build: .
|
|
4
|
+
image: technical-answer-validator:0.1.0
|
|
5
|
+
restart: unless-stopped
|
|
6
|
+
env_file: .env
|
|
7
|
+
ports:
|
|
8
|
+
- "127.0.0.1:8080:8080"
|
|
9
|
+
volumes:
|
|
10
|
+
- tav-usage:/data
|
|
11
|
+
read_only: true
|
|
12
|
+
tmpfs:
|
|
13
|
+
- /tmp:size=8m,noexec,nosuid
|
|
14
|
+
security_opt:
|
|
15
|
+
- no-new-privileges:true
|
|
16
|
+
cap_drop:
|
|
17
|
+
- ALL
|
|
18
|
+
pids_limit: 64
|
|
19
|
+
mem_limit: 128m
|
|
20
|
+
cpus: 0.50
|
|
21
|
+
healthcheck:
|
|
22
|
+
test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8080/healthz', timeout=2)"]
|
|
23
|
+
interval: 30s
|
|
24
|
+
timeout: 3s
|
|
25
|
+
retries: 3
|
|
26
|
+
start_period: 5s
|
|
27
|
+
|
|
28
|
+
volumes:
|
|
29
|
+
tav-usage:
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""MCP stdio server exposing the answer evaluator as an agent tool."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from mcp.server import MCPServer
|
|
10
|
+
|
|
11
|
+
from tav_core import RequestError, evaluate
|
|
12
|
+
|
|
13
|
+
logging.basicConfig(level=logging.WARNING, format="%(levelname)s %(message)s")
|
|
14
|
+
mcp = MCPServer(
|
|
15
|
+
"Technical Answer Validator",
|
|
16
|
+
instructions=(
|
|
17
|
+
"Evaluate a user's technical answer against a rubric supplied for this call. "
|
|
18
|
+
"Always pass the caller's rubric and answer. Explain that results are keyword-based "
|
|
19
|
+
"assistive feedback, never official exam grades; preserve review_required in your response."
|
|
20
|
+
),
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@mcp.tool()
|
|
25
|
+
def evaluate_answer(rubric: dict[str, Any], answer: str) -> dict[str, Any]:
|
|
26
|
+
"""Check an answer against caller-provided concepts, synonyms, and numeric requirements.
|
|
27
|
+
|
|
28
|
+
Rubric fields: required_concepts (1-50 strings), optional accepted_synonyms keyed by
|
|
29
|
+
concept, optional numeric_requirements [{value, unit?, tolerance?}], and optional
|
|
30
|
+
required_count. No question bank is bundled. Review the returned result; it is not
|
|
31
|
+
an official exam grade.
|
|
32
|
+
"""
|
|
33
|
+
try:
|
|
34
|
+
return evaluate({"rubric": rubric, "answer": answer})
|
|
35
|
+
except RequestError as exc:
|
|
36
|
+
# Return a structured error result so the agent can repair its inputs.
|
|
37
|
+
return {"status": "invalid_request", "error": str(exc), "review_required": True}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def main() -> None:
|
|
41
|
+
# FastMCP stdio transport uses stdout for protocol messages; keep diagnostics on stderr.
|
|
42
|
+
mcp.run(transport="stdio")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
if __name__ == "__main__":
|
|
46
|
+
main()
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
openapi: 3.1.0
|
|
2
|
+
info:
|
|
3
|
+
title: Technical Answer Validator API
|
|
4
|
+
version: 0.1.0
|
|
5
|
+
description: Deterministic, rubric-based assistive review for AI agents. Not an official exam grade.
|
|
6
|
+
servers:
|
|
7
|
+
- url: http://127.0.0.1:8080
|
|
8
|
+
description: Local development only; production must use HTTPS.
|
|
9
|
+
security:
|
|
10
|
+
- bearerAuth: []
|
|
11
|
+
paths:
|
|
12
|
+
/healthz:
|
|
13
|
+
get:
|
|
14
|
+
security: []
|
|
15
|
+
summary: Health check
|
|
16
|
+
responses:
|
|
17
|
+
"200":
|
|
18
|
+
description: Healthy
|
|
19
|
+
/v1/evaluate:
|
|
20
|
+
post:
|
|
21
|
+
summary: Evaluate one answer against a caller-supplied rubric
|
|
22
|
+
operationId: evaluateAnswer
|
|
23
|
+
requestBody:
|
|
24
|
+
required: true
|
|
25
|
+
content:
|
|
26
|
+
application/json:
|
|
27
|
+
schema:
|
|
28
|
+
$ref: "#/components/schemas/EvaluationRequest"
|
|
29
|
+
example:
|
|
30
|
+
rubric:
|
|
31
|
+
required_concepts: [isolation, lockout tag]
|
|
32
|
+
accepted_synonyms:
|
|
33
|
+
isolation: [energy isolation]
|
|
34
|
+
numeric_requirements: []
|
|
35
|
+
required_count: 2
|
|
36
|
+
answer: Apply energy isolation and attach a lockout tag.
|
|
37
|
+
responses:
|
|
38
|
+
"200":
|
|
39
|
+
description: Evaluation result; review is always required
|
|
40
|
+
content:
|
|
41
|
+
application/json:
|
|
42
|
+
schema:
|
|
43
|
+
$ref: "#/components/schemas/EvaluationResult"
|
|
44
|
+
"400":
|
|
45
|
+
$ref: "#/components/responses/ClientError"
|
|
46
|
+
"401":
|
|
47
|
+
$ref: "#/components/responses/Unauthorized"
|
|
48
|
+
"413":
|
|
49
|
+
description: Body exceeds 64 KiB
|
|
50
|
+
"415":
|
|
51
|
+
description: Content-Type must be application/json
|
|
52
|
+
"429":
|
|
53
|
+
description: Per-client request limit exceeded
|
|
54
|
+
/v1/usage:
|
|
55
|
+
get:
|
|
56
|
+
summary: Read this API key's UTC daily aggregate usage
|
|
57
|
+
operationId: getUsage
|
|
58
|
+
responses:
|
|
59
|
+
"200":
|
|
60
|
+
description: Aggregated counters; answer content is never included
|
|
61
|
+
content:
|
|
62
|
+
application/json:
|
|
63
|
+
schema:
|
|
64
|
+
type: object
|
|
65
|
+
properties:
|
|
66
|
+
date_utc: {type: string, format: date}
|
|
67
|
+
requests: {type: integer}
|
|
68
|
+
successful: {type: integer}
|
|
69
|
+
client_errors: {type: integer}
|
|
70
|
+
rate_limited: {type: integer}
|
|
71
|
+
"401":
|
|
72
|
+
$ref: "#/components/responses/Unauthorized"
|
|
73
|
+
components:
|
|
74
|
+
securitySchemes:
|
|
75
|
+
bearerAuth:
|
|
76
|
+
type: http
|
|
77
|
+
scheme: bearer
|
|
78
|
+
bearerFormat: API key
|
|
79
|
+
schemas:
|
|
80
|
+
NumericRequirement:
|
|
81
|
+
type: object
|
|
82
|
+
additionalProperties: false
|
|
83
|
+
required: [value]
|
|
84
|
+
properties:
|
|
85
|
+
value: {type: [string, number]}
|
|
86
|
+
unit: {type: string, maxLength: 24}
|
|
87
|
+
tolerance: {type: [string, number], default: 0}
|
|
88
|
+
Rubric:
|
|
89
|
+
type: object
|
|
90
|
+
additionalProperties: false
|
|
91
|
+
required: [required_concepts]
|
|
92
|
+
properties:
|
|
93
|
+
required_concepts:
|
|
94
|
+
type: array
|
|
95
|
+
minItems: 1
|
|
96
|
+
maxItems: 50
|
|
97
|
+
items: {type: string, minLength: 1, maxLength: 120}
|
|
98
|
+
accepted_synonyms:
|
|
99
|
+
type: object
|
|
100
|
+
additionalProperties:
|
|
101
|
+
type: array
|
|
102
|
+
maxItems: 20
|
|
103
|
+
items: {type: string, minLength: 1, maxLength: 120}
|
|
104
|
+
numeric_requirements:
|
|
105
|
+
type: array
|
|
106
|
+
maxItems: 20
|
|
107
|
+
items: {$ref: "#/components/schemas/NumericRequirement"}
|
|
108
|
+
required_count:
|
|
109
|
+
type: integer
|
|
110
|
+
minimum: 1
|
|
111
|
+
maximum: 50
|
|
112
|
+
EvaluationRequest:
|
|
113
|
+
type: object
|
|
114
|
+
additionalProperties: false
|
|
115
|
+
required: [rubric, answer]
|
|
116
|
+
properties:
|
|
117
|
+
rubric: {$ref: "#/components/schemas/Rubric"}
|
|
118
|
+
answer: {type: string, minLength: 1, maxLength: 8000}
|
|
119
|
+
EvaluationResult:
|
|
120
|
+
type: object
|
|
121
|
+
required: [api_version, status, score, verdict, matched_concepts, missing_concepts, numeric_checks, review_required, limitations]
|
|
122
|
+
properties:
|
|
123
|
+
api_version: {const: v1}
|
|
124
|
+
status: {const: graded}
|
|
125
|
+
score: {type: number, minimum: 0, maximum: 1}
|
|
126
|
+
verdict: {enum: [correct, partial, wrong]}
|
|
127
|
+
matched_concepts: {type: array, items: {type: string}}
|
|
128
|
+
missing_concepts: {type: array, items: {type: string}}
|
|
129
|
+
numeric_checks: {type: array, items: {type: object}}
|
|
130
|
+
notes: {type: array, items: {type: string}}
|
|
131
|
+
review_required: {const: true}
|
|
132
|
+
limitations: {type: array, items: {type: string}}
|
|
133
|
+
responses:
|
|
134
|
+
Unauthorized:
|
|
135
|
+
description: Missing or invalid bearer API key
|
|
136
|
+
ClientError:
|
|
137
|
+
description: Invalid request JSON or rubric
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.27,<2"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "technical-answer-validator"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "MCP and REST tools for deterministic rubric-based technical answer review"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
dependencies = ["mcp==2.2.0"]
|
|
14
|
+
|
|
15
|
+
[project.scripts]
|
|
16
|
+
tav-mcp = "mcp_server:main"
|
|
17
|
+
tav-api = "tav_api:main"
|
|
18
|
+
|
|
19
|
+
[tool.hatch.build.targets.wheel]
|
|
20
|
+
only-include = ["mcp_server.py", "tav_core.py", "tav_api.py"]
|
|
21
|
+
|
|
22
|
+
[tool.hatch.build.targets.wheel.force-include]
|
|
23
|
+
"mcp_server.py" = "mcp_server.py"
|
|
24
|
+
"tav_core.py" = "tav_core.py"
|
|
25
|
+
"tav_api.py" = "tav_api.py"
|
|
26
|
+
|
|
27
|
+
[tool.hatch.build.targets.sdist]
|
|
28
|
+
only-include = [
|
|
29
|
+
".dockerignore", ".env.example", ".gitignore", "Dockerfile", "NOTICE.md",
|
|
30
|
+
"PUBLISHING.md", "README.md", "claude-mcp-config.example.json",
|
|
31
|
+
"codex-mcp-config.example.toml", "compose.yaml", "mcp_server.py",
|
|
32
|
+
"openapi.yaml", "pyproject.toml", "requirements.txt", "server.json",
|
|
33
|
+
"tav_api.py", "tav_core.py", "uv.lock", ".github", "scripts", "tests"
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
[tool.pytest.ini_options]
|
|
37
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
mcp==2.2.0
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Generate a one-time API credential and its TAV_API_KEYS SHA-256 entry."""
|
|
2
|
+
import hashlib
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
import secrets
|
|
6
|
+
import sys
|
|
7
|
+
|
|
8
|
+
if len(sys.argv) != 2 or not re.fullmatch(r"[A-Za-z0-9_-]{1,48}", sys.argv[1]):
|
|
9
|
+
raise SystemExit("Usage: python scripts/create_api_key.py CLIENT_ID (letters, digits, _ or -; max 48)")
|
|
10
|
+
client_id = sys.argv[1]
|
|
11
|
+
token = "tav_" + secrets.token_urlsafe(36)
|
|
12
|
+
digest = hashlib.sha256(token.encode("utf-8")).hexdigest()
|
|
13
|
+
print("Save this API key now; it is shown only once:\n" + token)
|
|
14
|
+
print("Add this entry to your secret manager's TAV_API_KEYS JSON:")
|
|
15
|
+
print(json.dumps({client_id: digest}, separators=(",", ":")))
|