engineering-argument-language 3.2.3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eal/__init__.py +3 -0
- eal/abstractions.py +144 -0
- eal/acquisition_coordination.py +80 -0
- eal/api_load_methods.py +77 -0
- eal/aspic.py +478 -0
- eal/aspic_compiler.py +445 -0
- eal/aspic_export.py +309 -0
- eal/builtin_methods.py +115 -0
- eal/catalogue.py +384 -0
- eal/cli.py +204 -0
- eal/client_transports.py +69 -0
- eal/collection_identity.py +28 -0
- eal/collection_scheduler.py +72 -0
- eal/command_process.py +118 -0
- eal/command_supervisor.py +112 -0
- eal/composition.py +531 -0
- eal/credentials.py +33 -0
- eal/dialectic.py +321 -0
- eal/discovery.py +127 -0
- eal/evaluator.py +619 -0
- eal/expressions.py +404 -0
- eal/extensions.py +38 -0
- eal/formatter.py +156 -0
- eal/generated/EALLexer.py +434 -0
- eal/generated/EALParser.py +6559 -0
- eal/generated/EALVisitor.py +393 -0
- eal/generated/__init__.py +1 -0
- eal/host.py +179 -0
- eal/host_redaction.py +51 -0
- eal/knowledge.py +96 -0
- eal/limits.py +88 -0
- eal/mcp_guard.py +56 -0
- eal/methods.py +463 -0
- eal/model.py +266 -0
- eal/model_context.py +55 -0
- eal/modes.py +121 -0
- eal/observation_reuse.py +128 -0
- eal/operation_contracts.py +25 -0
- eal/operations.py +155 -0
- eal/packets.py +604 -0
- eal/parser.py +407 -0
- eal/planning.py +136 -0
- eal/propositions.py +231 -0
- eal/reachability.py +85 -0
- eal/reasoning/__init__.py +25 -0
- eal/reasoning/abductive.py +45 -0
- eal/reasoning/analogical.py +41 -0
- eal/reasoning/causal.py +46 -0
- eal/reasoning/counterfactual.py +64 -0
- eal/reasoning/deductive.py +72 -0
- eal/reasoning/inductive.py +29 -0
- eal/reasoning/strategy.py +18 -0
- eal/reasoning/structured.py +11 -0
- eal/reasoning/temporal.py +47 -0
- eal/reasoning/validation.py +90 -0
- eal/registered_assessment.py +181 -0
- eal/runtime.py +454 -0
- eal/sampled_negative.py +131 -0
- eal/scope_transfer.py +59 -0
- eal/semantics.py +656 -0
- eal/server.py +79 -0
- eal/server_auth.py +37 -0
- eal/server_settings.py +258 -0
- eal/source_printer.py +129 -0
- eal/store.py +272 -0
- eal/tool_acquisition.py +383 -0
- engineering_argument_language-3.2.3.dist-info/METADATA +188 -0
- engineering_argument_language-3.2.3.dist-info/RECORD +72 -0
- engineering_argument_language-3.2.3.dist-info/WHEEL +5 -0
- engineering_argument_language-3.2.3.dist-info/entry_points.txt +4 -0
- engineering_argument_language-3.2.3.dist-info/licenses/LICENSE +24 -0
- engineering_argument_language-3.2.3.dist-info/top_level.txt +1 -0
eal/store.py
ADDED
|
@@ -0,0 +1,272 @@
|
|
|
1
|
+
"""SQLite storage for immutable collection, observation and reasoning records."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterator
|
|
6
|
+
from contextlib import contextmanager
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import secrets
|
|
10
|
+
import sqlite3
|
|
11
|
+
import stat
|
|
12
|
+
from datetime import datetime, timezone
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any
|
|
15
|
+
from uuid import uuid4
|
|
16
|
+
|
|
17
|
+
from filelock import FileLock
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _check_private_directory(directory: Path) -> None:
|
|
21
|
+
details = directory.lstat()
|
|
22
|
+
if not stat.S_ISDIR(details.st_mode):
|
|
23
|
+
raise PermissionError("The run-store directory must be a regular directory")
|
|
24
|
+
if os.name == "posix" and (details.st_uid != os.getuid() or details.st_mode & 0o022):
|
|
25
|
+
raise PermissionError("The run-store directory must be owned by the process and not writable by others")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@contextmanager
|
|
29
|
+
def _store_initialisation_lock(database_path: Path, *, timeout: float = 30) -> Iterator[None]:
|
|
30
|
+
"""Coordinate store setup before SQLite or its private identity is opened.
|
|
31
|
+
|
|
32
|
+
Keep the lock file in place after release: unlinking it could let another
|
|
33
|
+
process acquire a different inode while a contender still holds this one.
|
|
34
|
+
The private parent excludes replacement by other operating-system users.
|
|
35
|
+
"""
|
|
36
|
+
_check_private_directory(database_path.parent)
|
|
37
|
+
path = database_path.with_name(database_path.name + ".initialisation.lock")
|
|
38
|
+
flags = os.O_RDWR | getattr(os, "O_NOFOLLOW", 0)
|
|
39
|
+
try:
|
|
40
|
+
descriptor = os.open(path, flags | os.O_CREAT | os.O_EXCL, 0o600)
|
|
41
|
+
except FileExistsError:
|
|
42
|
+
# Check before opening an existing object: opening a FIFO or device
|
|
43
|
+
# can block or have effects before descriptor validation is possible.
|
|
44
|
+
details = path.lstat()
|
|
45
|
+
if not stat.S_ISREG(details.st_mode):
|
|
46
|
+
raise PermissionError("The run-store initialisation lock must be a private regular 0600 file")
|
|
47
|
+
descriptor = os.open(path, flags)
|
|
48
|
+
try:
|
|
49
|
+
details = os.fstat(descriptor)
|
|
50
|
+
named = path.lstat()
|
|
51
|
+
if (not stat.S_ISREG(details.st_mode) or not stat.S_ISREG(named.st_mode)
|
|
52
|
+
or (details.st_dev, details.st_ino) != (named.st_dev, named.st_ino)
|
|
53
|
+
or stat.S_IMODE(details.st_mode) != 0o600
|
|
54
|
+
or (os.name == "posix" and details.st_uid != os.getuid())):
|
|
55
|
+
raise PermissionError("The run-store initialisation lock must be a private regular 0600 file")
|
|
56
|
+
finally:
|
|
57
|
+
os.close(descriptor)
|
|
58
|
+
with FileLock(path, timeout=timeout, mode=0o600):
|
|
59
|
+
# FileLock owns a separate descriptor. Check the pathname once more
|
|
60
|
+
# before touching storage; participating processes retain the file.
|
|
61
|
+
details = path.lstat()
|
|
62
|
+
if (not stat.S_ISREG(details.st_mode) or stat.S_IMODE(details.st_mode) != 0o600
|
|
63
|
+
or (os.name == "posix" and details.st_uid != os.getuid())):
|
|
64
|
+
raise PermissionError("The run-store initialisation lock must be a private regular 0600 file")
|
|
65
|
+
yield
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def utc_now() -> str:
|
|
69
|
+
return datetime.now(timezone.utc).isoformat().replace("+00:00", "Z")
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _private_binding_key(database_path: Path) -> bytes:
|
|
73
|
+
"""Atomically establish a private store-local identity key outside SQLite."""
|
|
74
|
+
parent = database_path.parent
|
|
75
|
+
_check_private_directory(parent)
|
|
76
|
+
key_path = database_path.with_name(database_path.name + ".binding-key")
|
|
77
|
+
temporary = key_path.with_name(key_path.name + "." + secrets.token_hex(16) + ".tmp")
|
|
78
|
+
descriptor = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL |
|
|
79
|
+
getattr(os, "O_NOFOLLOW", 0), 0o600)
|
|
80
|
+
try:
|
|
81
|
+
candidate = secrets.token_bytes(32)
|
|
82
|
+
with os.fdopen(descriptor, "wb") as stream:
|
|
83
|
+
stream.write(candidate)
|
|
84
|
+
stream.flush()
|
|
85
|
+
os.fsync(stream.fileno())
|
|
86
|
+
try:
|
|
87
|
+
os.link(temporary, key_path, follow_symlinks=False)
|
|
88
|
+
except FileExistsError:
|
|
89
|
+
pass
|
|
90
|
+
finally:
|
|
91
|
+
temporary.unlink(missing_ok=True)
|
|
92
|
+
descriptor = os.open(key_path, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0))
|
|
93
|
+
with os.fdopen(descriptor, "rb") as stream:
|
|
94
|
+
info = os.fstat(stream.fileno())
|
|
95
|
+
if not stat.S_ISREG(info.st_mode) or info.st_mode & 0o077:
|
|
96
|
+
raise PermissionError("The run-store binding key must be a private regular file")
|
|
97
|
+
if os.name == "posix" and info.st_uid != os.getuid():
|
|
98
|
+
raise PermissionError("The run-store binding key must be owned by the process")
|
|
99
|
+
key = stream.read(33)
|
|
100
|
+
if len(key) != 32:
|
|
101
|
+
raise ValueError("The run-store binding key must contain exactly 32 bytes")
|
|
102
|
+
return key
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _check_private_sqlite_files(database_path: Path, *, create_database: bool = False) -> None:
|
|
106
|
+
"""Refuse readable or replaceable SQLite files before opening the store."""
|
|
107
|
+
if create_database:
|
|
108
|
+
try:
|
|
109
|
+
descriptor = os.open(
|
|
110
|
+
database_path, os.O_RDWR | os.O_CREAT | os.O_EXCL |
|
|
111
|
+
getattr(os, "O_NOFOLLOW", 0), 0o600,
|
|
112
|
+
)
|
|
113
|
+
except FileExistsError:
|
|
114
|
+
pass
|
|
115
|
+
else:
|
|
116
|
+
os.close(descriptor)
|
|
117
|
+
for path in (database_path, database_path.with_name(database_path.name + "-wal"),
|
|
118
|
+
database_path.with_name(database_path.name + "-shm")):
|
|
119
|
+
try:
|
|
120
|
+
descriptor = os.open(path, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0))
|
|
121
|
+
except FileNotFoundError:
|
|
122
|
+
if path == database_path:
|
|
123
|
+
raise
|
|
124
|
+
continue
|
|
125
|
+
with os.fdopen(descriptor, "rb") as stream:
|
|
126
|
+
details = os.fstat(stream.fileno())
|
|
127
|
+
if (not stat.S_ISREG(details.st_mode) or stat.S_IMODE(details.st_mode) != 0o600
|
|
128
|
+
or (os.name == "posix" and details.st_uid != os.getuid())):
|
|
129
|
+
raise PermissionError(f"SQLite store file must be a private regular 0600 file: {path.name}")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
class RunStore:
|
|
133
|
+
def __init__(self, path: str | Path):
|
|
134
|
+
self.path = Path(path)
|
|
135
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
136
|
+
with _store_initialisation_lock(self.path):
|
|
137
|
+
self._binding_key = _private_binding_key(self.path)
|
|
138
|
+
_check_private_sqlite_files(self.path, create_database=True)
|
|
139
|
+
self._initialise_schema()
|
|
140
|
+
_check_private_sqlite_files(self.path)
|
|
141
|
+
|
|
142
|
+
def _initialise_schema(self) -> None:
|
|
143
|
+
"""Establish the current schema and backfill its index atomically."""
|
|
144
|
+
with self._connect() as connection:
|
|
145
|
+
connection.execute("PRAGMA journal_mode=WAL")
|
|
146
|
+
connection.execute("BEGIN IMMEDIATE")
|
|
147
|
+
connection.execute(
|
|
148
|
+
"CREATE TABLE IF NOT EXISTS records "
|
|
149
|
+
"(id TEXT PRIMARY KEY, kind TEXT NOT NULL, created_at TEXT NOT NULL, payload TEXT NOT NULL)"
|
|
150
|
+
)
|
|
151
|
+
connection.execute("CREATE INDEX IF NOT EXISTS records_kind ON records(kind, created_at)")
|
|
152
|
+
connection.execute(
|
|
153
|
+
"CREATE TABLE IF NOT EXISTS observation_index "
|
|
154
|
+
"(sequence INTEGER PRIMARY KEY, record_id TEXT NOT NULL UNIQUE, "
|
|
155
|
+
"evidence_id TEXT NOT NULL, environment TEXT NOT NULL, "
|
|
156
|
+
"request_digest TEXT NOT NULL, tool_binding_digest TEXT NOT NULL)"
|
|
157
|
+
)
|
|
158
|
+
connection.execute(
|
|
159
|
+
"CREATE INDEX IF NOT EXISTS observation_identity ON observation_index "
|
|
160
|
+
"(evidence_id, environment, request_digest, tool_binding_digest, sequence DESC)"
|
|
161
|
+
)
|
|
162
|
+
connection.execute(
|
|
163
|
+
"CREATE TABLE IF NOT EXISTS store_metadata "
|
|
164
|
+
"(key TEXT PRIMARY KEY, value TEXT NOT NULL)"
|
|
165
|
+
)
|
|
166
|
+
indexed = connection.execute(
|
|
167
|
+
"SELECT value FROM store_metadata WHERE key = 'observation-index-version'"
|
|
168
|
+
).fetchone()
|
|
169
|
+
if indexed is None:
|
|
170
|
+
# Existing stores predate the index. Scan once, including failed
|
|
171
|
+
# attempts which deliberately get no reusable index entry.
|
|
172
|
+
for record_id, encoded in connection.execute(
|
|
173
|
+
"SELECT id, payload FROM records WHERE kind = 'observation' ORDER BY rowid"
|
|
174
|
+
):
|
|
175
|
+
self._index_observation(connection, record_id, json.loads(encoded))
|
|
176
|
+
connection.execute(
|
|
177
|
+
"INSERT INTO store_metadata(key, value) VALUES ('observation-index-version', '1')"
|
|
178
|
+
)
|
|
179
|
+
elif indexed[0] != "1":
|
|
180
|
+
raise ValueError("Unsupported observation index version")
|
|
181
|
+
|
|
182
|
+
@contextmanager
|
|
183
|
+
def _connect(self) -> Iterator[sqlite3.Connection]:
|
|
184
|
+
"""Commit or roll back one operation and close its owned connection."""
|
|
185
|
+
_check_private_sqlite_files(self.path)
|
|
186
|
+
connection = sqlite3.connect(self.path, timeout=30)
|
|
187
|
+
try:
|
|
188
|
+
_check_private_sqlite_files(self.path)
|
|
189
|
+
with connection:
|
|
190
|
+
yield connection
|
|
191
|
+
finally:
|
|
192
|
+
connection.close()
|
|
193
|
+
|
|
194
|
+
def put(self, kind: str, payload: dict[str, Any], *, record_id: str | None = None) -> str:
|
|
195
|
+
record_id = record_id or str(uuid4())
|
|
196
|
+
encoded = json.dumps(payload, ensure_ascii=False, sort_keys=True, allow_nan=False)
|
|
197
|
+
with self._connect() as connection:
|
|
198
|
+
connection.execute(
|
|
199
|
+
"INSERT INTO records(id, kind, created_at, payload) VALUES (?, ?, ?, ?)",
|
|
200
|
+
(record_id, kind, utc_now(), encoded),
|
|
201
|
+
)
|
|
202
|
+
if kind == "observation":
|
|
203
|
+
self._index_observation(connection, record_id, payload)
|
|
204
|
+
return record_id
|
|
205
|
+
|
|
206
|
+
@staticmethod
|
|
207
|
+
def _index_observation(connection: sqlite3.Connection, record_id: str, payload: dict[str, Any]) -> None:
|
|
208
|
+
if payload.get("status") != "ok":
|
|
209
|
+
return
|
|
210
|
+
keys = (payload.get("evidence_id"), payload.get("environment"),
|
|
211
|
+
payload.get("request_digest"), payload.get("tool_binding_digest"))
|
|
212
|
+
if any(not isinstance(key, str) or not key for key in keys):
|
|
213
|
+
return
|
|
214
|
+
connection.execute(
|
|
215
|
+
"INSERT INTO observation_index(record_id, evidence_id, environment, request_digest, "
|
|
216
|
+
"tool_binding_digest) VALUES (?, ?, ?, ?, ?)", (record_id, *keys)
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
def put_batch(self, entries: list[tuple[str, dict[str, Any], str | None]]) -> list[str]:
|
|
220
|
+
"""Commit a collection and its observations as one transaction."""
|
|
221
|
+
if not entries:
|
|
222
|
+
raise ValueError("entries must not be empty")
|
|
223
|
+
ids = [record_id or str(uuid4()) for _, _, record_id in entries]
|
|
224
|
+
encoded = [json.dumps(payload, ensure_ascii=False, sort_keys=True, allow_nan=False)
|
|
225
|
+
for _, payload, _ in entries]
|
|
226
|
+
with self._connect() as connection:
|
|
227
|
+
for (kind, payload, _), record_id, value in zip(entries, ids, encoded):
|
|
228
|
+
connection.execute(
|
|
229
|
+
"INSERT INTO records(id, kind, created_at, payload) VALUES (?, ?, ?, ?)",
|
|
230
|
+
(record_id, kind, utc_now(), value),
|
|
231
|
+
)
|
|
232
|
+
if kind == "observation":
|
|
233
|
+
self._index_observation(connection, record_id, payload)
|
|
234
|
+
return ids
|
|
235
|
+
|
|
236
|
+
def find_observations(self, *, evidence_id: str, environment: str,
|
|
237
|
+
request_digest: str, tool_binding_digest: str,
|
|
238
|
+
limit: int = 100) -> list[dict[str, Any]]:
|
|
239
|
+
"""Return exact-acquisition candidates, newest first, for reuse."""
|
|
240
|
+
if not 1 <= limit <= 1000:
|
|
241
|
+
raise ValueError("limit must be between 1 and 1000")
|
|
242
|
+
with self._connect() as connection:
|
|
243
|
+
rows = connection.execute(
|
|
244
|
+
"SELECT records.payload FROM observation_index "
|
|
245
|
+
"JOIN records ON records.id = observation_index.record_id "
|
|
246
|
+
"WHERE evidence_id = ? AND environment = ? AND request_digest = ? "
|
|
247
|
+
"AND tool_binding_digest = ? ORDER BY sequence DESC LIMIT ?",
|
|
248
|
+
(evidence_id, environment, request_digest, tool_binding_digest, limit),
|
|
249
|
+
).fetchall()
|
|
250
|
+
return [json.loads(row[0]) for row in rows]
|
|
251
|
+
|
|
252
|
+
def get(self, record_id: str, *, kind: str | None = None) -> dict[str, Any]:
|
|
253
|
+
with self._connect() as connection:
|
|
254
|
+
row = connection.execute("SELECT kind, payload FROM records WHERE id = ?", (record_id,)).fetchone()
|
|
255
|
+
if row is None or (kind is not None and row[0] != kind):
|
|
256
|
+
raise KeyError(f"No {kind or 'stored'} record with ID {record_id!r}")
|
|
257
|
+
return json.loads(row[1])
|
|
258
|
+
|
|
259
|
+
def list(self, *, kind: str | None = None, limit: int = 100) -> list[dict[str, str]]:
|
|
260
|
+
if not 1 <= limit <= 1000:
|
|
261
|
+
raise ValueError("limit must be between 1 and 1000")
|
|
262
|
+
with self._connect() as connection:
|
|
263
|
+
if kind is None:
|
|
264
|
+
rows = connection.execute(
|
|
265
|
+
"SELECT id, kind, created_at FROM records ORDER BY rowid DESC LIMIT ?", (limit,)
|
|
266
|
+
).fetchall()
|
|
267
|
+
else:
|
|
268
|
+
rows = connection.execute(
|
|
269
|
+
"SELECT id, kind, created_at FROM records WHERE kind = ? ORDER BY rowid DESC LIMIT ?",
|
|
270
|
+
(kind, limit),
|
|
271
|
+
).fetchall()
|
|
272
|
+
return [dict(zip(("id", "kind", "created_at"), row)) for row in rows]
|
eal/tool_acquisition.py
ADDED
|
@@ -0,0 +1,383 @@
|
|
|
1
|
+
"""Trusted tool bindings and bounded observation acquisition adapters.
|
|
2
|
+
|
|
3
|
+
The operator registry selects an adapter. EAL source supplies JSON input and
|
|
4
|
+
cannot choose executable paths or file locations. These adapters are not a
|
|
5
|
+
process sandbox.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import dataclasses
|
|
11
|
+
import hashlib
|
|
12
|
+
import hmac
|
|
13
|
+
import math
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import re
|
|
17
|
+
import selectors
|
|
18
|
+
import stat
|
|
19
|
+
import subprocess
|
|
20
|
+
import time
|
|
21
|
+
import tomllib
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Any, Mapping
|
|
24
|
+
|
|
25
|
+
from .command_process import supervised_command
|
|
26
|
+
|
|
27
|
+
MAX_JSON_DEPTH = 128
|
|
28
|
+
MAX_REQUEST_BYTES = 1024 * 1024
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def keyed_digest(secret: bytes, domain: bytes, value: Any) -> str:
|
|
32
|
+
"""Public configuration identity without an offline low-entropy value oracle."""
|
|
33
|
+
from .evaluator import canonical_digest
|
|
34
|
+
|
|
35
|
+
if not isinstance(secret, bytes) or len(secret) != 32:
|
|
36
|
+
raise ValueError("A private 32-byte store key is required")
|
|
37
|
+
if not isinstance(domain, bytes) or not domain:
|
|
38
|
+
raise ValueError("A domain separator is required")
|
|
39
|
+
# Frozen acquisition hash domain, independent of the authored source version.
|
|
40
|
+
# Retain these bytes so EAL/3 can reuse compatible observation-record/1 data.
|
|
41
|
+
return hmac.new(secret, b"EAL/2\0" + domain + b"\0" +
|
|
42
|
+
bytes.fromhex(canonical_digest(value)), hashlib.sha256).hexdigest()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def strict_json(text: str) -> Any:
|
|
46
|
+
def pairs(items):
|
|
47
|
+
result = {}
|
|
48
|
+
for key, value in items:
|
|
49
|
+
if key in result:
|
|
50
|
+
raise ValueError(f"Duplicate JSON key {key!r}")
|
|
51
|
+
result[key] = value
|
|
52
|
+
return result
|
|
53
|
+
|
|
54
|
+
def invalid_constant(value):
|
|
55
|
+
raise ValueError(f"Non-finite JSON value {value}")
|
|
56
|
+
|
|
57
|
+
def finite_float(value):
|
|
58
|
+
parsed = float(value)
|
|
59
|
+
if not math.isfinite(parsed):
|
|
60
|
+
raise ValueError(f"Non-finite JSON value {value}")
|
|
61
|
+
return parsed
|
|
62
|
+
|
|
63
|
+
try:
|
|
64
|
+
value = json.loads(text, object_pairs_hook=pairs, parse_constant=invalid_constant, parse_float=finite_float)
|
|
65
|
+
pending = [(value, 0)]
|
|
66
|
+
while pending:
|
|
67
|
+
item, depth = pending.pop()
|
|
68
|
+
if isinstance(item, (dict, list)):
|
|
69
|
+
if depth >= MAX_JSON_DEPTH:
|
|
70
|
+
raise ValueError(f"JSON nesting exceeds {MAX_JSON_DEPTH} container levels")
|
|
71
|
+
children = item.values() if isinstance(item, dict) else item
|
|
72
|
+
pending.extend((child, depth + 1) for child in children)
|
|
73
|
+
# JSON escapes can encode lone UTF-16 surrogates which cannot be stored
|
|
74
|
+
# or hashed as UTF-8. Reject these before adding values to run records.
|
|
75
|
+
json.dumps(value, ensure_ascii=False, allow_nan=False).encode("utf-8")
|
|
76
|
+
except UnicodeError as exc:
|
|
77
|
+
raise ValueError("JSON strings must contain valid Unicode scalar values") from exc
|
|
78
|
+
except RecursionError as exc:
|
|
79
|
+
raise ValueError("JSON nesting exceeds decoder resources") from exc
|
|
80
|
+
return value
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def bounded_path(workspace: Path, name: str) -> Path:
|
|
84
|
+
"""Resolve symlinks and refuse traversal outside the configured workspace."""
|
|
85
|
+
path = (workspace / name).resolve()
|
|
86
|
+
if not path.is_relative_to(workspace.resolve()):
|
|
87
|
+
raise ValueError(f"Path is outside the workspace: {name}")
|
|
88
|
+
return path
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclasses.dataclass(frozen=True)
|
|
92
|
+
class ToolBinding:
|
|
93
|
+
name: str
|
|
94
|
+
kind: str
|
|
95
|
+
version: str
|
|
96
|
+
argv: tuple[str, ...] = ()
|
|
97
|
+
path: str | None = None
|
|
98
|
+
timeout_seconds: float = 30.0
|
|
99
|
+
max_output_bytes: int = 1024 * 1024
|
|
100
|
+
env: Mapping[str, str] = dataclasses.field(default_factory=dict, repr=False)
|
|
101
|
+
pinned_files: tuple[tuple[str, str], ...] = ()
|
|
102
|
+
# The operator asserts this collector is read-only and independent of
|
|
103
|
+
# other simultaneous acquisitions. Arbitrary commands default to serial.
|
|
104
|
+
parallel_safe: bool = False
|
|
105
|
+
# None preserves full host inheritance; an explicit list passes only the
|
|
106
|
+
# named variables that exist, plus the operator's configured env overlay.
|
|
107
|
+
inherit_env: tuple[str, ...] | None = None
|
|
108
|
+
|
|
109
|
+
def effective_environment(self) -> dict[str, str]:
|
|
110
|
+
if self.inherit_env is None:
|
|
111
|
+
current = dict(os.environ)
|
|
112
|
+
else:
|
|
113
|
+
current = {name: os.environ[name] for name in self.inherit_env if name in os.environ}
|
|
114
|
+
current.update(self.env)
|
|
115
|
+
return current
|
|
116
|
+
|
|
117
|
+
def process_environment_digest(self, secret: bytes, *,
|
|
118
|
+
effective_env: Mapping[str, str] | None = None) -> str | None:
|
|
119
|
+
"""Identity of the effective command environment without exposing secrets."""
|
|
120
|
+
if self.kind != "command":
|
|
121
|
+
return None
|
|
122
|
+
current = self.effective_environment() if effective_env is None else dict(effective_env)
|
|
123
|
+
return keyed_digest(secret, b"process-environment", current)
|
|
124
|
+
|
|
125
|
+
def binding_digest(self, secret: bytes, *, workspace: Path | None = None) -> str:
|
|
126
|
+
"""Keyed registry identity, checking each operator-pinned file's bytes."""
|
|
127
|
+
if self.pinned_files and workspace is None:
|
|
128
|
+
raise ValueError("Pinned collector files require a workspace")
|
|
129
|
+
for name, expected in self.pinned_files:
|
|
130
|
+
path = Path(name)
|
|
131
|
+
path = path.resolve() if path.is_absolute() else bounded_path(workspace, name)
|
|
132
|
+
try:
|
|
133
|
+
descriptor = os.open(path, os.O_RDONLY | os.O_NONBLOCK)
|
|
134
|
+
with os.fdopen(descriptor, "rb") as stream:
|
|
135
|
+
details = os.fstat(stream.fileno())
|
|
136
|
+
if not stat.S_ISREG(details.st_mode) or details.st_size > 16 * 1024 * 1024:
|
|
137
|
+
raise ValueError("Pinned collector file must be a regular file at most 16 MiB")
|
|
138
|
+
digest = hashlib.sha256(stream.read(16 * 1024 * 1024 + 1)).hexdigest()
|
|
139
|
+
except OSError as exc:
|
|
140
|
+
raise ValueError("Pinned collector file is inaccessible") from exc
|
|
141
|
+
if digest != expected:
|
|
142
|
+
raise ValueError("Pinned collector file identity differs from operator configuration")
|
|
143
|
+
identity = {
|
|
144
|
+
"name": self.name, "kind": self.kind,
|
|
145
|
+
"version": self.version, "argv": list(self.argv),
|
|
146
|
+
"path": self.path, "timeout_seconds": self.timeout_seconds,
|
|
147
|
+
"max_output_bytes": self.max_output_bytes,
|
|
148
|
+
"env": dict(self.env), "pinned_files": [{"path": path, "sha256": digest}
|
|
149
|
+
for path, digest in self.pinned_files]}
|
|
150
|
+
# Scheduling is a host choice, not a property of a measurement.
|
|
151
|
+
if self.inherit_env is not None:
|
|
152
|
+
identity["inherit_env"] = sorted(self.inherit_env)
|
|
153
|
+
return keyed_digest(secret, b"tool-binding", identity)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
@dataclasses.dataclass(frozen=True)
|
|
157
|
+
class AcquisitionResult:
|
|
158
|
+
stdout: bytes = b""
|
|
159
|
+
stderr: bytes = b""
|
|
160
|
+
metadata: Mapping[str, Any] = dataclasses.field(default_factory=dict)
|
|
161
|
+
error: Exception | None = None
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
class ToolRegistry:
|
|
165
|
+
def __init__(self, bindings: Mapping[str, ToolBinding] | None = None):
|
|
166
|
+
self.bindings = dict(bindings or {})
|
|
167
|
+
|
|
168
|
+
@classmethod
|
|
169
|
+
def load(cls, path: str | Path) -> "ToolRegistry":
|
|
170
|
+
with Path(path).open("rb") as stream:
|
|
171
|
+
document = tomllib.load(stream)
|
|
172
|
+
if set(document) - {"tools"} or not isinstance(document.get("tools", {}), dict):
|
|
173
|
+
raise ValueError("The registry must contain only a [tools] table")
|
|
174
|
+
bindings = {}
|
|
175
|
+
allowed = {"kind", "version", "argv", "path", "timeout_seconds", "max_output_bytes", "env", "pinned_files", "parallel_safe", "inherit_env"}
|
|
176
|
+
for name, raw in document.get("tools", {}).items():
|
|
177
|
+
if not isinstance(raw, dict) or set(raw) - allowed:
|
|
178
|
+
raise ValueError(f"Unknown registry settings for {name}")
|
|
179
|
+
if raw.get("kind") not in _ADAPTERS:
|
|
180
|
+
raise ValueError(f"{name}: kind must be command or json_file")
|
|
181
|
+
if not isinstance(raw.get("version"), str) or not raw["version"]:
|
|
182
|
+
raise ValueError(f"{name}: a nonempty version is required")
|
|
183
|
+
argv = raw.get("argv", [])
|
|
184
|
+
if not isinstance(argv, list) or not all(isinstance(arg, str) and arg for arg in argv):
|
|
185
|
+
raise ValueError(f"{name}: argv must contain nonempty strings")
|
|
186
|
+
if raw["kind"] == "command" and (not argv or "path" in raw):
|
|
187
|
+
raise ValueError(f"{name}: command requires argv and does not use path")
|
|
188
|
+
if raw["kind"] == "json_file" and (not isinstance(raw.get("path"), str) or not raw["path"] or argv):
|
|
189
|
+
raise ValueError(f"{name}: json_file requires a path and does not use argv")
|
|
190
|
+
timeout = raw.get("timeout_seconds", 30.0)
|
|
191
|
+
if isinstance(timeout, bool) or not isinstance(timeout, (int, float)) or not math.isfinite(timeout) or not 0 < timeout <= 3600:
|
|
192
|
+
raise ValueError(f"{name}: timeout_seconds must be finite, positive and at most 3600")
|
|
193
|
+
limit = raw.get("max_output_bytes", 1024 * 1024)
|
|
194
|
+
if isinstance(limit, bool) or not isinstance(limit, int) or not 1 <= limit <= 16 * 1024 * 1024:
|
|
195
|
+
raise ValueError(f"{name}: max_output_bytes must be an integer from 1 through 16777216")
|
|
196
|
+
env = raw.get("env", {})
|
|
197
|
+
if not isinstance(env, dict) or not all(isinstance(k, str) and isinstance(v, str) for k, v in env.items()):
|
|
198
|
+
raise ValueError(f"{name}: env must map strings to strings")
|
|
199
|
+
parallel_safe = raw.get("parallel_safe", False)
|
|
200
|
+
if type(parallel_safe) is not bool:
|
|
201
|
+
raise ValueError(f"{name}: parallel_safe must be a boolean")
|
|
202
|
+
inherited = raw.get("inherit_env")
|
|
203
|
+
if inherited is not None:
|
|
204
|
+
if (raw["kind"] != "command" or not isinstance(inherited, list)
|
|
205
|
+
or len(inherited) > 64 or any(
|
|
206
|
+
not isinstance(item, str)
|
|
207
|
+
or re.fullmatch(r"[A-Za-z_][A-Za-z_0-9]*", item, re.ASCII) is None
|
|
208
|
+
for item in inherited)
|
|
209
|
+
or len(set(inherited)) != len(inherited)):
|
|
210
|
+
raise ValueError(f"{name}: inherit_env requires up to 64 distinct command environment variable names")
|
|
211
|
+
pinned = raw.get("pinned_files", [])
|
|
212
|
+
if (not isinstance(pinned, list) or len(pinned) > 32
|
|
213
|
+
or any(not isinstance(item, dict) or set(item) != {"path", "sha256"}
|
|
214
|
+
or not isinstance(item["path"], str) or not item["path"]
|
|
215
|
+
or not isinstance(item["sha256"], str)
|
|
216
|
+
or re.fullmatch(r"[0-9a-f]{64}", item["sha256"]) is None for item in pinned)
|
|
217
|
+
or len({item["path"] for item in pinned}) != len(pinned)):
|
|
218
|
+
raise ValueError(f"{name}: pinned_files requires distinct paths and lowercase SHA-256 digests")
|
|
219
|
+
if raw["kind"] != "command" and pinned:
|
|
220
|
+
raise ValueError(f"{name}: only command bindings can pin executable files")
|
|
221
|
+
bindings[name] = ToolBinding(
|
|
222
|
+
name, raw["kind"], raw["version"], tuple(argv), raw.get("path"), float(timeout), limit, env,
|
|
223
|
+
tuple((item["path"], item["sha256"]) for item in pinned), parallel_safe,
|
|
224
|
+
None if inherited is None else tuple(inherited)
|
|
225
|
+
)
|
|
226
|
+
return cls(bindings)
|
|
227
|
+
|
|
228
|
+
def binding_for(self, name: str, *, version: str) -> ToolBinding:
|
|
229
|
+
binding = self.bindings.get(name)
|
|
230
|
+
if binding is None:
|
|
231
|
+
raise ValueError(f"Tool {name!r} is not in the operator registry")
|
|
232
|
+
if binding.version != version:
|
|
233
|
+
raise ValueError("Declared tool version differs from the operator registry")
|
|
234
|
+
return binding
|
|
235
|
+
|
|
236
|
+
def acquire(self, binding: ToolBinding, request: dict, workspace: Path,
|
|
237
|
+
*, secret: bytes) -> AcquisitionResult:
|
|
238
|
+
"""Choose the trusted adapter selected by the operator's tool binding."""
|
|
239
|
+
adapter = _ADAPTERS.get(binding.kind)
|
|
240
|
+
if adapter is None:
|
|
241
|
+
raise ValueError(f"Unsupported operator tool kind {binding.kind!r}")
|
|
242
|
+
return adapter(binding, request, workspace, secret)
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _request_stop(process: subprocess.Popen) -> None:
|
|
246
|
+
try:
|
|
247
|
+
# The supervisor stops the collector group before releasing its lease.
|
|
248
|
+
process.terminate()
|
|
249
|
+
except ProcessLookupError:
|
|
250
|
+
pass
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _execute(binding: ToolBinding, request: dict, workspace: Path, secret: bytes) -> tuple[bytes, bytes, int, dict]:
|
|
254
|
+
"""Read bounded stdout/stderr without retaining unbounded subprocess output."""
|
|
255
|
+
encoded_request = json.dumps(request, sort_keys=True, allow_nan=False).encode("utf-8") + b"\n"
|
|
256
|
+
if len(encoded_request) > MAX_REQUEST_BYTES:
|
|
257
|
+
raise ValueError(f"Tool request exceeds {MAX_REQUEST_BYTES} bytes")
|
|
258
|
+
effective_env = binding.effective_environment()
|
|
259
|
+
# Arguments may themselves contain credentials. The keyed binding digest
|
|
260
|
+
# identifies the configured command without copying argv into a record.
|
|
261
|
+
metadata = {"process_environment_digest": binding.process_environment_digest(
|
|
262
|
+
secret, effective_env=effective_env)}
|
|
263
|
+
if os.name != "posix":
|
|
264
|
+
raise RuntimeError("The bounded command adapter currently requires a POSIX host")
|
|
265
|
+
with supervised_command(binding.argv, workspace=workspace, environ=effective_env,
|
|
266
|
+
timeout=binding.timeout_seconds) as command:
|
|
267
|
+
process = command.process
|
|
268
|
+
stdout, stderr = bytearray(), bytearray()
|
|
269
|
+
deadline = time.monotonic() + binding.timeout_seconds
|
|
270
|
+
failure = None
|
|
271
|
+
# Nonblocking stdin matters: an adapter may never consume a large request.
|
|
272
|
+
try:
|
|
273
|
+
with selectors.DefaultSelector() as selector:
|
|
274
|
+
for stream, label in ((process.stdout, "stdout"), (process.stderr, "stderr"), (process.stdin, "stdin")):
|
|
275
|
+
os.set_blocking(stream.fileno(), False)
|
|
276
|
+
selector.register(stream, selectors.EVENT_WRITE if label == "stdin" else selectors.EVENT_READ, label)
|
|
277
|
+
remaining_input = memoryview(encoded_request)
|
|
278
|
+
while selector.get_map():
|
|
279
|
+
if time.monotonic() >= deadline:
|
|
280
|
+
failure = "timeout"
|
|
281
|
+
break
|
|
282
|
+
events = selector.select(min(0.1, max(0, deadline - time.monotonic())))
|
|
283
|
+
for key, _ in events:
|
|
284
|
+
if key.data == "stdin":
|
|
285
|
+
try:
|
|
286
|
+
written = os.write(key.fd, remaining_input[:65536])
|
|
287
|
+
remaining_input = remaining_input[written:]
|
|
288
|
+
except BrokenPipeError:
|
|
289
|
+
remaining_input = remaining_input[:0]
|
|
290
|
+
if not remaining_input:
|
|
291
|
+
selector.unregister(key.fileobj)
|
|
292
|
+
key.fileobj.close()
|
|
293
|
+
continue
|
|
294
|
+
chunk = os.read(key.fd, 65536)
|
|
295
|
+
if not chunk:
|
|
296
|
+
selector.unregister(key.fileobj)
|
|
297
|
+
key.fileobj.close()
|
|
298
|
+
continue
|
|
299
|
+
target = stdout if key.data == "stdout" else stderr
|
|
300
|
+
remaining = binding.max_output_bytes - len(stdout) - len(stderr)
|
|
301
|
+
target.extend(chunk[:max(0, remaining)])
|
|
302
|
+
if len(chunk) > remaining:
|
|
303
|
+
failure = "output_limit"
|
|
304
|
+
break
|
|
305
|
+
if failure:
|
|
306
|
+
break
|
|
307
|
+
if failure:
|
|
308
|
+
_request_stop(process)
|
|
309
|
+
try:
|
|
310
|
+
returncode = process.wait(timeout=max(0.01, deadline - time.monotonic()))
|
|
311
|
+
except subprocess.TimeoutExpired:
|
|
312
|
+
failure = failure or "timeout"
|
|
313
|
+
_request_stop(process)
|
|
314
|
+
returncode = process.wait(timeout=5)
|
|
315
|
+
except BaseException:
|
|
316
|
+
_request_stop(process)
|
|
317
|
+
process.wait(timeout=5)
|
|
318
|
+
raise
|
|
319
|
+
finally:
|
|
320
|
+
for stream in (process.stdin, process.stdout, process.stderr):
|
|
321
|
+
if stream and not stream.closed:
|
|
322
|
+
stream.close()
|
|
323
|
+
status = command.collector_status()
|
|
324
|
+
returncode = status.returncode
|
|
325
|
+
if status.timed_out and failure is None:
|
|
326
|
+
failure = "timeout"
|
|
327
|
+
metadata.update({"returncode": returncode, "output_truncated": failure == "output_limit"})
|
|
328
|
+
if failure:
|
|
329
|
+
metadata["execution_error"] = failure
|
|
330
|
+
return bytes(stdout), bytes(stderr), returncode, metadata
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def _acquire_command(binding: ToolBinding, request: dict, workspace: Path, secret: bytes) -> AcquisitionResult:
|
|
334
|
+
stdout, stderr, returncode, metadata = _execute(binding, request, workspace, secret)
|
|
335
|
+
if metadata.get("execution_error"):
|
|
336
|
+
error = ValueError(metadata["execution_error"])
|
|
337
|
+
elif returncode != 0:
|
|
338
|
+
error = ValueError(f"Tool exited with status {returncode}")
|
|
339
|
+
else:
|
|
340
|
+
error = None
|
|
341
|
+
return AcquisitionResult(stdout, stderr, metadata, error)
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def _acquire_json_file(binding: ToolBinding, request: dict, workspace: Path, secret: bytes) -> AcquisitionResult:
|
|
345
|
+
path = bounded_path(workspace, binding.path)
|
|
346
|
+
metadata = {"file": str(path.relative_to(workspace))}
|
|
347
|
+
# A FIFO or device can block indefinitely before a bounded read begins.
|
|
348
|
+
# File imports only accept regular files.
|
|
349
|
+
try:
|
|
350
|
+
descriptor = os.open(path, os.O_RDONLY | os.O_NONBLOCK)
|
|
351
|
+
with os.fdopen(descriptor, "rb") as stream:
|
|
352
|
+
if not stat.S_ISREG(os.fstat(stream.fileno()).st_mode):
|
|
353
|
+
raise ValueError("File observations require a regular file")
|
|
354
|
+
stdout = stream.read(binding.max_output_bytes + 1)
|
|
355
|
+
except (OSError, ValueError) as exc:
|
|
356
|
+
return AcquisitionResult(metadata=metadata, error=exc)
|
|
357
|
+
if len(stdout) > binding.max_output_bytes:
|
|
358
|
+
metadata["output_truncated"] = True
|
|
359
|
+
return AcquisitionResult(stdout[:binding.max_output_bytes], metadata=metadata,
|
|
360
|
+
error=ValueError("output_limit"))
|
|
361
|
+
return AcquisitionResult(stdout, metadata=metadata)
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
_ADAPTERS = {"command": _acquire_command, "json_file": _acquire_json_file}
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def validate_envelope(stdout: bytes, *, file_import: bool, context: dict,
|
|
368
|
+
acquisition: dict) -> dict:
|
|
369
|
+
"""Check observation shape and the declared acquisition identity."""
|
|
370
|
+
from .evaluator import canonical_digest
|
|
371
|
+
|
|
372
|
+
envelope = strict_json(stdout.decode("utf-8"))
|
|
373
|
+
if not isinstance(envelope, dict) or "value" not in envelope or set(envelope) - {"value", "observed_at", "context", "request", "details"}:
|
|
374
|
+
raise ValueError("Tool output must be an object with value and optional observed_at, context, request, details")
|
|
375
|
+
if file_import and not {"observed_at", "context", "request"} <= set(envelope):
|
|
376
|
+
raise ValueError("File observations require observed_at and context and request; import must preserve age, scope and acquisition identity")
|
|
377
|
+
if "request" in envelope:
|
|
378
|
+
if not isinstance(envelope["request"], dict) or canonical_digest(envelope["request"]) != canonical_digest(acquisition):
|
|
379
|
+
raise ValueError("Observation request differs from the declared acquisition request")
|
|
380
|
+
if "context" in envelope:
|
|
381
|
+
if not isinstance(envelope["context"], dict) or canonical_digest(envelope["context"]) != canonical_digest(context):
|
|
382
|
+
raise ValueError("Observation context differs from the requested context")
|
|
383
|
+
return envelope
|