engineering-argument-language 3.2.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. eal/__init__.py +3 -0
  2. eal/abstractions.py +144 -0
  3. eal/acquisition_coordination.py +80 -0
  4. eal/api_load_methods.py +77 -0
  5. eal/aspic.py +478 -0
  6. eal/aspic_compiler.py +445 -0
  7. eal/aspic_export.py +309 -0
  8. eal/builtin_methods.py +115 -0
  9. eal/catalogue.py +384 -0
  10. eal/cli.py +204 -0
  11. eal/client_transports.py +69 -0
  12. eal/collection_identity.py +28 -0
  13. eal/collection_scheduler.py +72 -0
  14. eal/command_process.py +118 -0
  15. eal/command_supervisor.py +112 -0
  16. eal/composition.py +531 -0
  17. eal/credentials.py +33 -0
  18. eal/dialectic.py +321 -0
  19. eal/discovery.py +127 -0
  20. eal/evaluator.py +619 -0
  21. eal/expressions.py +404 -0
  22. eal/extensions.py +38 -0
  23. eal/formatter.py +156 -0
  24. eal/generated/EALLexer.py +434 -0
  25. eal/generated/EALParser.py +6559 -0
  26. eal/generated/EALVisitor.py +393 -0
  27. eal/generated/__init__.py +1 -0
  28. eal/host.py +179 -0
  29. eal/host_redaction.py +51 -0
  30. eal/knowledge.py +96 -0
  31. eal/limits.py +88 -0
  32. eal/mcp_guard.py +56 -0
  33. eal/methods.py +463 -0
  34. eal/model.py +266 -0
  35. eal/model_context.py +55 -0
  36. eal/modes.py +121 -0
  37. eal/observation_reuse.py +128 -0
  38. eal/operation_contracts.py +25 -0
  39. eal/operations.py +155 -0
  40. eal/packets.py +604 -0
  41. eal/parser.py +407 -0
  42. eal/planning.py +136 -0
  43. eal/propositions.py +231 -0
  44. eal/reachability.py +85 -0
  45. eal/reasoning/__init__.py +25 -0
  46. eal/reasoning/abductive.py +45 -0
  47. eal/reasoning/analogical.py +41 -0
  48. eal/reasoning/causal.py +46 -0
  49. eal/reasoning/counterfactual.py +64 -0
  50. eal/reasoning/deductive.py +72 -0
  51. eal/reasoning/inductive.py +29 -0
  52. eal/reasoning/strategy.py +18 -0
  53. eal/reasoning/structured.py +11 -0
  54. eal/reasoning/temporal.py +47 -0
  55. eal/reasoning/validation.py +90 -0
  56. eal/registered_assessment.py +181 -0
  57. eal/runtime.py +454 -0
  58. eal/sampled_negative.py +131 -0
  59. eal/scope_transfer.py +59 -0
  60. eal/semantics.py +656 -0
  61. eal/server.py +79 -0
  62. eal/server_auth.py +37 -0
  63. eal/server_settings.py +258 -0
  64. eal/source_printer.py +129 -0
  65. eal/store.py +272 -0
  66. eal/tool_acquisition.py +383 -0
  67. engineering_argument_language-3.2.3.dist-info/METADATA +188 -0
  68. engineering_argument_language-3.2.3.dist-info/RECORD +72 -0
  69. engineering_argument_language-3.2.3.dist-info/WHEEL +5 -0
  70. engineering_argument_language-3.2.3.dist-info/entry_points.txt +4 -0
  71. engineering_argument_language-3.2.3.dist-info/licenses/LICENSE +24 -0
  72. engineering_argument_language-3.2.3.dist-info/top_level.txt +1 -0
eal/store.py ADDED
@@ -0,0 +1,272 @@
1
+ """SQLite storage for immutable collection, observation and reasoning records."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Iterator
6
+ from contextlib import contextmanager
7
+ import json
8
+ import os
9
+ import secrets
10
+ import sqlite3
11
+ import stat
12
+ from datetime import datetime, timezone
13
+ from pathlib import Path
14
+ from typing import Any
15
+ from uuid import uuid4
16
+
17
+ from filelock import FileLock
18
+
19
+
20
+ def _check_private_directory(directory: Path) -> None:
21
+ details = directory.lstat()
22
+ if not stat.S_ISDIR(details.st_mode):
23
+ raise PermissionError("The run-store directory must be a regular directory")
24
+ if os.name == "posix" and (details.st_uid != os.getuid() or details.st_mode & 0o022):
25
+ raise PermissionError("The run-store directory must be owned by the process and not writable by others")
26
+
27
+
28
+ @contextmanager
29
+ def _store_initialisation_lock(database_path: Path, *, timeout: float = 30) -> Iterator[None]:
30
+ """Coordinate store setup before SQLite or its private identity is opened.
31
+
32
+ Keep the lock file in place after release: unlinking it could let another
33
+ process acquire a different inode while a contender still holds this one.
34
+ The private parent excludes replacement by other operating-system users.
35
+ """
36
+ _check_private_directory(database_path.parent)
37
+ path = database_path.with_name(database_path.name + ".initialisation.lock")
38
+ flags = os.O_RDWR | getattr(os, "O_NOFOLLOW", 0)
39
+ try:
40
+ descriptor = os.open(path, flags | os.O_CREAT | os.O_EXCL, 0o600)
41
+ except FileExistsError:
42
+ # Check before opening an existing object: opening a FIFO or device
43
+ # can block or have effects before descriptor validation is possible.
44
+ details = path.lstat()
45
+ if not stat.S_ISREG(details.st_mode):
46
+ raise PermissionError("The run-store initialisation lock must be a private regular 0600 file")
47
+ descriptor = os.open(path, flags)
48
+ try:
49
+ details = os.fstat(descriptor)
50
+ named = path.lstat()
51
+ if (not stat.S_ISREG(details.st_mode) or not stat.S_ISREG(named.st_mode)
52
+ or (details.st_dev, details.st_ino) != (named.st_dev, named.st_ino)
53
+ or stat.S_IMODE(details.st_mode) != 0o600
54
+ or (os.name == "posix" and details.st_uid != os.getuid())):
55
+ raise PermissionError("The run-store initialisation lock must be a private regular 0600 file")
56
+ finally:
57
+ os.close(descriptor)
58
+ with FileLock(path, timeout=timeout, mode=0o600):
59
+ # FileLock owns a separate descriptor. Check the pathname once more
60
+ # before touching storage; participating processes retain the file.
61
+ details = path.lstat()
62
+ if (not stat.S_ISREG(details.st_mode) or stat.S_IMODE(details.st_mode) != 0o600
63
+ or (os.name == "posix" and details.st_uid != os.getuid())):
64
+ raise PermissionError("The run-store initialisation lock must be a private regular 0600 file")
65
+ yield
66
+
67
+
68
+ def utc_now() -> str:
69
+ return datetime.now(timezone.utc).isoformat().replace("+00:00", "Z")
70
+
71
+
72
+ def _private_binding_key(database_path: Path) -> bytes:
73
+ """Atomically establish a private store-local identity key outside SQLite."""
74
+ parent = database_path.parent
75
+ _check_private_directory(parent)
76
+ key_path = database_path.with_name(database_path.name + ".binding-key")
77
+ temporary = key_path.with_name(key_path.name + "." + secrets.token_hex(16) + ".tmp")
78
+ descriptor = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL |
79
+ getattr(os, "O_NOFOLLOW", 0), 0o600)
80
+ try:
81
+ candidate = secrets.token_bytes(32)
82
+ with os.fdopen(descriptor, "wb") as stream:
83
+ stream.write(candidate)
84
+ stream.flush()
85
+ os.fsync(stream.fileno())
86
+ try:
87
+ os.link(temporary, key_path, follow_symlinks=False)
88
+ except FileExistsError:
89
+ pass
90
+ finally:
91
+ temporary.unlink(missing_ok=True)
92
+ descriptor = os.open(key_path, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0))
93
+ with os.fdopen(descriptor, "rb") as stream:
94
+ info = os.fstat(stream.fileno())
95
+ if not stat.S_ISREG(info.st_mode) or info.st_mode & 0o077:
96
+ raise PermissionError("The run-store binding key must be a private regular file")
97
+ if os.name == "posix" and info.st_uid != os.getuid():
98
+ raise PermissionError("The run-store binding key must be owned by the process")
99
+ key = stream.read(33)
100
+ if len(key) != 32:
101
+ raise ValueError("The run-store binding key must contain exactly 32 bytes")
102
+ return key
103
+
104
+
105
+ def _check_private_sqlite_files(database_path: Path, *, create_database: bool = False) -> None:
106
+ """Refuse readable or replaceable SQLite files before opening the store."""
107
+ if create_database:
108
+ try:
109
+ descriptor = os.open(
110
+ database_path, os.O_RDWR | os.O_CREAT | os.O_EXCL |
111
+ getattr(os, "O_NOFOLLOW", 0), 0o600,
112
+ )
113
+ except FileExistsError:
114
+ pass
115
+ else:
116
+ os.close(descriptor)
117
+ for path in (database_path, database_path.with_name(database_path.name + "-wal"),
118
+ database_path.with_name(database_path.name + "-shm")):
119
+ try:
120
+ descriptor = os.open(path, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0))
121
+ except FileNotFoundError:
122
+ if path == database_path:
123
+ raise
124
+ continue
125
+ with os.fdopen(descriptor, "rb") as stream:
126
+ details = os.fstat(stream.fileno())
127
+ if (not stat.S_ISREG(details.st_mode) or stat.S_IMODE(details.st_mode) != 0o600
128
+ or (os.name == "posix" and details.st_uid != os.getuid())):
129
+ raise PermissionError(f"SQLite store file must be a private regular 0600 file: {path.name}")
130
+
131
+
132
+ class RunStore:
133
+ def __init__(self, path: str | Path):
134
+ self.path = Path(path)
135
+ self.path.parent.mkdir(parents=True, exist_ok=True)
136
+ with _store_initialisation_lock(self.path):
137
+ self._binding_key = _private_binding_key(self.path)
138
+ _check_private_sqlite_files(self.path, create_database=True)
139
+ self._initialise_schema()
140
+ _check_private_sqlite_files(self.path)
141
+
142
+ def _initialise_schema(self) -> None:
143
+ """Establish the current schema and backfill its index atomically."""
144
+ with self._connect() as connection:
145
+ connection.execute("PRAGMA journal_mode=WAL")
146
+ connection.execute("BEGIN IMMEDIATE")
147
+ connection.execute(
148
+ "CREATE TABLE IF NOT EXISTS records "
149
+ "(id TEXT PRIMARY KEY, kind TEXT NOT NULL, created_at TEXT NOT NULL, payload TEXT NOT NULL)"
150
+ )
151
+ connection.execute("CREATE INDEX IF NOT EXISTS records_kind ON records(kind, created_at)")
152
+ connection.execute(
153
+ "CREATE TABLE IF NOT EXISTS observation_index "
154
+ "(sequence INTEGER PRIMARY KEY, record_id TEXT NOT NULL UNIQUE, "
155
+ "evidence_id TEXT NOT NULL, environment TEXT NOT NULL, "
156
+ "request_digest TEXT NOT NULL, tool_binding_digest TEXT NOT NULL)"
157
+ )
158
+ connection.execute(
159
+ "CREATE INDEX IF NOT EXISTS observation_identity ON observation_index "
160
+ "(evidence_id, environment, request_digest, tool_binding_digest, sequence DESC)"
161
+ )
162
+ connection.execute(
163
+ "CREATE TABLE IF NOT EXISTS store_metadata "
164
+ "(key TEXT PRIMARY KEY, value TEXT NOT NULL)"
165
+ )
166
+ indexed = connection.execute(
167
+ "SELECT value FROM store_metadata WHERE key = 'observation-index-version'"
168
+ ).fetchone()
169
+ if indexed is None:
170
+ # Existing stores predate the index. Scan once, including failed
171
+ # attempts which deliberately get no reusable index entry.
172
+ for record_id, encoded in connection.execute(
173
+ "SELECT id, payload FROM records WHERE kind = 'observation' ORDER BY rowid"
174
+ ):
175
+ self._index_observation(connection, record_id, json.loads(encoded))
176
+ connection.execute(
177
+ "INSERT INTO store_metadata(key, value) VALUES ('observation-index-version', '1')"
178
+ )
179
+ elif indexed[0] != "1":
180
+ raise ValueError("Unsupported observation index version")
181
+
182
+ @contextmanager
183
+ def _connect(self) -> Iterator[sqlite3.Connection]:
184
+ """Commit or roll back one operation and close its owned connection."""
185
+ _check_private_sqlite_files(self.path)
186
+ connection = sqlite3.connect(self.path, timeout=30)
187
+ try:
188
+ _check_private_sqlite_files(self.path)
189
+ with connection:
190
+ yield connection
191
+ finally:
192
+ connection.close()
193
+
194
+ def put(self, kind: str, payload: dict[str, Any], *, record_id: str | None = None) -> str:
195
+ record_id = record_id or str(uuid4())
196
+ encoded = json.dumps(payload, ensure_ascii=False, sort_keys=True, allow_nan=False)
197
+ with self._connect() as connection:
198
+ connection.execute(
199
+ "INSERT INTO records(id, kind, created_at, payload) VALUES (?, ?, ?, ?)",
200
+ (record_id, kind, utc_now(), encoded),
201
+ )
202
+ if kind == "observation":
203
+ self._index_observation(connection, record_id, payload)
204
+ return record_id
205
+
206
+ @staticmethod
207
+ def _index_observation(connection: sqlite3.Connection, record_id: str, payload: dict[str, Any]) -> None:
208
+ if payload.get("status") != "ok":
209
+ return
210
+ keys = (payload.get("evidence_id"), payload.get("environment"),
211
+ payload.get("request_digest"), payload.get("tool_binding_digest"))
212
+ if any(not isinstance(key, str) or not key for key in keys):
213
+ return
214
+ connection.execute(
215
+ "INSERT INTO observation_index(record_id, evidence_id, environment, request_digest, "
216
+ "tool_binding_digest) VALUES (?, ?, ?, ?, ?)", (record_id, *keys)
217
+ )
218
+
219
+ def put_batch(self, entries: list[tuple[str, dict[str, Any], str | None]]) -> list[str]:
220
+ """Commit a collection and its observations as one transaction."""
221
+ if not entries:
222
+ raise ValueError("entries must not be empty")
223
+ ids = [record_id or str(uuid4()) for _, _, record_id in entries]
224
+ encoded = [json.dumps(payload, ensure_ascii=False, sort_keys=True, allow_nan=False)
225
+ for _, payload, _ in entries]
226
+ with self._connect() as connection:
227
+ for (kind, payload, _), record_id, value in zip(entries, ids, encoded):
228
+ connection.execute(
229
+ "INSERT INTO records(id, kind, created_at, payload) VALUES (?, ?, ?, ?)",
230
+ (record_id, kind, utc_now(), value),
231
+ )
232
+ if kind == "observation":
233
+ self._index_observation(connection, record_id, payload)
234
+ return ids
235
+
236
+ def find_observations(self, *, evidence_id: str, environment: str,
237
+ request_digest: str, tool_binding_digest: str,
238
+ limit: int = 100) -> list[dict[str, Any]]:
239
+ """Return exact-acquisition candidates, newest first, for reuse."""
240
+ if not 1 <= limit <= 1000:
241
+ raise ValueError("limit must be between 1 and 1000")
242
+ with self._connect() as connection:
243
+ rows = connection.execute(
244
+ "SELECT records.payload FROM observation_index "
245
+ "JOIN records ON records.id = observation_index.record_id "
246
+ "WHERE evidence_id = ? AND environment = ? AND request_digest = ? "
247
+ "AND tool_binding_digest = ? ORDER BY sequence DESC LIMIT ?",
248
+ (evidence_id, environment, request_digest, tool_binding_digest, limit),
249
+ ).fetchall()
250
+ return [json.loads(row[0]) for row in rows]
251
+
252
+ def get(self, record_id: str, *, kind: str | None = None) -> dict[str, Any]:
253
+ with self._connect() as connection:
254
+ row = connection.execute("SELECT kind, payload FROM records WHERE id = ?", (record_id,)).fetchone()
255
+ if row is None or (kind is not None and row[0] != kind):
256
+ raise KeyError(f"No {kind or 'stored'} record with ID {record_id!r}")
257
+ return json.loads(row[1])
258
+
259
+ def list(self, *, kind: str | None = None, limit: int = 100) -> list[dict[str, str]]:
260
+ if not 1 <= limit <= 1000:
261
+ raise ValueError("limit must be between 1 and 1000")
262
+ with self._connect() as connection:
263
+ if kind is None:
264
+ rows = connection.execute(
265
+ "SELECT id, kind, created_at FROM records ORDER BY rowid DESC LIMIT ?", (limit,)
266
+ ).fetchall()
267
+ else:
268
+ rows = connection.execute(
269
+ "SELECT id, kind, created_at FROM records WHERE kind = ? ORDER BY rowid DESC LIMIT ?",
270
+ (kind, limit),
271
+ ).fetchall()
272
+ return [dict(zip(("id", "kind", "created_at"), row)) for row in rows]
@@ -0,0 +1,383 @@
1
+ """Trusted tool bindings and bounded observation acquisition adapters.
2
+
3
+ The operator registry selects an adapter. EAL source supplies JSON input and
4
+ cannot choose executable paths or file locations. These adapters are not a
5
+ process sandbox.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import dataclasses
11
+ import hashlib
12
+ import hmac
13
+ import math
14
+ import json
15
+ import os
16
+ import re
17
+ import selectors
18
+ import stat
19
+ import subprocess
20
+ import time
21
+ import tomllib
22
+ from pathlib import Path
23
+ from typing import Any, Mapping
24
+
25
+ from .command_process import supervised_command
26
+
27
+ MAX_JSON_DEPTH = 128
28
+ MAX_REQUEST_BYTES = 1024 * 1024
29
+
30
+
31
+ def keyed_digest(secret: bytes, domain: bytes, value: Any) -> str:
32
+ """Public configuration identity without an offline low-entropy value oracle."""
33
+ from .evaluator import canonical_digest
34
+
35
+ if not isinstance(secret, bytes) or len(secret) != 32:
36
+ raise ValueError("A private 32-byte store key is required")
37
+ if not isinstance(domain, bytes) or not domain:
38
+ raise ValueError("A domain separator is required")
39
+ # Frozen acquisition hash domain, independent of the authored source version.
40
+ # Retain these bytes so EAL/3 can reuse compatible observation-record/1 data.
41
+ return hmac.new(secret, b"EAL/2\0" + domain + b"\0" +
42
+ bytes.fromhex(canonical_digest(value)), hashlib.sha256).hexdigest()
43
+
44
+
45
+ def strict_json(text: str) -> Any:
46
+ def pairs(items):
47
+ result = {}
48
+ for key, value in items:
49
+ if key in result:
50
+ raise ValueError(f"Duplicate JSON key {key!r}")
51
+ result[key] = value
52
+ return result
53
+
54
+ def invalid_constant(value):
55
+ raise ValueError(f"Non-finite JSON value {value}")
56
+
57
+ def finite_float(value):
58
+ parsed = float(value)
59
+ if not math.isfinite(parsed):
60
+ raise ValueError(f"Non-finite JSON value {value}")
61
+ return parsed
62
+
63
+ try:
64
+ value = json.loads(text, object_pairs_hook=pairs, parse_constant=invalid_constant, parse_float=finite_float)
65
+ pending = [(value, 0)]
66
+ while pending:
67
+ item, depth = pending.pop()
68
+ if isinstance(item, (dict, list)):
69
+ if depth >= MAX_JSON_DEPTH:
70
+ raise ValueError(f"JSON nesting exceeds {MAX_JSON_DEPTH} container levels")
71
+ children = item.values() if isinstance(item, dict) else item
72
+ pending.extend((child, depth + 1) for child in children)
73
+ # JSON escapes can encode lone UTF-16 surrogates which cannot be stored
74
+ # or hashed as UTF-8. Reject these before adding values to run records.
75
+ json.dumps(value, ensure_ascii=False, allow_nan=False).encode("utf-8")
76
+ except UnicodeError as exc:
77
+ raise ValueError("JSON strings must contain valid Unicode scalar values") from exc
78
+ except RecursionError as exc:
79
+ raise ValueError("JSON nesting exceeds decoder resources") from exc
80
+ return value
81
+
82
+
83
+ def bounded_path(workspace: Path, name: str) -> Path:
84
+ """Resolve symlinks and refuse traversal outside the configured workspace."""
85
+ path = (workspace / name).resolve()
86
+ if not path.is_relative_to(workspace.resolve()):
87
+ raise ValueError(f"Path is outside the workspace: {name}")
88
+ return path
89
+
90
+
91
+ @dataclasses.dataclass(frozen=True)
92
+ class ToolBinding:
93
+ name: str
94
+ kind: str
95
+ version: str
96
+ argv: tuple[str, ...] = ()
97
+ path: str | None = None
98
+ timeout_seconds: float = 30.0
99
+ max_output_bytes: int = 1024 * 1024
100
+ env: Mapping[str, str] = dataclasses.field(default_factory=dict, repr=False)
101
+ pinned_files: tuple[tuple[str, str], ...] = ()
102
+ # The operator asserts this collector is read-only and independent of
103
+ # other simultaneous acquisitions. Arbitrary commands default to serial.
104
+ parallel_safe: bool = False
105
+ # None preserves full host inheritance; an explicit list passes only the
106
+ # named variables that exist, plus the operator's configured env overlay.
107
+ inherit_env: tuple[str, ...] | None = None
108
+
109
+ def effective_environment(self) -> dict[str, str]:
110
+ if self.inherit_env is None:
111
+ current = dict(os.environ)
112
+ else:
113
+ current = {name: os.environ[name] for name in self.inherit_env if name in os.environ}
114
+ current.update(self.env)
115
+ return current
116
+
117
+ def process_environment_digest(self, secret: bytes, *,
118
+ effective_env: Mapping[str, str] | None = None) -> str | None:
119
+ """Identity of the effective command environment without exposing secrets."""
120
+ if self.kind != "command":
121
+ return None
122
+ current = self.effective_environment() if effective_env is None else dict(effective_env)
123
+ return keyed_digest(secret, b"process-environment", current)
124
+
125
+ def binding_digest(self, secret: bytes, *, workspace: Path | None = None) -> str:
126
+ """Keyed registry identity, checking each operator-pinned file's bytes."""
127
+ if self.pinned_files and workspace is None:
128
+ raise ValueError("Pinned collector files require a workspace")
129
+ for name, expected in self.pinned_files:
130
+ path = Path(name)
131
+ path = path.resolve() if path.is_absolute() else bounded_path(workspace, name)
132
+ try:
133
+ descriptor = os.open(path, os.O_RDONLY | os.O_NONBLOCK)
134
+ with os.fdopen(descriptor, "rb") as stream:
135
+ details = os.fstat(stream.fileno())
136
+ if not stat.S_ISREG(details.st_mode) or details.st_size > 16 * 1024 * 1024:
137
+ raise ValueError("Pinned collector file must be a regular file at most 16 MiB")
138
+ digest = hashlib.sha256(stream.read(16 * 1024 * 1024 + 1)).hexdigest()
139
+ except OSError as exc:
140
+ raise ValueError("Pinned collector file is inaccessible") from exc
141
+ if digest != expected:
142
+ raise ValueError("Pinned collector file identity differs from operator configuration")
143
+ identity = {
144
+ "name": self.name, "kind": self.kind,
145
+ "version": self.version, "argv": list(self.argv),
146
+ "path": self.path, "timeout_seconds": self.timeout_seconds,
147
+ "max_output_bytes": self.max_output_bytes,
148
+ "env": dict(self.env), "pinned_files": [{"path": path, "sha256": digest}
149
+ for path, digest in self.pinned_files]}
150
+ # Scheduling is a host choice, not a property of a measurement.
151
+ if self.inherit_env is not None:
152
+ identity["inherit_env"] = sorted(self.inherit_env)
153
+ return keyed_digest(secret, b"tool-binding", identity)
154
+
155
+
156
+ @dataclasses.dataclass(frozen=True)
157
+ class AcquisitionResult:
158
+ stdout: bytes = b""
159
+ stderr: bytes = b""
160
+ metadata: Mapping[str, Any] = dataclasses.field(default_factory=dict)
161
+ error: Exception | None = None
162
+
163
+
164
+ class ToolRegistry:
165
+ def __init__(self, bindings: Mapping[str, ToolBinding] | None = None):
166
+ self.bindings = dict(bindings or {})
167
+
168
+ @classmethod
169
+ def load(cls, path: str | Path) -> "ToolRegistry":
170
+ with Path(path).open("rb") as stream:
171
+ document = tomllib.load(stream)
172
+ if set(document) - {"tools"} or not isinstance(document.get("tools", {}), dict):
173
+ raise ValueError("The registry must contain only a [tools] table")
174
+ bindings = {}
175
+ allowed = {"kind", "version", "argv", "path", "timeout_seconds", "max_output_bytes", "env", "pinned_files", "parallel_safe", "inherit_env"}
176
+ for name, raw in document.get("tools", {}).items():
177
+ if not isinstance(raw, dict) or set(raw) - allowed:
178
+ raise ValueError(f"Unknown registry settings for {name}")
179
+ if raw.get("kind") not in _ADAPTERS:
180
+ raise ValueError(f"{name}: kind must be command or json_file")
181
+ if not isinstance(raw.get("version"), str) or not raw["version"]:
182
+ raise ValueError(f"{name}: a nonempty version is required")
183
+ argv = raw.get("argv", [])
184
+ if not isinstance(argv, list) or not all(isinstance(arg, str) and arg for arg in argv):
185
+ raise ValueError(f"{name}: argv must contain nonempty strings")
186
+ if raw["kind"] == "command" and (not argv or "path" in raw):
187
+ raise ValueError(f"{name}: command requires argv and does not use path")
188
+ if raw["kind"] == "json_file" and (not isinstance(raw.get("path"), str) or not raw["path"] or argv):
189
+ raise ValueError(f"{name}: json_file requires a path and does not use argv")
190
+ timeout = raw.get("timeout_seconds", 30.0)
191
+ if isinstance(timeout, bool) or not isinstance(timeout, (int, float)) or not math.isfinite(timeout) or not 0 < timeout <= 3600:
192
+ raise ValueError(f"{name}: timeout_seconds must be finite, positive and at most 3600")
193
+ limit = raw.get("max_output_bytes", 1024 * 1024)
194
+ if isinstance(limit, bool) or not isinstance(limit, int) or not 1 <= limit <= 16 * 1024 * 1024:
195
+ raise ValueError(f"{name}: max_output_bytes must be an integer from 1 through 16777216")
196
+ env = raw.get("env", {})
197
+ if not isinstance(env, dict) or not all(isinstance(k, str) and isinstance(v, str) for k, v in env.items()):
198
+ raise ValueError(f"{name}: env must map strings to strings")
199
+ parallel_safe = raw.get("parallel_safe", False)
200
+ if type(parallel_safe) is not bool:
201
+ raise ValueError(f"{name}: parallel_safe must be a boolean")
202
+ inherited = raw.get("inherit_env")
203
+ if inherited is not None:
204
+ if (raw["kind"] != "command" or not isinstance(inherited, list)
205
+ or len(inherited) > 64 or any(
206
+ not isinstance(item, str)
207
+ or re.fullmatch(r"[A-Za-z_][A-Za-z_0-9]*", item, re.ASCII) is None
208
+ for item in inherited)
209
+ or len(set(inherited)) != len(inherited)):
210
+ raise ValueError(f"{name}: inherit_env requires up to 64 distinct command environment variable names")
211
+ pinned = raw.get("pinned_files", [])
212
+ if (not isinstance(pinned, list) or len(pinned) > 32
213
+ or any(not isinstance(item, dict) or set(item) != {"path", "sha256"}
214
+ or not isinstance(item["path"], str) or not item["path"]
215
+ or not isinstance(item["sha256"], str)
216
+ or re.fullmatch(r"[0-9a-f]{64}", item["sha256"]) is None for item in pinned)
217
+ or len({item["path"] for item in pinned}) != len(pinned)):
218
+ raise ValueError(f"{name}: pinned_files requires distinct paths and lowercase SHA-256 digests")
219
+ if raw["kind"] != "command" and pinned:
220
+ raise ValueError(f"{name}: only command bindings can pin executable files")
221
+ bindings[name] = ToolBinding(
222
+ name, raw["kind"], raw["version"], tuple(argv), raw.get("path"), float(timeout), limit, env,
223
+ tuple((item["path"], item["sha256"]) for item in pinned), parallel_safe,
224
+ None if inherited is None else tuple(inherited)
225
+ )
226
+ return cls(bindings)
227
+
228
+ def binding_for(self, name: str, *, version: str) -> ToolBinding:
229
+ binding = self.bindings.get(name)
230
+ if binding is None:
231
+ raise ValueError(f"Tool {name!r} is not in the operator registry")
232
+ if binding.version != version:
233
+ raise ValueError("Declared tool version differs from the operator registry")
234
+ return binding
235
+
236
+ def acquire(self, binding: ToolBinding, request: dict, workspace: Path,
237
+ *, secret: bytes) -> AcquisitionResult:
238
+ """Choose the trusted adapter selected by the operator's tool binding."""
239
+ adapter = _ADAPTERS.get(binding.kind)
240
+ if adapter is None:
241
+ raise ValueError(f"Unsupported operator tool kind {binding.kind!r}")
242
+ return adapter(binding, request, workspace, secret)
243
+
244
+
245
+ def _request_stop(process: subprocess.Popen) -> None:
246
+ try:
247
+ # The supervisor stops the collector group before releasing its lease.
248
+ process.terminate()
249
+ except ProcessLookupError:
250
+ pass
251
+
252
+
253
+ def _execute(binding: ToolBinding, request: dict, workspace: Path, secret: bytes) -> tuple[bytes, bytes, int, dict]:
254
+ """Read bounded stdout/stderr without retaining unbounded subprocess output."""
255
+ encoded_request = json.dumps(request, sort_keys=True, allow_nan=False).encode("utf-8") + b"\n"
256
+ if len(encoded_request) > MAX_REQUEST_BYTES:
257
+ raise ValueError(f"Tool request exceeds {MAX_REQUEST_BYTES} bytes")
258
+ effective_env = binding.effective_environment()
259
+ # Arguments may themselves contain credentials. The keyed binding digest
260
+ # identifies the configured command without copying argv into a record.
261
+ metadata = {"process_environment_digest": binding.process_environment_digest(
262
+ secret, effective_env=effective_env)}
263
+ if os.name != "posix":
264
+ raise RuntimeError("The bounded command adapter currently requires a POSIX host")
265
+ with supervised_command(binding.argv, workspace=workspace, environ=effective_env,
266
+ timeout=binding.timeout_seconds) as command:
267
+ process = command.process
268
+ stdout, stderr = bytearray(), bytearray()
269
+ deadline = time.monotonic() + binding.timeout_seconds
270
+ failure = None
271
+ # Nonblocking stdin matters: an adapter may never consume a large request.
272
+ try:
273
+ with selectors.DefaultSelector() as selector:
274
+ for stream, label in ((process.stdout, "stdout"), (process.stderr, "stderr"), (process.stdin, "stdin")):
275
+ os.set_blocking(stream.fileno(), False)
276
+ selector.register(stream, selectors.EVENT_WRITE if label == "stdin" else selectors.EVENT_READ, label)
277
+ remaining_input = memoryview(encoded_request)
278
+ while selector.get_map():
279
+ if time.monotonic() >= deadline:
280
+ failure = "timeout"
281
+ break
282
+ events = selector.select(min(0.1, max(0, deadline - time.monotonic())))
283
+ for key, _ in events:
284
+ if key.data == "stdin":
285
+ try:
286
+ written = os.write(key.fd, remaining_input[:65536])
287
+ remaining_input = remaining_input[written:]
288
+ except BrokenPipeError:
289
+ remaining_input = remaining_input[:0]
290
+ if not remaining_input:
291
+ selector.unregister(key.fileobj)
292
+ key.fileobj.close()
293
+ continue
294
+ chunk = os.read(key.fd, 65536)
295
+ if not chunk:
296
+ selector.unregister(key.fileobj)
297
+ key.fileobj.close()
298
+ continue
299
+ target = stdout if key.data == "stdout" else stderr
300
+ remaining = binding.max_output_bytes - len(stdout) - len(stderr)
301
+ target.extend(chunk[:max(0, remaining)])
302
+ if len(chunk) > remaining:
303
+ failure = "output_limit"
304
+ break
305
+ if failure:
306
+ break
307
+ if failure:
308
+ _request_stop(process)
309
+ try:
310
+ returncode = process.wait(timeout=max(0.01, deadline - time.monotonic()))
311
+ except subprocess.TimeoutExpired:
312
+ failure = failure or "timeout"
313
+ _request_stop(process)
314
+ returncode = process.wait(timeout=5)
315
+ except BaseException:
316
+ _request_stop(process)
317
+ process.wait(timeout=5)
318
+ raise
319
+ finally:
320
+ for stream in (process.stdin, process.stdout, process.stderr):
321
+ if stream and not stream.closed:
322
+ stream.close()
323
+ status = command.collector_status()
324
+ returncode = status.returncode
325
+ if status.timed_out and failure is None:
326
+ failure = "timeout"
327
+ metadata.update({"returncode": returncode, "output_truncated": failure == "output_limit"})
328
+ if failure:
329
+ metadata["execution_error"] = failure
330
+ return bytes(stdout), bytes(stderr), returncode, metadata
331
+
332
+
333
+ def _acquire_command(binding: ToolBinding, request: dict, workspace: Path, secret: bytes) -> AcquisitionResult:
334
+ stdout, stderr, returncode, metadata = _execute(binding, request, workspace, secret)
335
+ if metadata.get("execution_error"):
336
+ error = ValueError(metadata["execution_error"])
337
+ elif returncode != 0:
338
+ error = ValueError(f"Tool exited with status {returncode}")
339
+ else:
340
+ error = None
341
+ return AcquisitionResult(stdout, stderr, metadata, error)
342
+
343
+
344
+ def _acquire_json_file(binding: ToolBinding, request: dict, workspace: Path, secret: bytes) -> AcquisitionResult:
345
+ path = bounded_path(workspace, binding.path)
346
+ metadata = {"file": str(path.relative_to(workspace))}
347
+ # A FIFO or device can block indefinitely before a bounded read begins.
348
+ # File imports only accept regular files.
349
+ try:
350
+ descriptor = os.open(path, os.O_RDONLY | os.O_NONBLOCK)
351
+ with os.fdopen(descriptor, "rb") as stream:
352
+ if not stat.S_ISREG(os.fstat(stream.fileno()).st_mode):
353
+ raise ValueError("File observations require a regular file")
354
+ stdout = stream.read(binding.max_output_bytes + 1)
355
+ except (OSError, ValueError) as exc:
356
+ return AcquisitionResult(metadata=metadata, error=exc)
357
+ if len(stdout) > binding.max_output_bytes:
358
+ metadata["output_truncated"] = True
359
+ return AcquisitionResult(stdout[:binding.max_output_bytes], metadata=metadata,
360
+ error=ValueError("output_limit"))
361
+ return AcquisitionResult(stdout, metadata=metadata)
362
+
363
+
364
+ _ADAPTERS = {"command": _acquire_command, "json_file": _acquire_json_file}
365
+
366
+
367
+ def validate_envelope(stdout: bytes, *, file_import: bool, context: dict,
368
+ acquisition: dict) -> dict:
369
+ """Check observation shape and the declared acquisition identity."""
370
+ from .evaluator import canonical_digest
371
+
372
+ envelope = strict_json(stdout.decode("utf-8"))
373
+ if not isinstance(envelope, dict) or "value" not in envelope or set(envelope) - {"value", "observed_at", "context", "request", "details"}:
374
+ raise ValueError("Tool output must be an object with value and optional observed_at, context, request, details")
375
+ if file_import and not {"observed_at", "context", "request"} <= set(envelope):
376
+ raise ValueError("File observations require observed_at and context and request; import must preserve age, scope and acquisition identity")
377
+ if "request" in envelope:
378
+ if not isinstance(envelope["request"], dict) or canonical_digest(envelope["request"]) != canonical_digest(acquisition):
379
+ raise ValueError("Observation request differs from the declared acquisition request")
380
+ if "context" in envelope:
381
+ if not isinstance(envelope["context"], dict) or canonical_digest(envelope["context"]) != canonical_digest(context):
382
+ raise ValueError("Observation context differs from the requested context")
383
+ return envelope