fable-engine 1.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fable_compressor.py +356 -0
- fable_engine/__init__.py +1 -0
- fable_engine/actions/__init__.py +291 -0
- fable_engine/actions/cas.py +182 -0
- fable_engine/actions/deliberation.py +523 -0
- fable_engine/actions/fleet.py +807 -0
- fable_engine/actions/lifecycle.py +298 -0
- fable_engine/actions/scrapers.py +116 -0
- fable_engine/actions/system3.py +815 -0
- fable_engine/browser.py +824 -0
- fable_engine/cas.py +974 -0
- fable_engine/fable_session.json +510 -0
- fable_engine/guards.py +283 -0
- fable_engine/schema.py +714 -0
- fable_engine/scrapers/__init__.py +32 -0
- fable_engine/scrapers/arxiv.py +115 -0
- fable_engine/scrapers/base.py +386 -0
- fable_engine/scrapers/github.py +129 -0
- fable_engine/scrapers/reddit.py +154 -0
- fable_engine/scrapers/web.py +120 -0
- fable_engine/scrapers/x.py +125 -0
- fable_engine/scrapers/youtube.py +132 -0
- fable_engine/server.py +414 -0
- fable_engine/session.py +1819 -0
- fable_engine/test_server.py +1362 -0
- fable_engine/updater.py +541 -0
- fable_engine-1.3.1.dist-info/LICENSE +22 -0
- fable_engine-1.3.1.dist-info/METADATA +173 -0
- fable_engine-1.3.1.dist-info/RECORD +104 -0
- fable_engine-1.3.1.dist-info/WHEEL +5 -0
- fable_engine-1.3.1.dist-info/entry_points.txt +5 -0
- fable_engine-1.3.1.dist-info/top_level.txt +6 -0
- fable_mode/__init__.py +3 -0
- fable_mode/__main__.py +4 -0
- fable_mode/adapters.py +1014 -0
- fable_mode/installer.py +553 -0
- fable_mode/launcher.py +437 -0
- fable_mode/manifest.py +142 -0
- fable_mode/resources.json +114 -0
- fable_mode/safety.py +103 -0
- fable_mode_entry.py +10 -0
- fable_v2/__init__.py +146 -0
- fable_v2/adapters.py +151 -0
- fable_v2/coder_fleet/__init__.py +100 -0
- fable_v2/coder_fleet/ast_tools.py +158 -0
- fable_v2/coder_fleet/compute.py +199 -0
- fable_v2/coder_fleet/design_engine.py +1316 -0
- fable_v2/coder_fleet/diagnostics.py +293 -0
- fable_v2/coder_fleet/fleet_dispatcher.py +214 -0
- fable_v2/coder_fleet/mock_auditor.py +306 -0
- fable_v2/coder_fleet/mutation.py +216 -0
- fable_v2/coder_fleet/property_oracle.py +260 -0
- fable_v2/coder_fleet/receipt_attestor.py +122 -0
- fable_v2/coder_fleet/red_team_swarm.py +908 -0
- fable_v2/coder_fleet/test_harness.py +198 -0
- fable_v2/coder_fleet/vector_engine.py +1287 -0
- fable_v2/coder_fleet/visual.py +357 -0
- fable_v2/coder_fleet/workspace.py +153 -0
- fable_v2/cortical/__init__.py +20 -0
- fable_v2/cortical/plasticity_engine.py +992 -0
- fable_v2/execution_broker.py +811 -0
- fable_v2/proof_engine.py +1141 -0
- fable_v2/protocol.py +485 -0
- fable_v2/runtime.py +1010 -0
- fable_v2/system3/__init__.py +204 -0
- fable_v2/system3/causal.py +558 -0
- fable_v2/system3/dialectical.py +577 -0
- fable_v2/system3/evolution.py +503 -0
- fable_v2/system3/executive.py +338 -0
- fable_v2/system3/free_energy.py +479 -0
- fable_v2/system3/hyperbolic.py +555 -0
- fable_v2/system3/induction.py +336 -0
- fable_v2/system3/kripke.py +548 -0
- fable_v2/system3/oracle.py +745 -0
- fable_v2/verifiers.py +72 -0
- tests/__init__.py +1 -0
- tests/test_anti_loop_circuit_breaker.py +64 -0
- tests/test_auto_updater.py +407 -0
- tests/test_coder_fleet.py +535 -0
- tests/test_delegation_compiler.py +54 -0
- tests/test_descriptor_boundaries.py +126 -0
- tests/test_design_engine.py +603 -0
- tests/test_epistemic_evidence_validator.py +66 -0
- tests/test_execution_broker.py +233 -0
- tests/test_fable_v2.py +406 -0
- tests/test_fleet_transitions.py +116 -0
- tests/test_fsm_redteam_evolution.py +406 -0
- tests/test_goal_rubric_and_pipeline.py +367 -0
- tests/test_hebbian_plasticity.py +585 -0
- tests/test_packaging_runtime.py +194 -0
- tests/test_proof_engine.py +259 -0
- tests/test_red_team_swarm.py +645 -0
- tests/test_redteam_remediation.py +169 -0
- tests/test_registration_transaction.py +375 -0
- tests/test_requested_regressions.py +467 -0
- tests/test_scrapers.py +370 -0
- tests/test_server_actions.py +93 -0
- tests/test_server_frontier_actions.py +269 -0
- tests/test_server_protocol.py +88 -0
- tests/test_stealth_browser.py +970 -0
- tests/test_system3.py +381 -0
- tests/test_system3_deep_integration.py +385 -0
- tests/test_system3_frontier.py +436 -0
- tests/test_vector_engine.py +608 -0
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
import hashlib
|
|
2
|
+
import io
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import subprocess
|
|
6
|
+
import sys
|
|
7
|
+
import tempfile
|
|
8
|
+
import time
|
|
9
|
+
import unittest
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from unittest.mock import patch
|
|
12
|
+
|
|
13
|
+
from fable_v2 import BrokerPolicy, ExecutionBroker
|
|
14
|
+
from fable_v2.execution_broker import MAX_ERROR_TEXT, MAX_FRAME_BYTES, serve
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class ExecutionBrokerTests(unittest.TestCase):
|
|
18
|
+
def setUp(self):
|
|
19
|
+
self.tempdir = tempfile.TemporaryDirectory()
|
|
20
|
+
self.workspace = Path(self.tempdir.name)
|
|
21
|
+
self.executable = Path(sys.executable).name
|
|
22
|
+
self.broker = ExecutionBroker(BrokerPolicy(
|
|
23
|
+
workspace=self.workspace,
|
|
24
|
+
allowed_executables=(self.executable,),
|
|
25
|
+
write_token_digest=hashlib.sha256(b"admin-token").hexdigest(),
|
|
26
|
+
))
|
|
27
|
+
|
|
28
|
+
def tearDown(self):
|
|
29
|
+
self.tempdir.cleanup()
|
|
30
|
+
|
|
31
|
+
def test_interpreters_are_blocked_before_write_authorization(self):
|
|
32
|
+
# shell=False does not stop Python from opening files directly.
|
|
33
|
+
with self.assertRaises(PermissionError):
|
|
34
|
+
self.broker.execute_command([
|
|
35
|
+
sys.executable, "-c", "open('unauthorized.txt', 'w').write('bypass')"
|
|
36
|
+
])
|
|
37
|
+
self.assertFalse((self.workspace / "unauthorized.txt").exists())
|
|
38
|
+
|
|
39
|
+
def test_command_is_allowlisted_and_runs_after_authorization(self):
|
|
40
|
+
with self.assertRaises(PermissionError):
|
|
41
|
+
self.broker.execute_command(["sh", "-c", "echo escaped"])
|
|
42
|
+
self.broker.unlock_writes("admin-token")
|
|
43
|
+
result = self.broker.execute_command([
|
|
44
|
+
sys.executable, "-c", "print('broker-ok')"
|
|
45
|
+
])
|
|
46
|
+
self.assertTrue(result["success"])
|
|
47
|
+
self.assertIn("broker-ok", result["stdout"])
|
|
48
|
+
|
|
49
|
+
def test_inspect_files_is_implemented_and_bounded(self):
|
|
50
|
+
target = self.workspace / "input.txt"
|
|
51
|
+
target.write_text("hello world", encoding="utf-8")
|
|
52
|
+
target.chmod(0o600)
|
|
53
|
+
result = self.broker.handle({"action": "inspect_files", "path": "input.txt"})
|
|
54
|
+
self.assertEqual(result["content"], "hello world")
|
|
55
|
+
self.assertFalse(result["truncated"])
|
|
56
|
+
self.assertEqual(result["content_hash"], hashlib.sha256(b"hello world").hexdigest())
|
|
57
|
+
self.assertEqual(self.broker.handle({"action": "probe_capabilities"}), self.broker.probe())
|
|
58
|
+
|
|
59
|
+
def test_subprocess_output_is_bounded_before_capture(self):
|
|
60
|
+
limited = ExecutionBroker(BrokerPolicy(
|
|
61
|
+
workspace=self.workspace,
|
|
62
|
+
allowed_executables=(self.executable,),
|
|
63
|
+
max_output_bytes=4096,
|
|
64
|
+
write_token_digest=hashlib.sha256(b"admin-token").hexdigest(),
|
|
65
|
+
))
|
|
66
|
+
limited.unlock_writes("admin-token")
|
|
67
|
+
result = limited.execute_command([
|
|
68
|
+
sys.executable, "-c", "import sys; sys.stdout.write('x' * 100000000)"
|
|
69
|
+
])
|
|
70
|
+
self.assertTrue(result["output_limited"])
|
|
71
|
+
self.assertFalse(result["success"])
|
|
72
|
+
self.assertLessEqual(len(result["stdout"].encode("utf-8")), 4096)
|
|
73
|
+
|
|
74
|
+
def test_same_basename_from_different_path_is_rejected(self):
|
|
75
|
+
fake = self.workspace / self.executable
|
|
76
|
+
fake.write_text("not the registered executable")
|
|
77
|
+
with self.assertRaises(PermissionError):
|
|
78
|
+
self.broker.execute_command([str(fake), "-c", "print('wrong')"])
|
|
79
|
+
|
|
80
|
+
def test_paths_cannot_escape_workspace(self):
|
|
81
|
+
self.broker.unlock_writes("admin-token")
|
|
82
|
+
with self.assertRaises(PermissionError):
|
|
83
|
+
self.broker.write_file("../outside.txt", "blocked")
|
|
84
|
+
|
|
85
|
+
def test_json_lines_frames_are_bounded_and_continue_after_oversize(self):
|
|
86
|
+
"""The model pipe can stay open; one bad frame cannot desynchronize it."""
|
|
87
|
+
output = io.StringIO()
|
|
88
|
+
import fable_v2.execution_broker as module
|
|
89
|
+
with patch.object(module.sys, "stdin", io.StringIO(
|
|
90
|
+
"x" * (MAX_FRAME_BYTES + 1) + "\n"
|
|
91
|
+
+ json.dumps({"action": "probe"}) + "\n")), \
|
|
92
|
+
patch.object(module.sys, "stdout", output):
|
|
93
|
+
serve(self.broker)
|
|
94
|
+
responses = [json.loads(line) for line in output.getvalue().splitlines()]
|
|
95
|
+
self.assertEqual(len(responses), 2)
|
|
96
|
+
self.assertFalse(responses[0]["ok"])
|
|
97
|
+
self.assertLessEqual(len(responses[0]["message"].encode()), MAX_ERROR_TEXT)
|
|
98
|
+
self.assertTrue(responses[1]["ok"])
|
|
99
|
+
self.assertIn("execute_command", responses[1]["result"]["capabilities"])
|
|
100
|
+
|
|
101
|
+
def test_malformed_json_is_a_bounded_controlled_error(self):
|
|
102
|
+
output = io.StringIO()
|
|
103
|
+
import fable_v2.execution_broker as module
|
|
104
|
+
with patch.object(module.sys, "stdin", io.StringIO("{" + "x" * 20000 + "\n")), \
|
|
105
|
+
patch.object(module.sys, "stdout", output):
|
|
106
|
+
serve(self.broker)
|
|
107
|
+
response = json.loads(output.getvalue())
|
|
108
|
+
self.assertFalse(response["ok"])
|
|
109
|
+
self.assertLessEqual(len(response["message"].encode()), MAX_ERROR_TEXT + 20)
|
|
110
|
+
|
|
111
|
+
def test_writes_are_locked_until_admin_authorization(self):
|
|
112
|
+
with self.assertRaises(PermissionError):
|
|
113
|
+
self.broker.write_file("result.txt", "blocked")
|
|
114
|
+
with self.assertRaises(PermissionError):
|
|
115
|
+
self.broker.unlock_writes("wrong-token")
|
|
116
|
+
self.broker.unlock_writes("admin-token")
|
|
117
|
+
result = self.broker.write_file("result.txt", "accepted")
|
|
118
|
+
self.assertTrue(result["writes_enabled"])
|
|
119
|
+
self.assertEqual((self.workspace / "result.txt").read_text(), "accepted")
|
|
120
|
+
|
|
121
|
+
@unittest.skipUnless(os.name != "nt", "inherited admin FD test is POSIX-only")
|
|
122
|
+
def test_broker_is_available_as_a_json_lines_child_process(self):
|
|
123
|
+
admin_read, admin_write = os.pipe()
|
|
124
|
+
env = os.environ.copy()
|
|
125
|
+
env["FABLE_BROKER_WRITE_TOKEN_DIGEST"] = hashlib.sha256(
|
|
126
|
+
b"admin-token"
|
|
127
|
+
).hexdigest()
|
|
128
|
+
process = subprocess.Popen(
|
|
129
|
+
[sys.executable, "-m", "fable_v2.execution_broker",
|
|
130
|
+
"--workspace", str(self.workspace),
|
|
131
|
+
"--allow-executable", self.executable,
|
|
132
|
+
"--admin-fd", str(admin_read)],
|
|
133
|
+
stdin=subprocess.PIPE, stdout=subprocess.PIPE,
|
|
134
|
+
stderr=subprocess.PIPE, text=True, env=env,
|
|
135
|
+
pass_fds=(admin_read,),
|
|
136
|
+
)
|
|
137
|
+
os.close(admin_read)
|
|
138
|
+
try:
|
|
139
|
+
os.write(admin_write, (json.dumps({
|
|
140
|
+
"action": "unlock_writes", "token": "admin-token",
|
|
141
|
+
}) + "\n").encode("utf-8"))
|
|
142
|
+
os.close(admin_write)
|
|
143
|
+
response = None
|
|
144
|
+
for _ in range(50):
|
|
145
|
+
process.stdin.write(json.dumps({"action": "probe"}) + "\n")
|
|
146
|
+
process.stdin.flush()
|
|
147
|
+
response = json.loads(process.stdout.readline())
|
|
148
|
+
if response["result"].get("writes_enabled"):
|
|
149
|
+
break
|
|
150
|
+
time.sleep(0.01)
|
|
151
|
+
self.assertTrue(response["ok"])
|
|
152
|
+
self.assertIn("execute_command", response["result"]["capabilities"])
|
|
153
|
+
self.assertTrue(response["result"]["writes_enabled"])
|
|
154
|
+
process.stdin.write(json.dumps({
|
|
155
|
+
"action": "write_file", "path": "cli.txt",
|
|
156
|
+
"content": "cli-authorized",
|
|
157
|
+
}) + "\n")
|
|
158
|
+
process.stdin.flush()
|
|
159
|
+
write_response = json.loads(process.stdout.readline())
|
|
160
|
+
self.assertTrue(write_response["ok"])
|
|
161
|
+
self.assertEqual((self.workspace / "cli.txt").read_text(), "cli-authorized")
|
|
162
|
+
finally:
|
|
163
|
+
process.terminate()
|
|
164
|
+
process.wait(timeout=5)
|
|
165
|
+
process.stdin.close()
|
|
166
|
+
process.stdout.close()
|
|
167
|
+
process.stderr.close()
|
|
168
|
+
|
|
169
|
+
def test_windows_backslash_path_resolution(self):
|
|
170
|
+
self.broker.unlock_writes("admin-token")
|
|
171
|
+
(self.workspace / "nested").mkdir(parents=True, exist_ok=True)
|
|
172
|
+
# Verify writing using Windows backslash
|
|
173
|
+
res = self.broker.write_file("nested\\win_test.txt", "backslash-content")
|
|
174
|
+
self.assertTrue(res["writes_enabled"])
|
|
175
|
+
self.assertEqual((self.workspace / "nested" / "win_test.txt").read_text(), "backslash-content")
|
|
176
|
+
|
|
177
|
+
# Verify inspection using Windows backslash
|
|
178
|
+
ins = self.broker.inspect_files("nested\\win_test.txt")
|
|
179
|
+
self.assertEqual(ins["content"], "backslash-content")
|
|
180
|
+
|
|
181
|
+
def test_broker_admin_file_unlocking(self):
|
|
182
|
+
"""Cross-platform administrative unlocking via --admin-file (Windows and POSIX compatible)."""
|
|
183
|
+
admin_cmd_file = self.workspace / ".admin_unlock.json"
|
|
184
|
+
env = os.environ.copy()
|
|
185
|
+
env["FABLE_BROKER_WRITE_TOKEN_DIGEST"] = hashlib.sha256(
|
|
186
|
+
b"admin-token"
|
|
187
|
+
).hexdigest()
|
|
188
|
+
process = subprocess.Popen(
|
|
189
|
+
[sys.executable, "-m", "fable_v2.execution_broker",
|
|
190
|
+
"--workspace", str(self.workspace),
|
|
191
|
+
"--allow-executable", self.executable,
|
|
192
|
+
"--admin-file", str(admin_cmd_file)],
|
|
193
|
+
stdin=subprocess.PIPE, stdout=subprocess.PIPE,
|
|
194
|
+
stderr=subprocess.PIPE, text=True, env=env,
|
|
195
|
+
)
|
|
196
|
+
try:
|
|
197
|
+
# Write token command to admin file
|
|
198
|
+
admin_cmd_file.write_text(json.dumps({
|
|
199
|
+
"action": "unlock_writes", "token": "admin-token",
|
|
200
|
+
}) + "\n", encoding="utf-8")
|
|
201
|
+
|
|
202
|
+
response = None
|
|
203
|
+
for _ in range(50):
|
|
204
|
+
process.stdin.write(json.dumps({"action": "probe"}) + "\n")
|
|
205
|
+
process.stdin.flush()
|
|
206
|
+
line = process.stdout.readline()
|
|
207
|
+
if line:
|
|
208
|
+
response = json.loads(line)
|
|
209
|
+
if response["result"].get("writes_enabled"):
|
|
210
|
+
break
|
|
211
|
+
time.sleep(0.05)
|
|
212
|
+
self.assertTrue(response["ok"])
|
|
213
|
+
self.assertTrue(response["result"]["writes_enabled"])
|
|
214
|
+
|
|
215
|
+
# Verify writes work now
|
|
216
|
+
process.stdin.write(json.dumps({
|
|
217
|
+
"action": "write_file", "path": "file_unlocked.txt",
|
|
218
|
+
"content": "authorized-via-file",
|
|
219
|
+
}) + "\n")
|
|
220
|
+
process.stdin.flush()
|
|
221
|
+
write_response = json.loads(process.stdout.readline())
|
|
222
|
+
self.assertTrue(write_response["ok"])
|
|
223
|
+
self.assertEqual((self.workspace / "file_unlocked.txt").read_text(), "authorized-via-file")
|
|
224
|
+
finally:
|
|
225
|
+
process.terminate()
|
|
226
|
+
process.wait(timeout=5)
|
|
227
|
+
process.stdin.close()
|
|
228
|
+
process.stdout.close()
|
|
229
|
+
process.stderr.close()
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
if __name__ == "__main__":
|
|
233
|
+
unittest.main()
|
tests/test_fable_v2.py
ADDED
|
@@ -0,0 +1,406 @@
|
|
|
1
|
+
import copy
|
|
2
|
+
import unittest
|
|
3
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
4
|
+
|
|
5
|
+
from fable_v2 import (
|
|
6
|
+
Candidate,
|
|
7
|
+
Evidence,
|
|
8
|
+
FableRun,
|
|
9
|
+
FunctionVerifier,
|
|
10
|
+
TaskSpec,
|
|
11
|
+
ToolReceipt,
|
|
12
|
+
VerificationPolicy,
|
|
13
|
+
VerificationResult,
|
|
14
|
+
get_profile,
|
|
15
|
+
new_run,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class FableV2RuntimeTests(unittest.TestCase):
|
|
20
|
+
def setUp(self):
|
|
21
|
+
self.task = TaskSpec(
|
|
22
|
+
task_id="coding-001",
|
|
23
|
+
objective="Repair the failing feature",
|
|
24
|
+
definition_of_done=("tests pass",),
|
|
25
|
+
required_capabilities=("inspect_files", "run_tests"),
|
|
26
|
+
required_evidence=("test-result",),
|
|
27
|
+
)
|
|
28
|
+
self.run = new_run("session-001", self.task)
|
|
29
|
+
self.inspect = ToolReceipt.from_result(
|
|
30
|
+
receipt_id="r-inspect", session_id="session-001",
|
|
31
|
+
capability="inspect_files", tool_name="grep",
|
|
32
|
+
tool_input={"query": "bug"}, tool_output={"matches": 2}, success=True,
|
|
33
|
+
)
|
|
34
|
+
self.tests = ToolReceipt.from_result(
|
|
35
|
+
receipt_id="r-tests", session_id="session-001",
|
|
36
|
+
capability="run_tests", tool_name="pytest",
|
|
37
|
+
tool_input={"target": "tests"}, tool_output={"passed": True}, success=True,
|
|
38
|
+
)
|
|
39
|
+
self.run.record_receipt(self.inspect)
|
|
40
|
+
self.run.record_receipt(self.tests)
|
|
41
|
+
self.evidence = Evidence.from_receipt(
|
|
42
|
+
self.tests,
|
|
43
|
+
evidence_id="e-tests",
|
|
44
|
+
claim="The regression tests pass",
|
|
45
|
+
kind="test-result",
|
|
46
|
+
source="pytest: tests",
|
|
47
|
+
)
|
|
48
|
+
self.run.attach_evidence(self.evidence)
|
|
49
|
+
self.candidate = Candidate(
|
|
50
|
+
candidate_id="candidate-001", session_id="session-001",
|
|
51
|
+
approach="minimal patch", artifact={"diff": "..."},
|
|
52
|
+
receipt_ids=("r-inspect", "r-tests"), evidence_ids=("e-tests",),
|
|
53
|
+
)
|
|
54
|
+
self.run.register_candidate(self.candidate)
|
|
55
|
+
|
|
56
|
+
def test_task_contract_requires_definition_of_done(self):
|
|
57
|
+
with self.assertRaises(ValueError):
|
|
58
|
+
TaskSpec(task_id="x", objective="y")
|
|
59
|
+
|
|
60
|
+
def test_receipts_are_hashed_and_recorded(self):
|
|
61
|
+
self.assertEqual(len(self.run.receipts), 2)
|
|
62
|
+
self.assertNotEqual(self.inspect.input_hash, self.inspect.output_hash)
|
|
63
|
+
self.assertTrue(self.evidence.integrity_bound)
|
|
64
|
+
self.assertEqual(self.run.successful_capabilities(), {"inspect_files", "run_tests"})
|
|
65
|
+
|
|
66
|
+
def test_direct_model_supplied_verification_is_rejected(self):
|
|
67
|
+
forged = VerificationResult("v", "session-001", "candidate-001", "model", True)
|
|
68
|
+
with self.assertRaises(PermissionError):
|
|
69
|
+
self.run.record_verification(forged)
|
|
70
|
+
with self.assertRaises(PermissionError):
|
|
71
|
+
self.run.finalize("candidate-001")
|
|
72
|
+
|
|
73
|
+
def test_required_capabilities_are_candidate_scoped(self):
|
|
74
|
+
task = TaskSpec(
|
|
75
|
+
task_id="scoped-capabilities", objective="x", definition_of_done=("done",),
|
|
76
|
+
required_capabilities=("inspect_files", "run_tests"),
|
|
77
|
+
verification_policy=VerificationPolicy(
|
|
78
|
+
required_verifier_classes=("deterministic",),
|
|
79
|
+
minimum_passing_verifiers=1,
|
|
80
|
+
require_independent=False,
|
|
81
|
+
),
|
|
82
|
+
)
|
|
83
|
+
run = new_run("session-scoped", task)
|
|
84
|
+
inspect = ToolReceipt.from_result(
|
|
85
|
+
receipt_id="r-scope-inspect", session_id="session-scoped",
|
|
86
|
+
capability="inspect_files", tool_name="grep", tool_input="x",
|
|
87
|
+
tool_output="inspection", success=True,
|
|
88
|
+
)
|
|
89
|
+
tests = ToolReceipt.from_result(
|
|
90
|
+
receipt_id="r-scope-tests", session_id="session-scoped",
|
|
91
|
+
capability="run_tests", tool_name="pytest", tool_input="x",
|
|
92
|
+
tool_output="tests", success=True,
|
|
93
|
+
)
|
|
94
|
+
run.record_receipt(inspect)
|
|
95
|
+
run.record_receipt(tests)
|
|
96
|
+
run.register_candidate(Candidate(
|
|
97
|
+
"inspect-only", "session-scoped", "inspection", "a", (inspect.receipt_id,)
|
|
98
|
+
))
|
|
99
|
+
run.register_candidate(Candidate(
|
|
100
|
+
"tests-only", "session-scoped", "tests", "b", (tests.receipt_id,)
|
|
101
|
+
))
|
|
102
|
+
self.assertEqual(run.successful_capabilities(), {"inspect_files", "run_tests"})
|
|
103
|
+
self.assertIn("run_tests", " ".join(run.missing_requirements("inspect-only")))
|
|
104
|
+
self.assertIn("inspect_files", " ".join(run.missing_requirements("tests-only")))
|
|
105
|
+
|
|
106
|
+
def test_finalization_requires_every_declared_capability(self):
|
|
107
|
+
task = TaskSpec(
|
|
108
|
+
task_id="missing", objective="x", definition_of_done=("done",),
|
|
109
|
+
required_capabilities=("inspect_files", "search_web"),
|
|
110
|
+
verification_policy=VerificationPolicy(
|
|
111
|
+
required_verifier_classes=("deterministic",),
|
|
112
|
+
minimum_passing_verifiers=1,
|
|
113
|
+
require_independent=False,
|
|
114
|
+
),
|
|
115
|
+
)
|
|
116
|
+
run = new_run("session-002", task)
|
|
117
|
+
run.record_receipt(ToolReceipt.from_result(
|
|
118
|
+
receipt_id="r", session_id="session-002", capability="inspect_files",
|
|
119
|
+
tool_name="grep", tool_input="x", tool_output="y", success=True,
|
|
120
|
+
))
|
|
121
|
+
run.attach_evidence(Evidence.from_receipt(
|
|
122
|
+
run.receipts["r"],
|
|
123
|
+
evidence_id="e", claim="inspection completed", kind="inspection",
|
|
124
|
+
source="grep",
|
|
125
|
+
))
|
|
126
|
+
candidate = Candidate("c", "session-002", "approach", "artifact", ("r",), ("e",))
|
|
127
|
+
run.register_candidate(candidate)
|
|
128
|
+
run.execute_verifier(FunctionVerifier(
|
|
129
|
+
"tests", lambda candidate: (True, ("checked",), 1.0), evidence_ids=("e",)
|
|
130
|
+
), "c")
|
|
131
|
+
with self.assertRaises(PermissionError) as error:
|
|
132
|
+
run.finalize("c")
|
|
133
|
+
self.assertIn("search_web", str(error.exception))
|
|
134
|
+
|
|
135
|
+
def test_deterministic_verifier_must_run_before_independent(self):
|
|
136
|
+
with self.assertRaises(PermissionError) as error:
|
|
137
|
+
self.run.execute_verifier(FunctionVerifier(
|
|
138
|
+
"independent-first", lambda candidate: (True, ("reviewed",), 1.0),
|
|
139
|
+
verifier_class="independent", independent=True, evidence_ids=("e-tests",),
|
|
140
|
+
), "candidate-001")
|
|
141
|
+
self.assertIn("after passing deterministic", str(error.exception))
|
|
142
|
+
|
|
143
|
+
def test_policy_requires_independent_verifier_class(self):
|
|
144
|
+
self.run.execute_verifier(FunctionVerifier(
|
|
145
|
+
"deterministic-tests", lambda candidate: (True, ("tests pass",), 1.0),
|
|
146
|
+
evidence_ids=("e-tests",),
|
|
147
|
+
), "candidate-001")
|
|
148
|
+
with self.assertRaises(PermissionError) as error:
|
|
149
|
+
self.run.finalize("candidate-001")
|
|
150
|
+
self.assertIn("independent", str(error.exception))
|
|
151
|
+
|
|
152
|
+
def test_verified_candidate_can_finalize(self):
|
|
153
|
+
self.run.execute_verifier(FunctionVerifier(
|
|
154
|
+
"deterministic-tests", lambda candidate: (True, ("tests pass",), 1.0),
|
|
155
|
+
evidence_ids=("e-tests",),
|
|
156
|
+
), "candidate-001")
|
|
157
|
+
self.run.execute_verifier(FunctionVerifier(
|
|
158
|
+
"independent-review", lambda candidate: (True, ("review passed",), 1.0),
|
|
159
|
+
verifier_class="independent", independent=True, evidence_ids=("e-tests",),
|
|
160
|
+
), "candidate-001")
|
|
161
|
+
result = self.run.finalize("candidate-001")
|
|
162
|
+
self.assertEqual(result.candidate_id, "candidate-001")
|
|
163
|
+
self.assertEqual(self.run.state.value, "finalized")
|
|
164
|
+
|
|
165
|
+
def test_failed_tool_cannot_anchor_evidence(self):
|
|
166
|
+
failed = ToolReceipt.from_result(
|
|
167
|
+
receipt_id="r-failed", session_id="session-001", capability="run_tests",
|
|
168
|
+
tool_name="pytest", tool_input="x", tool_output="failure", success=False,
|
|
169
|
+
)
|
|
170
|
+
self.run.record_receipt(failed)
|
|
171
|
+
with self.assertRaises(ValueError):
|
|
172
|
+
Evidence.from_receipt(
|
|
173
|
+
failed, evidence_id="e-failed", claim="it passed",
|
|
174
|
+
kind="test-result", source="pytest",
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
def test_evidence_hash_must_match_actual_content(self):
|
|
178
|
+
with self.assertRaises(ValueError):
|
|
179
|
+
Evidence(
|
|
180
|
+
"e-tampered", "session-001", "claim", "test-result", "pytest",
|
|
181
|
+
"r-tests", "not-the-output-hash", source_output_hash=self.tests.output_hash,
|
|
182
|
+
content=self.tests.output,
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
def test_in_process_verifier_cannot_claim_process_attestation(self):
|
|
186
|
+
verifier = FunctionVerifier(
|
|
187
|
+
"forged-boundary", lambda candidate: (True, ("claimed",), 1.0),
|
|
188
|
+
trust_boundary="process_attested",
|
|
189
|
+
)
|
|
190
|
+
with self.assertRaises(PermissionError):
|
|
191
|
+
self.run.execute_verifier(verifier, "candidate-001")
|
|
192
|
+
|
|
193
|
+
def test_process_attested_policy_rejects_in_process_results(self):
|
|
194
|
+
task = TaskSpec(
|
|
195
|
+
task_id="isolated", objective="x", definition_of_done=("done",),
|
|
196
|
+
required_capabilities=("inspect_files",),
|
|
197
|
+
verification_policy=VerificationPolicy(
|
|
198
|
+
required_verifier_classes=("deterministic",),
|
|
199
|
+
minimum_passing_verifiers=1,
|
|
200
|
+
require_independent=False,
|
|
201
|
+
minimum_trust_boundary="process_attested",
|
|
202
|
+
),
|
|
203
|
+
)
|
|
204
|
+
run = new_run("session-isolated", task)
|
|
205
|
+
receipt = ToolReceipt.from_result(
|
|
206
|
+
receipt_id="r-isolated", session_id="session-isolated",
|
|
207
|
+
capability="inspect_files", tool_name="grep", tool_input="x",
|
|
208
|
+
tool_output="y", success=True,
|
|
209
|
+
)
|
|
210
|
+
run.record_receipt(receipt)
|
|
211
|
+
evidence = Evidence.from_receipt(
|
|
212
|
+
receipt, evidence_id="e-isolated", claim="inspected", kind="inspection",
|
|
213
|
+
source="grep",
|
|
214
|
+
)
|
|
215
|
+
run.attach_evidence(evidence)
|
|
216
|
+
run.register_candidate(Candidate(
|
|
217
|
+
"c-isolated", "session-isolated", "approach", "artifact",
|
|
218
|
+
("r-isolated",), ("e-isolated",),
|
|
219
|
+
))
|
|
220
|
+
run.execute_verifier(FunctionVerifier(
|
|
221
|
+
"local", lambda candidate: (True, ("checked",), 1.0),
|
|
222
|
+
evidence_ids=("e-isolated",),
|
|
223
|
+
), "c-isolated")
|
|
224
|
+
with self.assertRaises(PermissionError):
|
|
225
|
+
run.finalize("c-isolated")
|
|
226
|
+
|
|
227
|
+
def test_verifier_function_is_composable(self):
|
|
228
|
+
verifier = FunctionVerifier("always-pass", lambda candidate: (True, ["ok"], 1.0))
|
|
229
|
+
result = verifier.verify(self.candidate)
|
|
230
|
+
self.assertTrue(result.passed)
|
|
231
|
+
self.assertEqual(result.verifier, "always-pass")
|
|
232
|
+
|
|
233
|
+
def test_verifier_cannot_return_a_result_for_another_candidate(self):
|
|
234
|
+
class WrongCandidateVerifier:
|
|
235
|
+
name = "wrong-candidate"
|
|
236
|
+
verifier_class = "deterministic"
|
|
237
|
+
independent = False
|
|
238
|
+
trust_boundary = "in_process"
|
|
239
|
+
|
|
240
|
+
def verify(self, candidate):
|
|
241
|
+
return VerificationResult(
|
|
242
|
+
"wrong", candidate.session_id, "another-candidate", self.name, True,
|
|
243
|
+
evidence_ids=("e-tests",),
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
with self.assertRaises(ValueError):
|
|
247
|
+
self.run.execute_verifier(WrongCandidateVerifier(), "candidate-001")
|
|
248
|
+
|
|
249
|
+
def test_verifier_evidence_must_belong_to_candidate(self):
|
|
250
|
+
candidate = Candidate(
|
|
251
|
+
"candidate-002", "session-001", "second approach", "other artifact",
|
|
252
|
+
("r-inspect",), (),
|
|
253
|
+
)
|
|
254
|
+
self.run.register_candidate(candidate)
|
|
255
|
+
with self.assertRaises(PermissionError):
|
|
256
|
+
self.run.execute_verifier(FunctionVerifier(
|
|
257
|
+
"wrong-evidence", lambda candidate: (True, ("checked",), 1.0),
|
|
258
|
+
evidence_ids=("e-tests",),
|
|
259
|
+
), "candidate-002")
|
|
260
|
+
|
|
261
|
+
def test_duplicate_or_contradictory_verification_is_rejected(self):
|
|
262
|
+
verifier = FunctionVerifier(
|
|
263
|
+
"single-use", lambda candidate: (True, ("checked",), 1.0),
|
|
264
|
+
evidence_ids=("e-tests",),
|
|
265
|
+
)
|
|
266
|
+
self.run.execute_verifier(verifier, "candidate-001")
|
|
267
|
+
with self.assertRaises(ValueError):
|
|
268
|
+
self.run.execute_verifier(verifier, "candidate-001")
|
|
269
|
+
with self.assertRaises(ValueError):
|
|
270
|
+
self.run.execute_verifier(FunctionVerifier(
|
|
271
|
+
"single-use", lambda candidate: (False, ("contradiction",), 0.0),
|
|
272
|
+
evidence_ids=("e-tests",),
|
|
273
|
+
), "candidate-001")
|
|
274
|
+
|
|
275
|
+
def test_finalization_rechecks_verifier_validity(self):
|
|
276
|
+
self.run.execute_verifier(FunctionVerifier(
|
|
277
|
+
"deterministic-tests", lambda candidate: (True, ("tests pass",), 1.0),
|
|
278
|
+
evidence_ids=("e-tests",),
|
|
279
|
+
), "candidate-001")
|
|
280
|
+
self.run.execute_verifier(FunctionVerifier(
|
|
281
|
+
"independent-review", lambda candidate: (True, ("review passed",), 1.0),
|
|
282
|
+
verifier_class="independent", independent=True, evidence_ids=("e-tests",),
|
|
283
|
+
), "candidate-001")
|
|
284
|
+
self.run.invalidate_verifier("independent-review", "calibration drift")
|
|
285
|
+
with self.assertRaises(PermissionError) as error:
|
|
286
|
+
self.run.finalize("candidate-001")
|
|
287
|
+
self.assertIn("independent", str(error.exception))
|
|
288
|
+
|
|
289
|
+
def test_malformed_or_reversed_timestamps_are_rejected(self):
|
|
290
|
+
with self.assertRaises(ValueError):
|
|
291
|
+
ToolReceipt.from_result(
|
|
292
|
+
receipt_id="bad-time", session_id="session-001", capability="x",
|
|
293
|
+
tool_name="x", tool_input="x", tool_output="x", success=True,
|
|
294
|
+
started_at="not-a-time", finished_at="not-a-time",
|
|
295
|
+
)
|
|
296
|
+
with self.assertRaises(ValueError):
|
|
297
|
+
ToolReceipt.from_result(
|
|
298
|
+
receipt_id="reversed", session_id="session-001", capability="x",
|
|
299
|
+
tool_name="x", tool_input="x", tool_output="x", success=True,
|
|
300
|
+
started_at="2026-08-28T12:00:00+00:00",
|
|
301
|
+
finished_at="2026-08-28T11:00:00+00:00",
|
|
302
|
+
)
|
|
303
|
+
|
|
304
|
+
def test_mutable_payloads_are_snapshotted(self):
|
|
305
|
+
output = {"passed": True}
|
|
306
|
+
metadata = {"host": {"name": "test"}}
|
|
307
|
+
receipt = ToolReceipt.from_result(
|
|
308
|
+
receipt_id="snapshot", session_id="session-001", capability="x",
|
|
309
|
+
tool_name="x", tool_input="x", tool_output=output, success=True,
|
|
310
|
+
metadata=metadata,
|
|
311
|
+
)
|
|
312
|
+
output["passed"] = False
|
|
313
|
+
metadata["host"]["name"] = "mutated"
|
|
314
|
+
self.assertTrue(receipt.output["passed"])
|
|
315
|
+
self.assertEqual(receipt.metadata["host"]["name"], "test")
|
|
316
|
+
|
|
317
|
+
artifact = {"items": [1]}
|
|
318
|
+
candidate = Candidate("snapshot-candidate", "session-001", "approach", artifact)
|
|
319
|
+
self.run.register_candidate(candidate)
|
|
320
|
+
artifact["items"].append(2)
|
|
321
|
+
self.assertEqual(self.run.candidates["snapshot-candidate"].artifact, {"items": [1]})
|
|
322
|
+
|
|
323
|
+
def test_run_serialization_round_trip(self):
|
|
324
|
+
restored = FableRun.from_dict(self.run.to_dict())
|
|
325
|
+
self.assertEqual(restored.status(), self.run.status())
|
|
326
|
+
restored.validate_event_history()
|
|
327
|
+
self.assertEqual(restored.evidence["e-tests"].content_hash, self.evidence.content_hash)
|
|
328
|
+
|
|
329
|
+
def test_parallel_candidate_registration_is_safe(self):
|
|
330
|
+
candidates = [
|
|
331
|
+
Candidate(f"parallel-{i}", "session-001", "parallel approach", {"i": i})
|
|
332
|
+
for i in range(32)
|
|
333
|
+
]
|
|
334
|
+
with ThreadPoolExecutor(max_workers=8) as pool:
|
|
335
|
+
list(pool.map(self.run.register_candidate, candidates))
|
|
336
|
+
self.assertEqual(len(self.run.candidates), 33)
|
|
337
|
+
self.run.validate_event_history()
|
|
338
|
+
|
|
339
|
+
def test_restored_verdict_is_checked_against_complete_attestation(self):
|
|
340
|
+
self.run.execute_verifier(FunctionVerifier(
|
|
341
|
+
"deterministic-tests", lambda candidate: (True, ("tests pass",), 1.0),
|
|
342
|
+
evidence_ids=("e-tests",),
|
|
343
|
+
), "candidate-001")
|
|
344
|
+
self.run.execute_verifier(FunctionVerifier(
|
|
345
|
+
"independent-review", lambda candidate: (True, ("review passed",), 1.0),
|
|
346
|
+
verifier_class="independent", independent=True, evidence_ids=("e-tests",),
|
|
347
|
+
), "candidate-001")
|
|
348
|
+
payload = self.run.to_dict()
|
|
349
|
+
self.assertEqual(FableRun.from_dict(payload).status()["verifications"], 2)
|
|
350
|
+
mutations = {
|
|
351
|
+
"passed": False,
|
|
352
|
+
"reasons": ["tampered"],
|
|
353
|
+
"evidence_ids": [],
|
|
354
|
+
"score": 0.0,
|
|
355
|
+
"independent": True,
|
|
356
|
+
"inspected_candidate": False,
|
|
357
|
+
}
|
|
358
|
+
for field, value in mutations.items():
|
|
359
|
+
tampered = copy.deepcopy(payload)
|
|
360
|
+
tampered["verifications"][0][field] = value
|
|
361
|
+
with self.assertRaises(PermissionError, msg=field):
|
|
362
|
+
FableRun.from_dict(tampered)
|
|
363
|
+
|
|
364
|
+
def test_dependency_graph_tampering_is_rejected_on_restore(self):
|
|
365
|
+
self.run.execute_verifier(FunctionVerifier(
|
|
366
|
+
"deterministic-tests", lambda candidate: (True, ("tests pass",), 1.0),
|
|
367
|
+
evidence_ids=("e-tests",),
|
|
368
|
+
), "candidate-001")
|
|
369
|
+
payload = self.run.to_dict()
|
|
370
|
+
self.assertTrue(payload["verifications"][0]["candidate_graph_hash"])
|
|
371
|
+
mutations = []
|
|
372
|
+
candidate_tamper = copy.deepcopy(payload)
|
|
373
|
+
candidate_tamper["candidates"][0]["receipt_ids"] = ["r-inspect"]
|
|
374
|
+
mutations.append(candidate_tamper)
|
|
375
|
+
receipt_tamper = copy.deepcopy(payload)
|
|
376
|
+
receipt_tamper["receipts"][0]["capability"] = "run_tests"
|
|
377
|
+
mutations.append(receipt_tamper)
|
|
378
|
+
evidence_tamper = copy.deepcopy(payload)
|
|
379
|
+
evidence_tamper["evidence"][0]["metadata"] = {"tampered": True}
|
|
380
|
+
mutations.append(evidence_tamper)
|
|
381
|
+
for tampered in mutations:
|
|
382
|
+
with self.assertRaises(PermissionError):
|
|
383
|
+
FableRun.from_dict(tampered)
|
|
384
|
+
|
|
385
|
+
def test_tampered_event_history_is_rejected_on_restore(self):
|
|
386
|
+
payload = self.run.to_dict()
|
|
387
|
+
payload["events"][0]["type"] = "tampered"
|
|
388
|
+
with self.assertRaises(ValueError):
|
|
389
|
+
FableRun.from_dict(payload)
|
|
390
|
+
|
|
391
|
+
def test_host_profiles_are_expected_until_probed(self):
|
|
392
|
+
profile = get_profile("antigravity")
|
|
393
|
+
self.assertFalse(profile.is_attested)
|
|
394
|
+
self.assertTrue(profile.supports("run_tests"))
|
|
395
|
+
self.assertFalse(profile.supports("run_tests", authoritative=True))
|
|
396
|
+
self.assertEqual(profile.normalize("run_command"), "execute_command")
|
|
397
|
+
self.assertFalse(profile.compatibility_report(["run_tests"])["compatible"])
|
|
398
|
+
attested = profile.attest(["run_tests"])
|
|
399
|
+
self.assertTrue(attested.is_attested)
|
|
400
|
+
self.assertTrue(attested.compatibility_report(["run_tests"])["compatible"])
|
|
401
|
+
unknown = get_profile("unknown-host")
|
|
402
|
+
self.assertFalse(unknown.compatibility_report(["run_tests"])["compatible"])
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
if __name__ == "__main__":
|
|
406
|
+
unittest.main()
|