fable-engine 1.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fable_compressor.py +356 -0
- fable_engine/__init__.py +1 -0
- fable_engine/actions/__init__.py +291 -0
- fable_engine/actions/cas.py +182 -0
- fable_engine/actions/deliberation.py +523 -0
- fable_engine/actions/fleet.py +807 -0
- fable_engine/actions/lifecycle.py +298 -0
- fable_engine/actions/scrapers.py +116 -0
- fable_engine/actions/system3.py +815 -0
- fable_engine/browser.py +824 -0
- fable_engine/cas.py +974 -0
- fable_engine/fable_session.json +510 -0
- fable_engine/guards.py +283 -0
- fable_engine/schema.py +714 -0
- fable_engine/scrapers/__init__.py +32 -0
- fable_engine/scrapers/arxiv.py +115 -0
- fable_engine/scrapers/base.py +386 -0
- fable_engine/scrapers/github.py +129 -0
- fable_engine/scrapers/reddit.py +154 -0
- fable_engine/scrapers/web.py +120 -0
- fable_engine/scrapers/x.py +125 -0
- fable_engine/scrapers/youtube.py +132 -0
- fable_engine/server.py +414 -0
- fable_engine/session.py +1819 -0
- fable_engine/test_server.py +1362 -0
- fable_engine/updater.py +541 -0
- fable_engine-1.3.1.dist-info/LICENSE +22 -0
- fable_engine-1.3.1.dist-info/METADATA +173 -0
- fable_engine-1.3.1.dist-info/RECORD +104 -0
- fable_engine-1.3.1.dist-info/WHEEL +5 -0
- fable_engine-1.3.1.dist-info/entry_points.txt +5 -0
- fable_engine-1.3.1.dist-info/top_level.txt +6 -0
- fable_mode/__init__.py +3 -0
- fable_mode/__main__.py +4 -0
- fable_mode/adapters.py +1014 -0
- fable_mode/installer.py +553 -0
- fable_mode/launcher.py +437 -0
- fable_mode/manifest.py +142 -0
- fable_mode/resources.json +114 -0
- fable_mode/safety.py +103 -0
- fable_mode_entry.py +10 -0
- fable_v2/__init__.py +146 -0
- fable_v2/adapters.py +151 -0
- fable_v2/coder_fleet/__init__.py +100 -0
- fable_v2/coder_fleet/ast_tools.py +158 -0
- fable_v2/coder_fleet/compute.py +199 -0
- fable_v2/coder_fleet/design_engine.py +1316 -0
- fable_v2/coder_fleet/diagnostics.py +293 -0
- fable_v2/coder_fleet/fleet_dispatcher.py +214 -0
- fable_v2/coder_fleet/mock_auditor.py +306 -0
- fable_v2/coder_fleet/mutation.py +216 -0
- fable_v2/coder_fleet/property_oracle.py +260 -0
- fable_v2/coder_fleet/receipt_attestor.py +122 -0
- fable_v2/coder_fleet/red_team_swarm.py +908 -0
- fable_v2/coder_fleet/test_harness.py +198 -0
- fable_v2/coder_fleet/vector_engine.py +1287 -0
- fable_v2/coder_fleet/visual.py +357 -0
- fable_v2/coder_fleet/workspace.py +153 -0
- fable_v2/cortical/__init__.py +20 -0
- fable_v2/cortical/plasticity_engine.py +992 -0
- fable_v2/execution_broker.py +811 -0
- fable_v2/proof_engine.py +1141 -0
- fable_v2/protocol.py +485 -0
- fable_v2/runtime.py +1010 -0
- fable_v2/system3/__init__.py +204 -0
- fable_v2/system3/causal.py +558 -0
- fable_v2/system3/dialectical.py +577 -0
- fable_v2/system3/evolution.py +503 -0
- fable_v2/system3/executive.py +338 -0
- fable_v2/system3/free_energy.py +479 -0
- fable_v2/system3/hyperbolic.py +555 -0
- fable_v2/system3/induction.py +336 -0
- fable_v2/system3/kripke.py +548 -0
- fable_v2/system3/oracle.py +745 -0
- fable_v2/verifiers.py +72 -0
- tests/__init__.py +1 -0
- tests/test_anti_loop_circuit_breaker.py +64 -0
- tests/test_auto_updater.py +407 -0
- tests/test_coder_fleet.py +535 -0
- tests/test_delegation_compiler.py +54 -0
- tests/test_descriptor_boundaries.py +126 -0
- tests/test_design_engine.py +603 -0
- tests/test_epistemic_evidence_validator.py +66 -0
- tests/test_execution_broker.py +233 -0
- tests/test_fable_v2.py +406 -0
- tests/test_fleet_transitions.py +116 -0
- tests/test_fsm_redteam_evolution.py +406 -0
- tests/test_goal_rubric_and_pipeline.py +367 -0
- tests/test_hebbian_plasticity.py +585 -0
- tests/test_packaging_runtime.py +194 -0
- tests/test_proof_engine.py +259 -0
- tests/test_red_team_swarm.py +645 -0
- tests/test_redteam_remediation.py +169 -0
- tests/test_registration_transaction.py +375 -0
- tests/test_requested_regressions.py +467 -0
- tests/test_scrapers.py +370 -0
- tests/test_server_actions.py +93 -0
- tests/test_server_frontier_actions.py +269 -0
- tests/test_server_protocol.py +88 -0
- tests/test_stealth_browser.py +970 -0
- tests/test_system3.py +381 -0
- tests/test_system3_deep_integration.py +385 -0
- tests/test_system3_frontier.py +436 -0
- tests/test_vector_engine.py +608 -0
|
@@ -0,0 +1,645 @@
|
|
|
1
|
+
"""Comprehensive Unit Test Suite for Modular Fable Part 1: Adversarial Code Review Swarm.
|
|
2
|
+
|
|
3
|
+
Tests:
|
|
4
|
+
- Scenario generation across all 5 attack vectors + custom hypotheses
|
|
5
|
+
- Break detection on fragile/broken callables
|
|
6
|
+
- Resilience verification on hardened callables
|
|
7
|
+
- to_dict() and to_markdown() serialization & GitHub alert formatting
|
|
8
|
+
- Full closed-loop ping-pong remediation cycle
|
|
9
|
+
- CoderFleetDispatcher routing for all 5 red team actions
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import copy
|
|
14
|
+
import os
|
|
15
|
+
import sys
|
|
16
|
+
import tempfile
|
|
17
|
+
import threading
|
|
18
|
+
import unittest
|
|
19
|
+
import uuid
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
# Ensure workspace root is in sys.path
|
|
24
|
+
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
25
|
+
|
|
26
|
+
from fable_v2.coder_fleet import (
|
|
27
|
+
AttackVector,
|
|
28
|
+
BreakFinding,
|
|
29
|
+
BreakScenario,
|
|
30
|
+
CoderFleetDispatcher,
|
|
31
|
+
RedTeamBreakageReport,
|
|
32
|
+
RedTeamSwarm,
|
|
33
|
+
)
|
|
34
|
+
from fable_v2.cortical import HebbianPlasticityEngine
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class TestRedTeamSwarmScenarios(unittest.TestCase):
|
|
38
|
+
def setUp(self) -> None:
|
|
39
|
+
self.swarm = RedTeamSwarm()
|
|
40
|
+
|
|
41
|
+
def test_scenario_generation_all_vectors(self) -> None:
|
|
42
|
+
scenarios = self.swarm.generate_break_scenarios(
|
|
43
|
+
target_name="test_target",
|
|
44
|
+
custom_hypotheses=[
|
|
45
|
+
"What will happen if a concurrent race condition occurs?",
|
|
46
|
+
"What will happen if byzantine payload injections arrive?",
|
|
47
|
+
],
|
|
48
|
+
)
|
|
49
|
+
self.assertGreaterEqual(len(scenarios), 10)
|
|
50
|
+
|
|
51
|
+
vectors_present = {s.vector for s in scenarios}
|
|
52
|
+
self.assertIn(AttackVector.CHAOS_ENVIRONMENT, vectors_present)
|
|
53
|
+
self.assertIn(AttackVector.BYZANTINE_PAYLOAD, vectors_present)
|
|
54
|
+
self.assertIn(AttackVector.CONCURRENCY_RACE, vectors_present)
|
|
55
|
+
self.assertIn(AttackVector.RESOURCE_EXHAUSTION, vectors_present)
|
|
56
|
+
self.assertIn(AttackVector.STATE_INVARIANT, vectors_present)
|
|
57
|
+
|
|
58
|
+
custom_scenarios = [s for s in scenarios if s.metadata.get("custom")]
|
|
59
|
+
self.assertEqual(len(custom_scenarios), 2)
|
|
60
|
+
self.assertEqual(custom_scenarios[0].vector, AttackVector.CONCURRENCY_RACE)
|
|
61
|
+
self.assertEqual(custom_scenarios[1].vector, AttackVector.BYZANTINE_PAYLOAD)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class TestRedTeamSwarmExecution(unittest.TestCase):
|
|
65
|
+
def setUp(self) -> None:
|
|
66
|
+
self.swarm = RedTeamSwarm()
|
|
67
|
+
|
|
68
|
+
def test_break_detection_fragile_callable(self) -> None:
|
|
69
|
+
def fragile_target(x: Any = None) -> str:
|
|
70
|
+
if x is None:
|
|
71
|
+
raise AttributeError("'NoneType' object has no attribute 'split'")
|
|
72
|
+
if isinstance(x, str) and "\x00" in x:
|
|
73
|
+
raise KeyError("Null byte detected in string dictionary index")
|
|
74
|
+
return "ok"
|
|
75
|
+
|
|
76
|
+
report = self.swarm.execute_swarm_attack(fragile_target)
|
|
77
|
+
self.assertFalse(report.passed)
|
|
78
|
+
self.assertGreater(report.broken_count, 0)
|
|
79
|
+
|
|
80
|
+
# Ensure broken findings contain rich diagnostics
|
|
81
|
+
broken_findings = [f for f in report.findings if f.broken]
|
|
82
|
+
self.assertTrue(len(broken_findings) > 0)
|
|
83
|
+
first_broken = broken_findings[0]
|
|
84
|
+
self.assertIsNotNone(first_broken.error_message)
|
|
85
|
+
self.assertIsNotNone(first_broken.reproduction_code)
|
|
86
|
+
self.assertIn(first_broken.severity, ("CRITICAL", "HIGH", "MEDIUM"))
|
|
87
|
+
self.assertGreater(len(report.remediation_directives), 0)
|
|
88
|
+
|
|
89
|
+
def test_resilience_hardened_callable(self) -> None:
|
|
90
|
+
lock = threading.Lock()
|
|
91
|
+
state = {"calls": 0}
|
|
92
|
+
|
|
93
|
+
def hardened_target(payload: Any = None) -> dict[str, Any]:
|
|
94
|
+
with lock:
|
|
95
|
+
state["calls"] += 1
|
|
96
|
+
if payload is None:
|
|
97
|
+
return {"status": "ok", "data": "default"}
|
|
98
|
+
if isinstance(payload, (int, float)):
|
|
99
|
+
return {"status": "ok", "data": "number"}
|
|
100
|
+
if isinstance(payload, str):
|
|
101
|
+
# Defensively sanitize null bytes and bound size
|
|
102
|
+
sanitized = payload.replace("\x00", "")[:1000]
|
|
103
|
+
return {"status": "ok", "data": sanitized}
|
|
104
|
+
if isinstance(payload, dict):
|
|
105
|
+
return {"status": "ok", "data": "dict_received"}
|
|
106
|
+
return {"status": "ok", "data": str(payload)[:100]}
|
|
107
|
+
|
|
108
|
+
report = self.swarm.execute_swarm_attack(hardened_target)
|
|
109
|
+
self.assertTrue(report.passed)
|
|
110
|
+
self.assertEqual(report.broken_count, 0)
|
|
111
|
+
self.assertEqual(len([f for f in report.findings if f.broken]), 0)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
class TestRedTeamSwarmSecurityBoundary(unittest.TestCase):
|
|
115
|
+
def setUp(self) -> None:
|
|
116
|
+
self.swarm = RedTeamSwarm()
|
|
117
|
+
|
|
118
|
+
def test_none_or_unexecutable_target_fails_closed(self) -> None:
|
|
119
|
+
report = self.swarm.execute_swarm_attack(None)
|
|
120
|
+
self.assertFalse(report.passed)
|
|
121
|
+
self.assertEqual(report.broken_count, 1)
|
|
122
|
+
self.assertEqual(report.findings[0].scenario_id, "target_not_executable")
|
|
123
|
+
self.assertFalse(report.findings[0].details.get("target_executable", True))
|
|
124
|
+
self.assertIn("TypeError", report.findings[0].error_message or "")
|
|
125
|
+
|
|
126
|
+
def test_source_string_target_fails_closed(self) -> None:
|
|
127
|
+
code_snippet = "def safe_fn(x=None):\n return 'ok'\n"
|
|
128
|
+
report = self.swarm.execute_swarm_attack(code_snippet)
|
|
129
|
+
self.assertFalse(report.passed)
|
|
130
|
+
self.assertEqual(report.broken_count, 1)
|
|
131
|
+
self.assertFalse(report.findings[0].details.get("target_executable", True))
|
|
132
|
+
|
|
133
|
+
def test_callable_object_succeeds(self) -> None:
|
|
134
|
+
def safe_fn(x=None):
|
|
135
|
+
return "ok"
|
|
136
|
+
|
|
137
|
+
report = self.swarm.execute_swarm_attack(safe_fn)
|
|
138
|
+
self.assertTrue(report.passed)
|
|
139
|
+
self.assertEqual(report.broken_count, 0)
|
|
140
|
+
|
|
141
|
+
def test_false_bool_callable_object_succeeds(self) -> None:
|
|
142
|
+
class FalseCallable:
|
|
143
|
+
def __bool__(self) -> bool:
|
|
144
|
+
return False
|
|
145
|
+
|
|
146
|
+
def __call__(self, x: Any = None) -> str:
|
|
147
|
+
return "ok"
|
|
148
|
+
|
|
149
|
+
false_callable = FalseCallable()
|
|
150
|
+
report = self.swarm.execute_swarm_attack(false_callable)
|
|
151
|
+
self.assertTrue(report.passed)
|
|
152
|
+
self.assertEqual(report.broken_count, 0)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
class TestReportFormattingAndSerialization(unittest.TestCase):
|
|
156
|
+
def setUp(self) -> None:
|
|
157
|
+
self.swarm = RedTeamSwarm()
|
|
158
|
+
|
|
159
|
+
def test_report_to_dict_and_from_dict(self) -> None:
|
|
160
|
+
finding = BreakFinding(
|
|
161
|
+
scenario_id="sc_01",
|
|
162
|
+
vector="byzantine_payload",
|
|
163
|
+
hypothesis="What will happen on null byte?",
|
|
164
|
+
broken=True,
|
|
165
|
+
error_message="KeyError: null byte",
|
|
166
|
+
traceback_snippet="Traceback ...",
|
|
167
|
+
reproduction_code="target('\x00')",
|
|
168
|
+
severity="HIGH",
|
|
169
|
+
details={"duration_ms": 1.2},
|
|
170
|
+
)
|
|
171
|
+
report = RedTeamBreakageReport(
|
|
172
|
+
report_id="rep_test_123",
|
|
173
|
+
target_name="AuthService",
|
|
174
|
+
total_probes=1,
|
|
175
|
+
broken_count=1,
|
|
176
|
+
passed=False,
|
|
177
|
+
findings=[finding],
|
|
178
|
+
created_at="2026-09-04T12:00:00Z",
|
|
179
|
+
remediation_directives=["Sanitize input null bytes"],
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
data = report.to_dict()
|
|
183
|
+
self.assertEqual(data["report_id"], "rep_test_123")
|
|
184
|
+
self.assertEqual(data["broken_count"], 1)
|
|
185
|
+
self.assertFalse(data["passed"])
|
|
186
|
+
|
|
187
|
+
restored = RedTeamBreakageReport.from_dict(data)
|
|
188
|
+
self.assertEqual(restored.report_id, report.report_id)
|
|
189
|
+
self.assertEqual(restored.target_name, report.target_name)
|
|
190
|
+
self.assertEqual(len(restored.findings), 1)
|
|
191
|
+
self.assertTrue(restored.findings[0].broken)
|
|
192
|
+
|
|
193
|
+
def test_to_markdown_formatting_failed(self) -> None:
|
|
194
|
+
finding = BreakFinding(
|
|
195
|
+
scenario_id="sc_01",
|
|
196
|
+
vector="byzantine_payload",
|
|
197
|
+
hypothesis="What will happen on null byte?",
|
|
198
|
+
broken=True,
|
|
199
|
+
error_message="ValueError: invalid byte",
|
|
200
|
+
traceback_snippet="Traceback ...",
|
|
201
|
+
reproduction_code="target_fn('\x00')",
|
|
202
|
+
severity="HIGH",
|
|
203
|
+
)
|
|
204
|
+
report = RedTeamBreakageReport(
|
|
205
|
+
report_id="rep_test_fail",
|
|
206
|
+
target_name="payment_processor",
|
|
207
|
+
total_probes=5,
|
|
208
|
+
broken_count=1,
|
|
209
|
+
passed=False,
|
|
210
|
+
findings=[finding],
|
|
211
|
+
created_at="2026-09-04T12:00:00Z",
|
|
212
|
+
remediation_directives=["Strip null bytes before processing"],
|
|
213
|
+
)
|
|
214
|
+
md = report.to_markdown()
|
|
215
|
+
self.assertIn("# 🚨 Adversarial Red Team Breakage Report: `payment_processor`", md)
|
|
216
|
+
self.assertIn("> [!CAUTION]", md)
|
|
217
|
+
self.assertIn("Strip null bytes before processing", md)
|
|
218
|
+
self.assertIn("target_fn('\x00')", md)
|
|
219
|
+
|
|
220
|
+
def test_to_markdown_formatting_passed(self) -> None:
|
|
221
|
+
report = RedTeamBreakageReport(
|
|
222
|
+
report_id="rep_test_pass",
|
|
223
|
+
target_name="hardened_crypto",
|
|
224
|
+
total_probes=10,
|
|
225
|
+
broken_count=0,
|
|
226
|
+
passed=True,
|
|
227
|
+
findings=[],
|
|
228
|
+
created_at="2026-09-04T12:00:00Z",
|
|
229
|
+
)
|
|
230
|
+
md = report.to_markdown()
|
|
231
|
+
self.assertIn("# 🛡️ Adversarial Red Team Resilient Attestation: `hardened_crypto`", md)
|
|
232
|
+
self.assertIn("> [!NOTE]", md)
|
|
233
|
+
self.assertIn("🟢 **RESILIENT (PASSED)**", md)
|
|
234
|
+
|
|
235
|
+
def test_document_breakage_to_file(self) -> None:
|
|
236
|
+
report = RedTeamBreakageReport(
|
|
237
|
+
report_id="rep_test_doc",
|
|
238
|
+
target_name="doc_target",
|
|
239
|
+
total_probes=2,
|
|
240
|
+
broken_count=0,
|
|
241
|
+
passed=True,
|
|
242
|
+
findings=[],
|
|
243
|
+
created_at="2026-09-04T12:00:00Z",
|
|
244
|
+
)
|
|
245
|
+
with tempfile.NamedTemporaryFile(suffix=".md", delete=False) as tmp:
|
|
246
|
+
tmp_path = tmp.name
|
|
247
|
+
|
|
248
|
+
try:
|
|
249
|
+
res_md = self.swarm.document_breakage(report, output_path=tmp_path)
|
|
250
|
+
self.assertTrue(os.path.exists(tmp_path))
|
|
251
|
+
with open(tmp_path, "r", encoding="utf-8") as f:
|
|
252
|
+
content = f.read()
|
|
253
|
+
self.assertEqual(res_md, content)
|
|
254
|
+
self.assertIn("doc_target", content)
|
|
255
|
+
finally:
|
|
256
|
+
if os.path.exists(tmp_path):
|
|
257
|
+
os.remove(tmp_path)
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
class TestPingPongRemediationCycle(unittest.TestCase):
|
|
261
|
+
def setUp(self) -> None:
|
|
262
|
+
self.temp_dir = tempfile.TemporaryDirectory()
|
|
263
|
+
self.cortex_path = Path(self.temp_dir.name)
|
|
264
|
+
self.engine = HebbianPlasticityEngine(cortex_dir=self.cortex_path)
|
|
265
|
+
self.swarm = RedTeamSwarm(plasticity_engine=self.engine)
|
|
266
|
+
|
|
267
|
+
def tearDown(self) -> None:
|
|
268
|
+
self.temp_dir.cleanup()
|
|
269
|
+
|
|
270
|
+
def test_full_ping_pong_hardening_cycle(self) -> None:
|
|
271
|
+
# Stage 1: Fragile initial implementation submitted by subagent
|
|
272
|
+
def candidate_v1(input_data: Any = None) -> str:
|
|
273
|
+
# Fragile: crashes unhandled on None, null bytes, and empty inputs
|
|
274
|
+
if input_data is None:
|
|
275
|
+
raise AttributeError("'NoneType' object has no attribute 'strip'")
|
|
276
|
+
if "\x00" in input_data:
|
|
277
|
+
raise KeyError("Byzantine null byte rejected with raw error")
|
|
278
|
+
return f"processed_{input_data}"
|
|
279
|
+
|
|
280
|
+
# Stage 2: Swarm attack breaks candidate_v1 -> auto-consolidates with LTD
|
|
281
|
+
initial_report = self.swarm.run_full_review_cycle(
|
|
282
|
+
candidate_v1,
|
|
283
|
+
target_name="candidate_service",
|
|
284
|
+
)
|
|
285
|
+
self.assertFalse(initial_report.passed)
|
|
286
|
+
self.assertGreater(initial_report.broken_count, 0)
|
|
287
|
+
self.assertGreater(len(initial_report.remediation_directives), 0)
|
|
288
|
+
|
|
289
|
+
# Verify automated closed-loop LTD consolidation
|
|
290
|
+
lobe = self.engine._load_or_create_lobe("candidate_service")
|
|
291
|
+
self.assertGreaterEqual(len(lobe.antibodies), 1)
|
|
292
|
+
self.assertIn("test_harness", lobe.synaptic_weights)
|
|
293
|
+
# Weight depressed under failure
|
|
294
|
+
self.assertLess(lobe.synaptic_weights["test_harness"], 0.30)
|
|
295
|
+
|
|
296
|
+
# Stage 3: Subagent remediates and hardens implementation
|
|
297
|
+
lock = threading.Lock()
|
|
298
|
+
|
|
299
|
+
def candidate_v2(input_data: Any = None) -> str:
|
|
300
|
+
with lock:
|
|
301
|
+
if input_data is None:
|
|
302
|
+
return "processed_default"
|
|
303
|
+
if isinstance(input_data, str):
|
|
304
|
+
clean = input_data.replace("\x00", "")[:200]
|
|
305
|
+
return f"processed_{clean}"
|
|
306
|
+
return f"processed_{str(input_data)[:200]}"
|
|
307
|
+
|
|
308
|
+
# Stage 4: Swarm re-attacks and verifies remediation -> auto-consolidates with LTP
|
|
309
|
+
all_fixed, new_report = self.swarm.verify_remediation(
|
|
310
|
+
target_callable=candidate_v2,
|
|
311
|
+
prior_report=initial_report,
|
|
312
|
+
)
|
|
313
|
+
|
|
314
|
+
self.assertTrue(all_fixed)
|
|
315
|
+
self.assertTrue(new_report.passed)
|
|
316
|
+
self.assertEqual(new_report.broken_count, 0)
|
|
317
|
+
|
|
318
|
+
# Verify automated closed-loop LTP consolidation and antibody minting
|
|
319
|
+
lobe_reloaded = self.engine._load_or_create_lobe("candidate_service")
|
|
320
|
+
self.assertGreater(len(lobe_reloaded.antibodies), 0)
|
|
321
|
+
self.assertIn("mutation", lobe_reloaded.synaptic_weights)
|
|
322
|
+
# Remediated pathway potentiated (LTP)
|
|
323
|
+
self.assertGreater(lobe_reloaded.synaptic_weights["mutation"], 0.30)
|
|
324
|
+
self.assertGreater(len(lobe_reloaded.specialized_heuristics), 0)
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
class TestPublicActionHandlers(unittest.TestCase):
|
|
328
|
+
def test_handle_red_team_code_review_rejects_source_string_without_mutation(self) -> None:
|
|
329
|
+
from fable_engine.actions.fleet import _handle_red_team_code_review
|
|
330
|
+
from fable_engine.session import FableSession, ACTIVE_SESSIONS, SESSIONS_DIR
|
|
331
|
+
session_name = f"test_no_mutate_{uuid.uuid4().hex[:8]}"
|
|
332
|
+
session_file = SESSIONS_DIR / f"{session_name}.json"
|
|
333
|
+
session = FableSession(session_name=session_name, objective="Test no mutation", time_budget_minutes=5.0)
|
|
334
|
+
ACTIVE_SESSIONS[session_name] = session
|
|
335
|
+
snapshot = copy.deepcopy(session.to_dict())
|
|
336
|
+
|
|
337
|
+
try:
|
|
338
|
+
resp = _handle_red_team_code_review({
|
|
339
|
+
"action": "red_team_code_review",
|
|
340
|
+
"session_name": session_name,
|
|
341
|
+
"target_code": "def process(): pass",
|
|
342
|
+
})
|
|
343
|
+
self.assertIn("Error: Source-code strings cannot be evaluated in-process for security reasons", resp)
|
|
344
|
+
self.assertEqual(session.to_dict(), snapshot)
|
|
345
|
+
finally:
|
|
346
|
+
ACTIVE_SESSIONS.pop(session_name, None)
|
|
347
|
+
if session_file.exists():
|
|
348
|
+
session_file.unlink()
|
|
349
|
+
|
|
350
|
+
def test_handle_verify_red_team_remediation_rejects_source_string_without_mutation(self) -> None:
|
|
351
|
+
from fable_engine.actions.fleet import _handle_verify_red_team_remediation
|
|
352
|
+
from fable_engine.session import FableSession, ACTIVE_SESSIONS, SESSIONS_DIR
|
|
353
|
+
session_name = f"test_no_mutate_{uuid.uuid4().hex[:8]}"
|
|
354
|
+
session_file = SESSIONS_DIR / f"{session_name}.json"
|
|
355
|
+
session = FableSession(session_name=session_name, objective="Test no mutation", time_budget_minutes=5.0)
|
|
356
|
+
ACTIVE_SESSIONS[session_name] = session
|
|
357
|
+
snapshot = copy.deepcopy(session.to_dict())
|
|
358
|
+
|
|
359
|
+
try:
|
|
360
|
+
resp = _handle_verify_red_team_remediation({
|
|
361
|
+
"action": "verify_red_team_remediation",
|
|
362
|
+
"session_name": session_name,
|
|
363
|
+
"remediated_code": "def process(): pass",
|
|
364
|
+
})
|
|
365
|
+
self.assertIn("Error: Source-code strings cannot be evaluated in-process for security reasons", resp)
|
|
366
|
+
self.assertEqual(session.to_dict(), snapshot)
|
|
367
|
+
finally:
|
|
368
|
+
ACTIVE_SESSIONS.pop(session_name, None)
|
|
369
|
+
if session_file.exists():
|
|
370
|
+
session_file.unlink()
|
|
371
|
+
|
|
372
|
+
def test_rejected_source_string_with_nonexistent_session_name(self) -> None:
|
|
373
|
+
from fable_engine.actions.fleet import _handle_red_team_code_review, _handle_verify_red_team_remediation
|
|
374
|
+
from fable_engine.session import ACTIVE_SESSIONS, SESSIONS_DIR
|
|
375
|
+
|
|
376
|
+
nonexistent_name = f"nonexistent_session_{uuid.uuid4().hex[:8]}"
|
|
377
|
+
session_file = SESSIONS_DIR / f"{nonexistent_name}.json"
|
|
378
|
+
|
|
379
|
+
ACTIVE_SESSIONS.pop(nonexistent_name, None)
|
|
380
|
+
if session_file.exists():
|
|
381
|
+
session_file.unlink()
|
|
382
|
+
|
|
383
|
+
try:
|
|
384
|
+
self.assertNotIn(nonexistent_name, ACTIVE_SESSIONS)
|
|
385
|
+
|
|
386
|
+
resp1 = _handle_red_team_code_review({
|
|
387
|
+
"action": "red_team_code_review",
|
|
388
|
+
"session_name": nonexistent_name,
|
|
389
|
+
"target_code": "def process(): pass",
|
|
390
|
+
})
|
|
391
|
+
self.assertIn("Error: Source-code strings cannot be evaluated in-process for security reasons", resp1)
|
|
392
|
+
self.assertNotIn(nonexistent_name, ACTIVE_SESSIONS)
|
|
393
|
+
self.assertFalse(session_file.exists())
|
|
394
|
+
|
|
395
|
+
resp2 = _handle_verify_red_team_remediation({
|
|
396
|
+
"action": "verify_red_team_remediation",
|
|
397
|
+
"session_name": nonexistent_name,
|
|
398
|
+
"remediated_code": "def process(): pass",
|
|
399
|
+
})
|
|
400
|
+
self.assertIn("Error: Source-code strings cannot be evaluated in-process for security reasons", resp2)
|
|
401
|
+
self.assertNotIn(nonexistent_name, ACTIVE_SESSIONS)
|
|
402
|
+
self.assertFalse(session_file.exists())
|
|
403
|
+
finally:
|
|
404
|
+
ACTIVE_SESSIONS.pop(nonexistent_name, None)
|
|
405
|
+
if session_file.exists():
|
|
406
|
+
session_file.unlink()
|
|
407
|
+
|
|
408
|
+
def test_missing_session_name_does_not_create_session_or_file(self) -> None:
|
|
409
|
+
from fable_engine.actions.fleet import _handle_red_team_code_review, _handle_verify_red_team_remediation
|
|
410
|
+
from fable_engine.session import ACTIVE_SESSIONS, SESSIONS_DIR
|
|
411
|
+
|
|
412
|
+
initial_active_keys = set(ACTIVE_SESSIONS.keys())
|
|
413
|
+
|
|
414
|
+
resp1 = _handle_red_team_code_review({"action": "red_team_code_review"})
|
|
415
|
+
self.assertIn("Error: 'session_name' is required", resp1)
|
|
416
|
+
|
|
417
|
+
resp2 = _handle_verify_red_team_remediation({"action": "verify_red_team_remediation"})
|
|
418
|
+
self.assertIn("Error: 'session_name' is required", resp2)
|
|
419
|
+
|
|
420
|
+
self.assertEqual(set(ACTIVE_SESSIONS.keys()), initial_active_keys)
|
|
421
|
+
self.assertFalse((SESSIONS_DIR / ".json").exists())
|
|
422
|
+
|
|
423
|
+
def test_falsey_callable_object_accepted_in_public_action_handlers(self) -> None:
|
|
424
|
+
from fable_engine.actions.fleet import _handle_red_team_code_review
|
|
425
|
+
from fable_engine.session import FableSession, ACTIVE_SESSIONS, SESSIONS_DIR, SessionState
|
|
426
|
+
|
|
427
|
+
class FalseCallable:
|
|
428
|
+
def __bool__(self) -> bool:
|
|
429
|
+
return False
|
|
430
|
+
|
|
431
|
+
def __call__(self, x: Any = None) -> str:
|
|
432
|
+
return "ok"
|
|
433
|
+
|
|
434
|
+
session_name = f"test_falsey_callable_{uuid.uuid4().hex[:8]}"
|
|
435
|
+
session_file = SESSIONS_DIR / f"{session_name}.json"
|
|
436
|
+
|
|
437
|
+
session = FableSession(session_name=session_name, objective="Test falsey callable", time_budget_minutes=5.0)
|
|
438
|
+
session.set_timer(5.0)
|
|
439
|
+
session.log_epistemic_item("PROVEN", "Evidence item 1", evidence="README.md:L1")
|
|
440
|
+
session.log_epistemic_item("PROVEN", "Evidence item 2", evidence="README.md:L5")
|
|
441
|
+
session.proof_receipts.append({"receipt_id": "falsey-review-receipt", "verified": True})
|
|
442
|
+
session.set_goal_rubric("Test Rubric", [{
|
|
443
|
+
"pointer_id": "P1",
|
|
444
|
+
"description": "Check 1",
|
|
445
|
+
"satisfied": True,
|
|
446
|
+
"score": 1.0,
|
|
447
|
+
"verifier_command": "python -m unittest tests.test_red_team_swarm",
|
|
448
|
+
"evidence_receipt_id": "falsey-review-receipt",
|
|
449
|
+
}])
|
|
450
|
+
session.log_refinement_cycle("refine", "core", "bottleneck", "refinement")
|
|
451
|
+
session.execution_locked = False
|
|
452
|
+
session.can_execute_code = True
|
|
453
|
+
session.transition_to(SessionState.IMPLEMENTATION, "Implemented")
|
|
454
|
+
session.track_file_change("sample.py", "created", "Added initial implementation")
|
|
455
|
+
session.transition_to(SessionState.RED_TEAM_GATE, "Ready for gate")
|
|
456
|
+
ACTIVE_SESSIONS[session_name] = session
|
|
457
|
+
|
|
458
|
+
try:
|
|
459
|
+
resp = _handle_red_team_code_review({
|
|
460
|
+
"action": "red_team_code_review",
|
|
461
|
+
"session_name": session_name,
|
|
462
|
+
"target_code": FalseCallable(),
|
|
463
|
+
})
|
|
464
|
+
self.assertNotIn("Error: Source-code strings cannot be evaluated in-process", resp)
|
|
465
|
+
self.assertIn("Adversarial Red Team Resilient Attestation", resp)
|
|
466
|
+
finally:
|
|
467
|
+
ACTIVE_SESSIONS.pop(session_name, None)
|
|
468
|
+
if session_file.exists():
|
|
469
|
+
session_file.unlink()
|
|
470
|
+
|
|
471
|
+
def test_falsey_callable_object_accepted_in_verify_red_team_remediation(self) -> None:
|
|
472
|
+
from fable_engine.actions.fleet import _handle_verify_red_team_remediation
|
|
473
|
+
from fable_engine.session import FableSession, ACTIVE_SESSIONS, SESSIONS_DIR, SessionState
|
|
474
|
+
|
|
475
|
+
class FalseCallable:
|
|
476
|
+
def __bool__(self) -> bool:
|
|
477
|
+
return False
|
|
478
|
+
|
|
479
|
+
def __call__(self, x: Any = None) -> str:
|
|
480
|
+
return "ok"
|
|
481
|
+
|
|
482
|
+
session_name = f"test_falsey_remediation_{uuid.uuid4().hex[:8]}"
|
|
483
|
+
session_file = SESSIONS_DIR / f"{session_name}.json"
|
|
484
|
+
|
|
485
|
+
session = FableSession(session_name=session_name, objective="Test falsey remediation", time_budget_minutes=5.0)
|
|
486
|
+
session.set_timer(5.0)
|
|
487
|
+
session.log_epistemic_item("PROVEN", "Evidence item 1", evidence="README.md:L1")
|
|
488
|
+
session.log_epistemic_item("PROVEN", "Evidence item 2", evidence="README.md:L5")
|
|
489
|
+
session.proof_receipts.append({"receipt_id": "falsey-remediation-receipt", "verified": True})
|
|
490
|
+
session.set_goal_rubric("Test Rubric", [{
|
|
491
|
+
"pointer_id": "P1",
|
|
492
|
+
"description": "Check 1",
|
|
493
|
+
"satisfied": True,
|
|
494
|
+
"score": 1.0,
|
|
495
|
+
"verifier_command": "python -m unittest tests.test_red_team_swarm",
|
|
496
|
+
"evidence_receipt_id": "falsey-remediation-receipt",
|
|
497
|
+
}])
|
|
498
|
+
session.log_refinement_cycle("refine", "core", "bottleneck", "refinement")
|
|
499
|
+
session.execution_locked = False
|
|
500
|
+
session.can_execute_code = True
|
|
501
|
+
session.transition_to(SessionState.IMPLEMENTATION, "Implemented")
|
|
502
|
+
session.track_file_change("sample.py", "created", "Added initial implementation")
|
|
503
|
+
session.transition_to(SessionState.RED_TEAM_GATE, "Ready for gate")
|
|
504
|
+
ACTIVE_SESSIONS[session_name] = session
|
|
505
|
+
|
|
506
|
+
prior_report = {
|
|
507
|
+
"report_id": "rep_prior_falsey",
|
|
508
|
+
"target_name": "target",
|
|
509
|
+
"broken_count": 1,
|
|
510
|
+
"findings": [
|
|
511
|
+
{
|
|
512
|
+
"scenario_id": "target_chaos_01_missing_path",
|
|
513
|
+
"vector": "chaos_environment",
|
|
514
|
+
"hypothesis": "Hypothesis",
|
|
515
|
+
"broken": True,
|
|
516
|
+
}
|
|
517
|
+
],
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
try:
|
|
521
|
+
resp = _handle_verify_red_team_remediation({
|
|
522
|
+
"action": "verify_red_team_remediation",
|
|
523
|
+
"session_name": session_name,
|
|
524
|
+
"remediated_code": FalseCallable(),
|
|
525
|
+
"prior_report": prior_report,
|
|
526
|
+
})
|
|
527
|
+
self.assertNotIn("Error: Source-code strings cannot be evaluated in-process", resp)
|
|
528
|
+
self.assertIn("TASK COMPLETED: 0 breakages remain. Code sealed.", resp)
|
|
529
|
+
finally:
|
|
530
|
+
ACTIVE_SESSIONS.pop(session_name, None)
|
|
531
|
+
if session_file.exists():
|
|
532
|
+
session_file.unlink()
|
|
533
|
+
|
|
534
|
+
def test_verify_remediation_does_not_seal_unbound_satisfied_criterion(self) -> None:
|
|
535
|
+
from fable_engine.actions.fleet import _handle_verify_red_team_remediation
|
|
536
|
+
from fable_engine.session import FableSession, ACTIVE_SESSIONS, SESSIONS_DIR, SessionState
|
|
537
|
+
|
|
538
|
+
def remediated(value: Any = None) -> str:
|
|
539
|
+
return "ok"
|
|
540
|
+
|
|
541
|
+
session_name = f"test_unbound_rubric_{uuid.uuid4().hex[:8]}"
|
|
542
|
+
session_file = SESSIONS_DIR / f"{session_name}.json"
|
|
543
|
+
session = FableSession(session_name, "Reject unbound criterion", 5.0)
|
|
544
|
+
session.set_timer(5.0)
|
|
545
|
+
session.log_epistemic_item("PROVEN", "Evidence item one", evidence="README.md:L1")
|
|
546
|
+
session.log_epistemic_item("PROVEN", "Evidence item two", evidence="README.md:L5")
|
|
547
|
+
rubric = session.set_goal_rubric(
|
|
548
|
+
"Unbound rubric",
|
|
549
|
+
[{"pointer_id": "P1", "satisfied": True, "score": 1.0}],
|
|
550
|
+
)
|
|
551
|
+
self.assertNotEqual(rubric["status"], "achieved")
|
|
552
|
+
session.log_refinement_cycle("refine", "sealing", "unbound evidence", "require evidence binding")
|
|
553
|
+
session.execution_locked = False
|
|
554
|
+
session.can_execute_code = True
|
|
555
|
+
session.transition_to(SessionState.IMPLEMENTATION, "Implemented")
|
|
556
|
+
session.track_file_change("sample.py", "modified", "Hardened implementation")
|
|
557
|
+
session.transition_to(SessionState.RED_TEAM_GATE, "Ready for gate")
|
|
558
|
+
ACTIVE_SESSIONS[session_name] = session
|
|
559
|
+
prior_report = {
|
|
560
|
+
"report_id": "rep_unbound",
|
|
561
|
+
"target_name": "target",
|
|
562
|
+
"total_probes": 1,
|
|
563
|
+
"broken_count": 1,
|
|
564
|
+
"passed": False,
|
|
565
|
+
"findings": [{
|
|
566
|
+
"scenario_id": "target_chaos_01_missing_path",
|
|
567
|
+
"vector": "chaos_environment",
|
|
568
|
+
"hypothesis": "Missing path",
|
|
569
|
+
"broken": True,
|
|
570
|
+
}],
|
|
571
|
+
}
|
|
572
|
+
|
|
573
|
+
try:
|
|
574
|
+
response = _handle_verify_red_team_remediation({
|
|
575
|
+
"session_name": session_name,
|
|
576
|
+
"remediated_code": remediated,
|
|
577
|
+
"prior_report": prior_report,
|
|
578
|
+
})
|
|
579
|
+
self.assertIn("Satisfied rubric criteria require", response)
|
|
580
|
+
self.assertNotIn("Code sealed", response)
|
|
581
|
+
self.assertNotEqual(session.current_state, SessionState.SEALED)
|
|
582
|
+
finally:
|
|
583
|
+
ACTIVE_SESSIONS.pop(session_name, None)
|
|
584
|
+
if session_file.exists():
|
|
585
|
+
session_file.unlink()
|
|
586
|
+
|
|
587
|
+
|
|
588
|
+
class TestCoderFleetDispatcherRedTeamActions(unittest.TestCase):
|
|
589
|
+
def setUp(self) -> None:
|
|
590
|
+
self.temp_dir = tempfile.TemporaryDirectory()
|
|
591
|
+
self.cortex_path = Path(self.temp_dir.name)
|
|
592
|
+
self.engine = HebbianPlasticityEngine(cortex_dir=self.cortex_path)
|
|
593
|
+
self.dispatcher = CoderFleetDispatcher(plasticity_engine=self.engine)
|
|
594
|
+
|
|
595
|
+
def tearDown(self) -> None:
|
|
596
|
+
self.temp_dir.cleanup()
|
|
597
|
+
|
|
598
|
+
def test_registered_red_team_actions(self) -> None:
|
|
599
|
+
actions = self.dispatcher.list_actions()
|
|
600
|
+
expected = [
|
|
601
|
+
"red_team_generate_scenarios",
|
|
602
|
+
"red_team_execute_attack",
|
|
603
|
+
"red_team_document_breakage",
|
|
604
|
+
"red_team_verify_remediation",
|
|
605
|
+
"red_team_full_review_cycle",
|
|
606
|
+
]
|
|
607
|
+
for act in expected:
|
|
608
|
+
self.assertIn(act, actions)
|
|
609
|
+
|
|
610
|
+
def test_dispatch_red_team_generate_scenarios(self) -> None:
|
|
611
|
+
res = self.dispatcher.dispatch(
|
|
612
|
+
"red_team_generate_scenarios",
|
|
613
|
+
{"target_name": "AuthModule", "custom_hypotheses": ["What if token is empty?"]},
|
|
614
|
+
)
|
|
615
|
+
self.assertTrue(res["success"])
|
|
616
|
+
scenarios = res["result"]
|
|
617
|
+
self.assertIsInstance(scenarios, list)
|
|
618
|
+
self.assertGreater(len(scenarios), 0)
|
|
619
|
+
|
|
620
|
+
def test_dispatch_red_team_full_review_cycle(self) -> None:
|
|
621
|
+
def safe_fn(x=None):
|
|
622
|
+
return "safe"
|
|
623
|
+
|
|
624
|
+
res = self.dispatcher.dispatch(
|
|
625
|
+
"red_team_full_review_cycle",
|
|
626
|
+
{"target_callable": safe_fn, "target_name": "safe_fn"},
|
|
627
|
+
)
|
|
628
|
+
self.assertTrue(res["success"])
|
|
629
|
+
report = res["result"]
|
|
630
|
+
self.assertTrue(isinstance(report, RedTeamBreakageReport))
|
|
631
|
+
self.assertTrue(report.passed)
|
|
632
|
+
|
|
633
|
+
# Source code strings produce fail-closed breakage reports
|
|
634
|
+
res_str = self.dispatcher.dispatch(
|
|
635
|
+
"red_team_full_review_cycle",
|
|
636
|
+
{"target_callable": "def safe_fn(x=None):\n return 'safe'\n", "target_name": "safe_fn"},
|
|
637
|
+
)
|
|
638
|
+
self.assertTrue(res_str["success"])
|
|
639
|
+
report_str = res_str["result"]
|
|
640
|
+
self.assertFalse(report_str.passed)
|
|
641
|
+
self.assertEqual(report_str.broken_count, 1)
|
|
642
|
+
|
|
643
|
+
|
|
644
|
+
if __name__ == "__main__":
|
|
645
|
+
unittest.main()
|