fable-engine 1.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fable_compressor.py +356 -0
- fable_engine/__init__.py +1 -0
- fable_engine/actions/__init__.py +291 -0
- fable_engine/actions/cas.py +182 -0
- fable_engine/actions/deliberation.py +523 -0
- fable_engine/actions/fleet.py +807 -0
- fable_engine/actions/lifecycle.py +298 -0
- fable_engine/actions/scrapers.py +116 -0
- fable_engine/actions/system3.py +815 -0
- fable_engine/browser.py +824 -0
- fable_engine/cas.py +974 -0
- fable_engine/fable_session.json +510 -0
- fable_engine/guards.py +283 -0
- fable_engine/schema.py +714 -0
- fable_engine/scrapers/__init__.py +32 -0
- fable_engine/scrapers/arxiv.py +115 -0
- fable_engine/scrapers/base.py +386 -0
- fable_engine/scrapers/github.py +129 -0
- fable_engine/scrapers/reddit.py +154 -0
- fable_engine/scrapers/web.py +120 -0
- fable_engine/scrapers/x.py +125 -0
- fable_engine/scrapers/youtube.py +132 -0
- fable_engine/server.py +414 -0
- fable_engine/session.py +1819 -0
- fable_engine/test_server.py +1362 -0
- fable_engine/updater.py +541 -0
- fable_engine-1.3.1.dist-info/LICENSE +22 -0
- fable_engine-1.3.1.dist-info/METADATA +173 -0
- fable_engine-1.3.1.dist-info/RECORD +104 -0
- fable_engine-1.3.1.dist-info/WHEEL +5 -0
- fable_engine-1.3.1.dist-info/entry_points.txt +5 -0
- fable_engine-1.3.1.dist-info/top_level.txt +6 -0
- fable_mode/__init__.py +3 -0
- fable_mode/__main__.py +4 -0
- fable_mode/adapters.py +1014 -0
- fable_mode/installer.py +553 -0
- fable_mode/launcher.py +437 -0
- fable_mode/manifest.py +142 -0
- fable_mode/resources.json +114 -0
- fable_mode/safety.py +103 -0
- fable_mode_entry.py +10 -0
- fable_v2/__init__.py +146 -0
- fable_v2/adapters.py +151 -0
- fable_v2/coder_fleet/__init__.py +100 -0
- fable_v2/coder_fleet/ast_tools.py +158 -0
- fable_v2/coder_fleet/compute.py +199 -0
- fable_v2/coder_fleet/design_engine.py +1316 -0
- fable_v2/coder_fleet/diagnostics.py +293 -0
- fable_v2/coder_fleet/fleet_dispatcher.py +214 -0
- fable_v2/coder_fleet/mock_auditor.py +306 -0
- fable_v2/coder_fleet/mutation.py +216 -0
- fable_v2/coder_fleet/property_oracle.py +260 -0
- fable_v2/coder_fleet/receipt_attestor.py +122 -0
- fable_v2/coder_fleet/red_team_swarm.py +908 -0
- fable_v2/coder_fleet/test_harness.py +198 -0
- fable_v2/coder_fleet/vector_engine.py +1287 -0
- fable_v2/coder_fleet/visual.py +357 -0
- fable_v2/coder_fleet/workspace.py +153 -0
- fable_v2/cortical/__init__.py +20 -0
- fable_v2/cortical/plasticity_engine.py +992 -0
- fable_v2/execution_broker.py +811 -0
- fable_v2/proof_engine.py +1141 -0
- fable_v2/protocol.py +485 -0
- fable_v2/runtime.py +1010 -0
- fable_v2/system3/__init__.py +204 -0
- fable_v2/system3/causal.py +558 -0
- fable_v2/system3/dialectical.py +577 -0
- fable_v2/system3/evolution.py +503 -0
- fable_v2/system3/executive.py +338 -0
- fable_v2/system3/free_energy.py +479 -0
- fable_v2/system3/hyperbolic.py +555 -0
- fable_v2/system3/induction.py +336 -0
- fable_v2/system3/kripke.py +548 -0
- fable_v2/system3/oracle.py +745 -0
- fable_v2/verifiers.py +72 -0
- tests/__init__.py +1 -0
- tests/test_anti_loop_circuit_breaker.py +64 -0
- tests/test_auto_updater.py +407 -0
- tests/test_coder_fleet.py +535 -0
- tests/test_delegation_compiler.py +54 -0
- tests/test_descriptor_boundaries.py +126 -0
- tests/test_design_engine.py +603 -0
- tests/test_epistemic_evidence_validator.py +66 -0
- tests/test_execution_broker.py +233 -0
- tests/test_fable_v2.py +406 -0
- tests/test_fleet_transitions.py +116 -0
- tests/test_fsm_redteam_evolution.py +406 -0
- tests/test_goal_rubric_and_pipeline.py +367 -0
- tests/test_hebbian_plasticity.py +585 -0
- tests/test_packaging_runtime.py +194 -0
- tests/test_proof_engine.py +259 -0
- tests/test_red_team_swarm.py +645 -0
- tests/test_redteam_remediation.py +169 -0
- tests/test_registration_transaction.py +375 -0
- tests/test_requested_regressions.py +467 -0
- tests/test_scrapers.py +370 -0
- tests/test_server_actions.py +93 -0
- tests/test_server_frontier_actions.py +269 -0
- tests/test_server_protocol.py +88 -0
- tests/test_stealth_browser.py +970 -0
- tests/test_system3.py +381 -0
- tests/test_system3_deep_integration.py +385 -0
- tests/test_system3_frontier.py +436 -0
- tests/test_vector_engine.py +608 -0
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
import tempfile
|
|
2
|
+
import unittest
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from types import SimpleNamespace
|
|
5
|
+
from unittest.mock import patch
|
|
6
|
+
|
|
7
|
+
from fable_engine.actions.fleet import (
|
|
8
|
+
_handle_evolve_cortex,
|
|
9
|
+
_handle_record_breakage_report,
|
|
10
|
+
_handle_verify_red_team_remediation,
|
|
11
|
+
)
|
|
12
|
+
from fable_engine.session import FableSession, SessionState
|
|
13
|
+
from fable_v2.cortical import HebbianPlasticityEngine
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class FleetTransitionRegressionTests(unittest.TestCase):
|
|
17
|
+
@staticmethod
|
|
18
|
+
def _red_team_session(name: str) -> FableSession:
|
|
19
|
+
session = FableSession(name, "Fleet transition regression", 5.0)
|
|
20
|
+
session.set_timer(5.0)
|
|
21
|
+
session.execution_locked = False
|
|
22
|
+
session.can_execute_code = True
|
|
23
|
+
session.transition_to(SessionState.IMPLEMENTATION, "Execution unlocked")
|
|
24
|
+
session.track_file_change("sample.py", "modified", "Exercise fleet transition")
|
|
25
|
+
session.transition_to(SessionState.RED_TEAM_GATE, "Code submitted for review")
|
|
26
|
+
return session
|
|
27
|
+
|
|
28
|
+
def test_record_breakage_report_handler_transitions_to_remediation(self) -> None:
|
|
29
|
+
session = self._red_team_session("record_handler_transition")
|
|
30
|
+
with (
|
|
31
|
+
patch("fable_engine.actions.fleet.get_or_load_session", return_value=session),
|
|
32
|
+
patch.object(session, "save"),
|
|
33
|
+
):
|
|
34
|
+
response = _handle_record_breakage_report({
|
|
35
|
+
"session_name": session.session_name,
|
|
36
|
+
"findings": [{"scenario_id": "missing-broken-is-active"}],
|
|
37
|
+
})
|
|
38
|
+
|
|
39
|
+
self.assertIn("TASK REJECTED", response)
|
|
40
|
+
self.assertEqual(session.current_state, SessionState.REMEDIATION_REQUIRED)
|
|
41
|
+
self.assertEqual(len(session.active_breakages), 1)
|
|
42
|
+
|
|
43
|
+
def test_verify_remediation_handler_keeps_failed_session_unsealed(self) -> None:
|
|
44
|
+
session = self._red_team_session("verify_handler_transition")
|
|
45
|
+
session.record_breakage_report({
|
|
46
|
+
"report_id": "prior",
|
|
47
|
+
"total_probes": 1,
|
|
48
|
+
"broken_count": 1,
|
|
49
|
+
"passed": False,
|
|
50
|
+
"findings": [{"scenario_id": "still-broken", "broken": True}],
|
|
51
|
+
})
|
|
52
|
+
failed_report = {
|
|
53
|
+
"report_id": "verification",
|
|
54
|
+
"total_probes": 1,
|
|
55
|
+
"broken_count": 1,
|
|
56
|
+
"passed": False,
|
|
57
|
+
"findings": [{"scenario_id": "still-broken", "broken": True}],
|
|
58
|
+
}
|
|
59
|
+
swarm = SimpleNamespace(
|
|
60
|
+
verify_remediation=lambda **_kwargs: (
|
|
61
|
+
False,
|
|
62
|
+
SimpleNamespace(broken_count=1, to_dict=lambda: failed_report),
|
|
63
|
+
)
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
with (
|
|
67
|
+
patch("fable_engine.actions.fleet.get_or_load_session", return_value=session),
|
|
68
|
+
patch("fable_engine.actions.fleet._get_swarm", return_value=swarm),
|
|
69
|
+
patch.object(session, "save"),
|
|
70
|
+
):
|
|
71
|
+
response = _handle_verify_red_team_remediation({
|
|
72
|
+
"session_name": session.session_name,
|
|
73
|
+
"prior_report": session.breakage_reports[-1],
|
|
74
|
+
"remediated_code": lambda: None,
|
|
75
|
+
})
|
|
76
|
+
|
|
77
|
+
self.assertIn("TASK REJECTED", response)
|
|
78
|
+
self.assertEqual(session.current_state, SessionState.REMEDIATION_REQUIRED)
|
|
79
|
+
self.assertEqual(session.remediation_attempt_count, 2)
|
|
80
|
+
|
|
81
|
+
def test_direct_seal_requires_validated_clean_report(self) -> None:
|
|
82
|
+
session = self._red_team_session("direct_seal_guard")
|
|
83
|
+
session.transition_to(SessionState.ARBITRATION, "Review complete")
|
|
84
|
+
|
|
85
|
+
with self.assertRaisesRegex(ValueError, "requires a clean report"):
|
|
86
|
+
session.transition_to(SessionState.SEALED, "Direct seal probe")
|
|
87
|
+
|
|
88
|
+
def test_evolve_cortex_reports_effective_storage_path(self) -> None:
|
|
89
|
+
session = self._red_team_session("effective_cortex_path")
|
|
90
|
+
session.transition_to(SessionState.ARBITRATION, "Review complete")
|
|
91
|
+
session._sealing_authorized = True
|
|
92
|
+
try:
|
|
93
|
+
session.transition_to(SessionState.SEALED, "Validated test seal")
|
|
94
|
+
finally:
|
|
95
|
+
session._sealing_authorized = False
|
|
96
|
+
|
|
97
|
+
with tempfile.TemporaryDirectory() as temp_dir:
|
|
98
|
+
engine = HebbianPlasticityEngine(cortex_dir=Path(temp_dir) / "cortex")
|
|
99
|
+
expected_path = engine._get_lobe_path("python").resolve()
|
|
100
|
+
with (
|
|
101
|
+
patch("fable_engine.actions.fleet.get_or_load_session", return_value=session),
|
|
102
|
+
patch("fable_engine.actions.fleet._get_cortex", return_value=engine),
|
|
103
|
+
patch.object(session, "save"),
|
|
104
|
+
):
|
|
105
|
+
response = _handle_evolve_cortex({
|
|
106
|
+
"session_name": session.session_name,
|
|
107
|
+
"domain": "python",
|
|
108
|
+
})
|
|
109
|
+
|
|
110
|
+
self.assertEqual(session.current_state, SessionState.EVOLVED)
|
|
111
|
+
self.assertEqual(response.count(f"`{expected_path}`"), 2)
|
|
112
|
+
self.assertNotIn("skills/fable-mode/cortex", response)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
if __name__ == "__main__":
|
|
116
|
+
unittest.main()
|
|
@@ -0,0 +1,406 @@
|
|
|
1
|
+
"""Unit test suite verifying the MCP-Governed Closed-Loop Red-Team & Evolution Engine Architecture.
|
|
2
|
+
|
|
3
|
+
Validates:
|
|
4
|
+
1. test_fsm_illegal_transitions_blocked
|
|
5
|
+
2. test_red_team_gating_and_rejection_order
|
|
6
|
+
3. test_closed_loop_ping_pong_remediation_and_sealing
|
|
7
|
+
4. test_post_success_cortical_evolution
|
|
8
|
+
"""
|
|
9
|
+
import copy
|
|
10
|
+
import json
|
|
11
|
+
import os
|
|
12
|
+
import sys
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
import shutil
|
|
15
|
+
import tempfile
|
|
16
|
+
import time
|
|
17
|
+
import unittest
|
|
18
|
+
|
|
19
|
+
# Ensure workspace root is in sys.path
|
|
20
|
+
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
21
|
+
|
|
22
|
+
from fable_engine.server import (
|
|
23
|
+
FableSession,
|
|
24
|
+
SessionState,
|
|
25
|
+
VALID_TRANSITIONS,
|
|
26
|
+
PHASES,
|
|
27
|
+
handle_fable_session,
|
|
28
|
+
GLOBAL_PLASTICITY_ENGINE,
|
|
29
|
+
ACTIVE_SESSIONS,
|
|
30
|
+
SESSIONS_DIR,
|
|
31
|
+
get_or_load_session,
|
|
32
|
+
)
|
|
33
|
+
from fable_engine.session import RED_TEAM_ATTACK_VECTORS
|
|
34
|
+
from fable_v2.coder_fleet.red_team_swarm import RedTeamBreakageReport, BreakFinding
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class TestFSMRedTeamEvolution(unittest.TestCase):
|
|
38
|
+
def setUp(self):
|
|
39
|
+
ACTIVE_SESSIONS.clear()
|
|
40
|
+
self.test_dir = tempfile.mkdtemp(prefix="fable_fsm_test_")
|
|
41
|
+
self._orig_cortex_dir = GLOBAL_PLASTICITY_ENGINE.cortex_dir
|
|
42
|
+
self._orig_matrix_path = GLOBAL_PLASTICITY_ENGINE.matrix_path
|
|
43
|
+
temp_cortex = Path(self.test_dir) / "cortex"
|
|
44
|
+
temp_cortex.mkdir(parents=True, exist_ok=True)
|
|
45
|
+
if self._orig_cortex_dir.exists():
|
|
46
|
+
shutil.copytree(self._orig_cortex_dir, temp_cortex, dirs_exist_ok=True)
|
|
47
|
+
GLOBAL_PLASTICITY_ENGINE.cortex_dir = temp_cortex
|
|
48
|
+
GLOBAL_PLASTICITY_ENGINE.matrix_path = temp_cortex / "synaptic_matrix.json"
|
|
49
|
+
from fable_engine.session import get_red_team_swarm
|
|
50
|
+
get_red_team_swarm().plasticity_engine = GLOBAL_PLASTICITY_ENGINE
|
|
51
|
+
for name in ("test_illegal_fsm", "test_red_team_gate", "test_ping_pong_loop", "test_cortical_evo"):
|
|
52
|
+
p = SESSIONS_DIR / f"{name}.json"
|
|
53
|
+
if p.exists():
|
|
54
|
+
try:
|
|
55
|
+
p.unlink()
|
|
56
|
+
except Exception:
|
|
57
|
+
pass
|
|
58
|
+
|
|
59
|
+
def tearDown(self):
|
|
60
|
+
ACTIVE_SESSIONS.clear()
|
|
61
|
+
GLOBAL_PLASTICITY_ENGINE.cortex_dir = self._orig_cortex_dir
|
|
62
|
+
GLOBAL_PLASTICITY_ENGINE.matrix_path = self._orig_matrix_path
|
|
63
|
+
from fable_engine.session import get_red_team_swarm
|
|
64
|
+
get_red_team_swarm().plasticity_engine = GLOBAL_PLASTICITY_ENGINE
|
|
65
|
+
for name in ("test_illegal_fsm", "test_red_team_gate", "test_ping_pong_loop", "test_cortical_evo"):
|
|
66
|
+
p = SESSIONS_DIR / f"{name}.json"
|
|
67
|
+
if p.exists():
|
|
68
|
+
try:
|
|
69
|
+
p.unlink()
|
|
70
|
+
except Exception:
|
|
71
|
+
pass
|
|
72
|
+
if os.path.exists(self.test_dir):
|
|
73
|
+
shutil.rmtree(self.test_dir, ignore_errors=True)
|
|
74
|
+
|
|
75
|
+
def test_fsm_illegal_transitions_blocked(self):
|
|
76
|
+
"""1. Verify FSM tracks state and strictly blocks illegal leaps (e.g. INIT -> SEALED/EVOLVED)."""
|
|
77
|
+
session_name = "test_illegal_fsm"
|
|
78
|
+
session = FableSession(session_name=session_name, objective="Test FSM invariants", time_budget_minutes=10.0)
|
|
79
|
+
ACTIVE_SESSIONS[session_name] = session
|
|
80
|
+
|
|
81
|
+
# Verify initial state
|
|
82
|
+
self.assertEqual(session.current_state, SessionState.INIT)
|
|
83
|
+
self.assertEqual(session.iteration_count, 0)
|
|
84
|
+
self.assertEqual(session.active_breakages, [])
|
|
85
|
+
self.assertEqual(session.remediation_history, [])
|
|
86
|
+
|
|
87
|
+
# Direct leap from INIT -> SEALED must raise ValueError
|
|
88
|
+
with self.assertRaises(ValueError) as ctx:
|
|
89
|
+
session.transition_to(SessionState.SEALED, "Attempt premature seal")
|
|
90
|
+
self.assertIn("Illegal state transition from INIT to SEALED", str(ctx.exception))
|
|
91
|
+
|
|
92
|
+
# Direct leap from INIT -> EVOLVED must raise ValueError
|
|
93
|
+
with self.assertRaises(ValueError) as ctx:
|
|
94
|
+
session.transition_to(SessionState.EVOLVED, "Attempt premature evolve")
|
|
95
|
+
self.assertIn("Illegal state transition from INIT to EVOLVED", str(ctx.exception))
|
|
96
|
+
|
|
97
|
+
# Direct leap from INIT -> RED_TEAM_GATE must raise ValueError
|
|
98
|
+
with self.assertRaises(ValueError) as ctx:
|
|
99
|
+
session.transition_to(SessionState.RED_TEAM_GATE, "Skip implementation")
|
|
100
|
+
self.assertIn("Illegal state transition", str(ctx.exception))
|
|
101
|
+
|
|
102
|
+
# Move to DEEPTHINK_TIMELOCK via set_timer
|
|
103
|
+
session.set_timer(10.0)
|
|
104
|
+
self.assertEqual(session.current_state, SessionState.DEEPTHINK_TIMELOCK)
|
|
105
|
+
|
|
106
|
+
# Direct leap from DEEPTHINK_TIMELOCK -> SEALED or EVOLVED must raise ValueError
|
|
107
|
+
with self.assertRaises(ValueError):
|
|
108
|
+
session.transition_to(SessionState.SEALED, "Premature seal from timelock")
|
|
109
|
+
with self.assertRaises(ValueError):
|
|
110
|
+
session.transition_to(SessionState.EVOLVED, "Premature evolve from timelock")
|
|
111
|
+
|
|
112
|
+
# Attempting jump from Phase 1 to Phase 4 is blocked by phase order
|
|
113
|
+
adv_res_jump = handle_fable_session({
|
|
114
|
+
"action": "advance_phase",
|
|
115
|
+
"session_name": session_name,
|
|
116
|
+
"next_phase": "Phase 4: Subagent Fleet Delegation",
|
|
117
|
+
"phase_summary": "Attempting jump while in Phase 1"
|
|
118
|
+
})
|
|
119
|
+
self.assertIn("Invalid phase transition", adv_res_jump)
|
|
120
|
+
|
|
121
|
+
# In Phase 3, advancing to Phase 4 while execution is locked must fail with execution locked
|
|
122
|
+
session.active_phase = PHASES[2] # Phase 3
|
|
123
|
+
adv_res = handle_fable_session({
|
|
124
|
+
"action": "advance_phase",
|
|
125
|
+
"session_name": session_name,
|
|
126
|
+
"next_phase": "Phase 4: Subagent Fleet Delegation",
|
|
127
|
+
"phase_summary": "Attempting jump while locked"
|
|
128
|
+
})
|
|
129
|
+
self.assertTrue("Execution must be unlocked" in adv_res or "Execution is still locked" in adv_res)
|
|
130
|
+
|
|
131
|
+
# advance_phase to Phase 5 or Phase 6 while in DEEPTHINK_TIMELOCK must fail
|
|
132
|
+
adv_res_p5 = handle_fable_session({
|
|
133
|
+
"action": "advance_phase",
|
|
134
|
+
"session_name": session_name,
|
|
135
|
+
"next_phase": "Phase 5: Multi-Tier Verification & Gatekeeping",
|
|
136
|
+
"phase_summary": "Skipping implementation"
|
|
137
|
+
})
|
|
138
|
+
self.assertTrue(
|
|
139
|
+
"Cannot advance to Phase 5" in adv_res_p5 or "Execution is still locked" in adv_res_p5 or "Error" in adv_res_p5
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
# Transition to IMPLEMENTATION
|
|
143
|
+
session.execution_locked = False
|
|
144
|
+
session.can_execute_code = True
|
|
145
|
+
session.transition_to(SessionState.IMPLEMENTATION, "Execution unlocked")
|
|
146
|
+
self.assertEqual(session.current_state, SessionState.IMPLEMENTATION)
|
|
147
|
+
|
|
148
|
+
# From IMPLEMENTATION, jumping directly to SEALED or EVOLVED is illegal
|
|
149
|
+
with self.assertRaises(ValueError):
|
|
150
|
+
session.transition_to(SessionState.SEALED)
|
|
151
|
+
with self.assertRaises(ValueError):
|
|
152
|
+
session.transition_to(SessionState.EVOLVED)
|
|
153
|
+
|
|
154
|
+
def test_red_team_gating_and_rejection_order(self):
|
|
155
|
+
"""2. Verify Red Team Swarm attack results strictly gate the session, returning TASK REJECTED."""
|
|
156
|
+
session_name = "test_red_team_gate"
|
|
157
|
+
session = FableSession(session_name=session_name, objective="Red team gate test", time_budget_minutes=5.0)
|
|
158
|
+
session.set_timer(5.0)
|
|
159
|
+
session.execution_locked = False
|
|
160
|
+
session.can_execute_code = True
|
|
161
|
+
session.transition_to(SessionState.IMPLEMENTATION, "Implementation started")
|
|
162
|
+
session.transition_to(SessionState.RED_TEAM_GATE, "Code submitted for swarm audit")
|
|
163
|
+
session.save()
|
|
164
|
+
ACTIVE_SESSIONS[session_name] = session
|
|
165
|
+
|
|
166
|
+
# Record breakage report with 2 broken scenarios
|
|
167
|
+
broken_scenarios = [
|
|
168
|
+
{
|
|
169
|
+
"scenario_id": "auth_byzantine_null_byte",
|
|
170
|
+
"vector": "byzantine_payload",
|
|
171
|
+
"hypothesis": "Null byte injection causes unhandled C-string truncation",
|
|
172
|
+
"broken": True,
|
|
173
|
+
"error_message": "ValueError: embedded null byte",
|
|
174
|
+
"reproduction_code": "authenticate('admin\\x00token')",
|
|
175
|
+
"severity": "CRITICAL",
|
|
176
|
+
},
|
|
177
|
+
{
|
|
178
|
+
"scenario_id": "auth_concurrency_toctou",
|
|
179
|
+
"vector": "concurrency_race",
|
|
180
|
+
"hypothesis": "Concurrent token revocation allows TOCTOU reuse",
|
|
181
|
+
"broken": True,
|
|
182
|
+
"error_message": "AssertionError: token was used after revocation",
|
|
183
|
+
"reproduction_code": "concurrent_revoke_and_use(token)",
|
|
184
|
+
"severity": "HIGH",
|
|
185
|
+
},
|
|
186
|
+
]
|
|
187
|
+
|
|
188
|
+
resp = handle_fable_session({
|
|
189
|
+
"action": "record_breakage_report",
|
|
190
|
+
"session_name": session_name,
|
|
191
|
+
"broken_scenarios": broken_scenarios,
|
|
192
|
+
})
|
|
193
|
+
session = get_or_load_session(session_name)
|
|
194
|
+
|
|
195
|
+
# Verify exact machine order string
|
|
196
|
+
expected_machine_order = "TASK REJECTED: 2 breakages detected. Deploy subagent to fix findings."
|
|
197
|
+
self.assertIn(expected_machine_order, resp)
|
|
198
|
+
|
|
199
|
+
# Verify session state became REMEDIATION_REQUIRED
|
|
200
|
+
self.assertEqual(session.current_state, SessionState.REMEDIATION_REQUIRED)
|
|
201
|
+
self.assertEqual(len(session.active_breakages), 2)
|
|
202
|
+
self.assertEqual(session.iteration_count, 1)
|
|
203
|
+
|
|
204
|
+
# Verify advancing phase is blocked while active breakages exist
|
|
205
|
+
adv_resp = handle_fable_session({
|
|
206
|
+
"action": "advance_phase",
|
|
207
|
+
"session_name": session_name,
|
|
208
|
+
"next_phase": "Phase 5: Multi-Tier Verification & Gatekeeping",
|
|
209
|
+
"phase_summary": "Attempting advance with active breakages"
|
|
210
|
+
})
|
|
211
|
+
self.assertTrue("active breakages" in adv_resp or "REMEDIATION_REQUIRED" in adv_resp or "Error" in adv_resp)
|
|
212
|
+
|
|
213
|
+
def test_closed_loop_ping_pong_remediation_and_sealing(self):
|
|
214
|
+
"""3. Verify ping-pong while-loop remediation and final SEALED state transition."""
|
|
215
|
+
session_name = "test_ping_pong_loop"
|
|
216
|
+
session = FableSession(session_name=session_name, objective="Ping-pong hardening test", time_budget_minutes=5.0)
|
|
217
|
+
session.set_timer(5.0)
|
|
218
|
+
session.log_epistemic_item("PROVEN", "Ping-pong evidence one", "README.md:L1")
|
|
219
|
+
session.log_epistemic_item("PROVEN", "Ping-pong evidence two", "README.md:L5")
|
|
220
|
+
session.proof_receipts.append({"receipt_id": "ping-pong-receipt", "verified": True})
|
|
221
|
+
session.set_goal_rubric(
|
|
222
|
+
"Ping-pong rubric",
|
|
223
|
+
[{"pointer_id": "P1", "satisfied": True, "score": 1.0, "verifier_command": "unittest", "evidence_receipt_id": "ping-pong-receipt"}],
|
|
224
|
+
)
|
|
225
|
+
session.log_refinement_cycle("security", "remediation", "breakage", "harden implementation")
|
|
226
|
+
session.execution_locked = False
|
|
227
|
+
session.can_execute_code = True
|
|
228
|
+
session.transition_to(SessionState.IMPLEMENTATION, "Implemented")
|
|
229
|
+
session.track_file_change("sample.py", "modified", "remediated implementation")
|
|
230
|
+
session.transition_to(SessionState.RED_TEAM_GATE, "Code written and ready for audit")
|
|
231
|
+
session.save()
|
|
232
|
+
ACTIVE_SESSIONS[session_name] = session
|
|
233
|
+
|
|
234
|
+
# 1. Initial breakage recorded
|
|
235
|
+
initial_breakages = [
|
|
236
|
+
{
|
|
237
|
+
"scenario_id": "sec_01",
|
|
238
|
+
"vector": "byzantine_payload",
|
|
239
|
+
"hypothesis": "Large payload crashes memory",
|
|
240
|
+
"broken": True,
|
|
241
|
+
"error_message": "MemoryError: payload too large",
|
|
242
|
+
"reproduction_code": "process('A' * 1000000)",
|
|
243
|
+
"severity": "HIGH",
|
|
244
|
+
}
|
|
245
|
+
]
|
|
246
|
+
resp1 = handle_fable_session({
|
|
247
|
+
"action": "record_breakage_report",
|
|
248
|
+
"session_name": session_name,
|
|
249
|
+
"broken_scenarios": initial_breakages,
|
|
250
|
+
})
|
|
251
|
+
session = get_or_load_session(session_name)
|
|
252
|
+
self.assertIn("TASK REJECTED: 1 breakages detected. Deploy subagent to fix findings.", resp1)
|
|
253
|
+
self.assertEqual(session.current_state, SessionState.REMEDIATION_REQUIRED)
|
|
254
|
+
|
|
255
|
+
# 2. Subagent submits remediated code, but it still breaks
|
|
256
|
+
broken_report = {
|
|
257
|
+
"report_id": "rep_prior_01",
|
|
258
|
+
"target_name": "process",
|
|
259
|
+
"broken_count": 1,
|
|
260
|
+
"findings": [
|
|
261
|
+
{
|
|
262
|
+
"scenario_id": "sec_01",
|
|
263
|
+
"vector": "byzantine_payload",
|
|
264
|
+
"hypothesis": "Large payload crashes memory",
|
|
265
|
+
"broken": True,
|
|
266
|
+
"error_message": "MemoryError",
|
|
267
|
+
}
|
|
268
|
+
]
|
|
269
|
+
}
|
|
270
|
+
def flawed_code(x=None):
|
|
271
|
+
raise MemoryError('Still leaking')
|
|
272
|
+
|
|
273
|
+
resp_flawed = handle_fable_session({
|
|
274
|
+
"action": "verify_red_team_remediation",
|
|
275
|
+
"session_name": session_name,
|
|
276
|
+
"remediated_code": flawed_code,
|
|
277
|
+
"prior_report": broken_report,
|
|
278
|
+
})
|
|
279
|
+
session = get_or_load_session(session_name)
|
|
280
|
+
self.assertIn("TASK REJECTED:", resp_flawed)
|
|
281
|
+
self.assertEqual(session.current_state, SessionState.REMEDIATION_REQUIRED)
|
|
282
|
+
|
|
283
|
+
# 3. Subagent submits fully fixed code that survives the prior breaking probe
|
|
284
|
+
def fixed_code(x=None):
|
|
285
|
+
return 'clean'
|
|
286
|
+
|
|
287
|
+
resp_fixed = handle_fable_session({
|
|
288
|
+
"action": "verify_red_team_remediation",
|
|
289
|
+
"session_name": session_name,
|
|
290
|
+
"remediated_code": fixed_code,
|
|
291
|
+
"prior_report": broken_report,
|
|
292
|
+
})
|
|
293
|
+
session = get_or_load_session(session_name)
|
|
294
|
+
|
|
295
|
+
# Verify exact machine completion string
|
|
296
|
+
expected_completion = "TASK COMPLETED: 0 breakages remain. Code sealed."
|
|
297
|
+
self.assertIn(expected_completion, resp_fixed)
|
|
298
|
+
self.assertEqual(session.current_state, SessionState.SEALED)
|
|
299
|
+
self.assertEqual(len(session.active_breakages), 0)
|
|
300
|
+
self.assertTrue(len(session.remediation_history) >= 1)
|
|
301
|
+
|
|
302
|
+
def test_post_success_cortical_evolution(self):
|
|
303
|
+
"""4. Verify post-success cortical evolution applies LTP weight updates, antibodies, and disk save."""
|
|
304
|
+
session_name = "test_cortical_evo"
|
|
305
|
+
session = FableSession(session_name=session_name, objective="Cortical evolution test", time_budget_minutes=5.0)
|
|
306
|
+
session.save()
|
|
307
|
+
ACTIVE_SESSIONS[session_name] = session
|
|
308
|
+
|
|
309
|
+
# Calling evolve_cortex on an unsealed session (INIT) must be rejected
|
|
310
|
+
unsealed_resp = handle_fable_session({
|
|
311
|
+
"action": "evolve_cortex",
|
|
312
|
+
"session_name": session_name,
|
|
313
|
+
"domain": "security",
|
|
314
|
+
"task_id": "test_task_evo",
|
|
315
|
+
})
|
|
316
|
+
self.assertIn("Error: evolve_cortex rejected: Session must be in SEALED or EVOLVED state", unsealed_resp)
|
|
317
|
+
|
|
318
|
+
# Advance session legitimately to SEALED state
|
|
319
|
+
session.set_timer(5.0)
|
|
320
|
+
session.log_epistemic_item("PROVEN", "Evolution evidence one", "README.md:L1")
|
|
321
|
+
session.log_epistemic_item("PROVEN", "Evolution evidence two", "README.md:L5")
|
|
322
|
+
session.proof_receipts.append({"receipt_id": "evolution-receipt", "verified": True})
|
|
323
|
+
session.set_goal_rubric(
|
|
324
|
+
"Evolution rubric",
|
|
325
|
+
[{"pointer_id": "P1", "satisfied": True, "score": 1.0, "evidence_receipt_id": "evolution-receipt"}],
|
|
326
|
+
)
|
|
327
|
+
session.log_refinement_cycle("security", "sealing", "direct transition", "validate clean report")
|
|
328
|
+
session.execution_locked = False
|
|
329
|
+
session.can_execute_code = True
|
|
330
|
+
session.transition_to(SessionState.IMPLEMENTATION, "Implemented")
|
|
331
|
+
session.track_file_change("sample.py", "modified", "Implementation ready for evolution")
|
|
332
|
+
session.transition_to(SessionState.RED_TEAM_GATE, "Code written and ready for audit")
|
|
333
|
+
session.transition_to(SessionState.ARBITRATION, "Arbitration")
|
|
334
|
+
with self.assertRaises(ValueError):
|
|
335
|
+
session.transition_to(SessionState.SEALED, "Direct seal without validated report")
|
|
336
|
+
clean_report = {
|
|
337
|
+
"report_id": "evolution-clean",
|
|
338
|
+
"target_name": "evolution",
|
|
339
|
+
"total_probes": len(RED_TEAM_ATTACK_VECTORS),
|
|
340
|
+
"broken_count": 0,
|
|
341
|
+
"passed": True,
|
|
342
|
+
"findings": [
|
|
343
|
+
{"scenario_id": vector, "vector": vector, "broken": False}
|
|
344
|
+
for vector in RED_TEAM_ATTACK_VECTORS
|
|
345
|
+
],
|
|
346
|
+
"report_origin": "red_team_swarm",
|
|
347
|
+
"reviewed_change_id": session.derive_reviewed_change_id(),
|
|
348
|
+
}
|
|
349
|
+
clean_report["attack_vector_results"] = session._attack_vector_results(clean_report)
|
|
350
|
+
clean_report["red_team_receipt"] = session.issue_red_team_receipt(
|
|
351
|
+
clean_report, clean_report["reviewed_change_id"]
|
|
352
|
+
)
|
|
353
|
+
session.record_breakage_report(clean_report)
|
|
354
|
+
session.save()
|
|
355
|
+
ACTIVE_SESSIONS[session_name] = session
|
|
356
|
+
|
|
357
|
+
# Record a past breakage that was neutralized
|
|
358
|
+
neutralized_breakages = [
|
|
359
|
+
{
|
|
360
|
+
"scenario_id": "sec_sqli_01",
|
|
361
|
+
"vector": "byzantine_payload",
|
|
362
|
+
"hypothesis": "Unsanitized input in query",
|
|
363
|
+
"error_message": "SQLSyntaxError",
|
|
364
|
+
"reproduction_code": "query(\"' OR 1=1 --\")",
|
|
365
|
+
"severity": "CRITICAL",
|
|
366
|
+
"prescribed_defense": "Enforce parameterized queries with atomic binding",
|
|
367
|
+
}
|
|
368
|
+
]
|
|
369
|
+
|
|
370
|
+
# Call evolve_cortex
|
|
371
|
+
evo_resp = handle_fable_session({
|
|
372
|
+
"action": "evolve_cortex",
|
|
373
|
+
"session_name": session_name,
|
|
374
|
+
"domain": "security",
|
|
375
|
+
"task_id": "task_sec_01",
|
|
376
|
+
"broken_scenarios": neutralized_breakages,
|
|
377
|
+
"co_activated_nodes": ["red_team_swarm", "mutation", "test_harness"],
|
|
378
|
+
})
|
|
379
|
+
session = get_or_load_session(session_name)
|
|
380
|
+
|
|
381
|
+
# Verify receipt contents
|
|
382
|
+
self.assertIn("### 🧬 Cortical Evolution Receipt: EVOLVED", evo_resp)
|
|
383
|
+
self.assertIn("LTP (Long-Term Potentiation)", evo_resp)
|
|
384
|
+
self.assertIn("+0.10 * A_domain * A_node (LTP)", evo_resp)
|
|
385
|
+
self.assertIn("ab_security_sec_sqli_01", evo_resp)
|
|
386
|
+
|
|
387
|
+
# Verify session state transitioned to EVOLVED
|
|
388
|
+
self.assertEqual(session.current_state, SessionState.EVOLVED)
|
|
389
|
+
|
|
390
|
+
# Verify disk persistence in cortex/<domain>.md
|
|
391
|
+
lobe = GLOBAL_PLASTICITY_ENGINE.activate_lobe(domain="security")
|
|
392
|
+
self.assertIsNotNone(lobe)
|
|
393
|
+
lobe_path = GLOBAL_PLASTICITY_ENGINE.cortex_dir / "security.md"
|
|
394
|
+
self.assertTrue(lobe_path.exists())
|
|
395
|
+
|
|
396
|
+
lobe_content = lobe_path.read_text(encoding="utf-8")
|
|
397
|
+
self.assertIn("ab_security_sec_sqli_01", lobe_content)
|
|
398
|
+
self.assertIn("Unsanitized input in query", lobe_content)
|
|
399
|
+
|
|
400
|
+
# Verify synaptic weights reflect potentiated LTP updates
|
|
401
|
+
self.assertIn("red_team_swarm", lobe.synaptic_weights)
|
|
402
|
+
self.assertGreaterEqual(lobe.synaptic_weights["red_team_swarm"], 0.05)
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
if __name__ == "__main__":
|
|
406
|
+
unittest.main()
|