fable-engine 1.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fable_compressor.py +356 -0
- fable_engine/__init__.py +1 -0
- fable_engine/actions/__init__.py +291 -0
- fable_engine/actions/cas.py +182 -0
- fable_engine/actions/deliberation.py +523 -0
- fable_engine/actions/fleet.py +807 -0
- fable_engine/actions/lifecycle.py +298 -0
- fable_engine/actions/scrapers.py +116 -0
- fable_engine/actions/system3.py +815 -0
- fable_engine/browser.py +824 -0
- fable_engine/cas.py +974 -0
- fable_engine/fable_session.json +510 -0
- fable_engine/guards.py +283 -0
- fable_engine/schema.py +714 -0
- fable_engine/scrapers/__init__.py +32 -0
- fable_engine/scrapers/arxiv.py +115 -0
- fable_engine/scrapers/base.py +386 -0
- fable_engine/scrapers/github.py +129 -0
- fable_engine/scrapers/reddit.py +154 -0
- fable_engine/scrapers/web.py +120 -0
- fable_engine/scrapers/x.py +125 -0
- fable_engine/scrapers/youtube.py +132 -0
- fable_engine/server.py +414 -0
- fable_engine/session.py +1819 -0
- fable_engine/test_server.py +1362 -0
- fable_engine/updater.py +541 -0
- fable_engine-1.3.1.dist-info/LICENSE +22 -0
- fable_engine-1.3.1.dist-info/METADATA +173 -0
- fable_engine-1.3.1.dist-info/RECORD +104 -0
- fable_engine-1.3.1.dist-info/WHEEL +5 -0
- fable_engine-1.3.1.dist-info/entry_points.txt +5 -0
- fable_engine-1.3.1.dist-info/top_level.txt +6 -0
- fable_mode/__init__.py +3 -0
- fable_mode/__main__.py +4 -0
- fable_mode/adapters.py +1014 -0
- fable_mode/installer.py +553 -0
- fable_mode/launcher.py +437 -0
- fable_mode/manifest.py +142 -0
- fable_mode/resources.json +114 -0
- fable_mode/safety.py +103 -0
- fable_mode_entry.py +10 -0
- fable_v2/__init__.py +146 -0
- fable_v2/adapters.py +151 -0
- fable_v2/coder_fleet/__init__.py +100 -0
- fable_v2/coder_fleet/ast_tools.py +158 -0
- fable_v2/coder_fleet/compute.py +199 -0
- fable_v2/coder_fleet/design_engine.py +1316 -0
- fable_v2/coder_fleet/diagnostics.py +293 -0
- fable_v2/coder_fleet/fleet_dispatcher.py +214 -0
- fable_v2/coder_fleet/mock_auditor.py +306 -0
- fable_v2/coder_fleet/mutation.py +216 -0
- fable_v2/coder_fleet/property_oracle.py +260 -0
- fable_v2/coder_fleet/receipt_attestor.py +122 -0
- fable_v2/coder_fleet/red_team_swarm.py +908 -0
- fable_v2/coder_fleet/test_harness.py +198 -0
- fable_v2/coder_fleet/vector_engine.py +1287 -0
- fable_v2/coder_fleet/visual.py +357 -0
- fable_v2/coder_fleet/workspace.py +153 -0
- fable_v2/cortical/__init__.py +20 -0
- fable_v2/cortical/plasticity_engine.py +992 -0
- fable_v2/execution_broker.py +811 -0
- fable_v2/proof_engine.py +1141 -0
- fable_v2/protocol.py +485 -0
- fable_v2/runtime.py +1010 -0
- fable_v2/system3/__init__.py +204 -0
- fable_v2/system3/causal.py +558 -0
- fable_v2/system3/dialectical.py +577 -0
- fable_v2/system3/evolution.py +503 -0
- fable_v2/system3/executive.py +338 -0
- fable_v2/system3/free_energy.py +479 -0
- fable_v2/system3/hyperbolic.py +555 -0
- fable_v2/system3/induction.py +336 -0
- fable_v2/system3/kripke.py +548 -0
- fable_v2/system3/oracle.py +745 -0
- fable_v2/verifiers.py +72 -0
- tests/__init__.py +1 -0
- tests/test_anti_loop_circuit_breaker.py +64 -0
- tests/test_auto_updater.py +407 -0
- tests/test_coder_fleet.py +535 -0
- tests/test_delegation_compiler.py +54 -0
- tests/test_descriptor_boundaries.py +126 -0
- tests/test_design_engine.py +603 -0
- tests/test_epistemic_evidence_validator.py +66 -0
- tests/test_execution_broker.py +233 -0
- tests/test_fable_v2.py +406 -0
- tests/test_fleet_transitions.py +116 -0
- tests/test_fsm_redteam_evolution.py +406 -0
- tests/test_goal_rubric_and_pipeline.py +367 -0
- tests/test_hebbian_plasticity.py +585 -0
- tests/test_packaging_runtime.py +194 -0
- tests/test_proof_engine.py +259 -0
- tests/test_red_team_swarm.py +645 -0
- tests/test_redteam_remediation.py +169 -0
- tests/test_registration_transaction.py +375 -0
- tests/test_requested_regressions.py +467 -0
- tests/test_scrapers.py +370 -0
- tests/test_server_actions.py +93 -0
- tests/test_server_frontier_actions.py +269 -0
- tests/test_server_protocol.py +88 -0
- tests/test_stealth_browser.py +970 -0
- tests/test_system3.py +381 -0
- tests/test_system3_deep_integration.py +385 -0
- tests/test_system3_frontier.py +436 -0
- tests/test_vector_engine.py +608 -0
|
@@ -0,0 +1,367 @@
|
|
|
1
|
+
import unittest
|
|
2
|
+
import json
|
|
3
|
+
import time
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
import sys
|
|
6
|
+
|
|
7
|
+
BASE_DIR = Path(__file__).resolve().parent.parent
|
|
8
|
+
for p in [str(BASE_DIR), str(BASE_DIR / "fable_engine"), str(Path(__file__).resolve().parent)]:
|
|
9
|
+
if p not in sys.path:
|
|
10
|
+
sys.path.insert(0, p)
|
|
11
|
+
|
|
12
|
+
from server import (
|
|
13
|
+
handle_fable_session,
|
|
14
|
+
ACTIVE_SESSIONS,
|
|
15
|
+
SESSIONS_DIR,
|
|
16
|
+
FableSession,
|
|
17
|
+
get_or_load_session,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class TestGoalRubricAndPipeline(unittest.TestCase):
|
|
22
|
+
def setUp(self):
|
|
23
|
+
self.session_name = f"test_rubric_pipe_{int(time.time() * 1000)}"
|
|
24
|
+
|
|
25
|
+
def tearDown(self):
|
|
26
|
+
if self.session_name in ACTIVE_SESSIONS:
|
|
27
|
+
del ACTIVE_SESSIONS[self.session_name]
|
|
28
|
+
session_file = SESSIONS_DIR / f"{self.session_name}.json"
|
|
29
|
+
if session_file.exists():
|
|
30
|
+
try:
|
|
31
|
+
session_file.unlink()
|
|
32
|
+
except Exception:
|
|
33
|
+
pass
|
|
34
|
+
|
|
35
|
+
def _create_test_session(self, time_budget=10.0, objective="Test Rubrics and Pipelines"):
|
|
36
|
+
return handle_fable_session({
|
|
37
|
+
"action": "create_session",
|
|
38
|
+
"session_name": self.session_name,
|
|
39
|
+
"objective": objective,
|
|
40
|
+
"time_budget_minutes": time_budget
|
|
41
|
+
})
|
|
42
|
+
|
|
43
|
+
def test_set_goal_rubric_defaults_and_custom_weights(self):
|
|
44
|
+
self._create_test_session()
|
|
45
|
+
|
|
46
|
+
# 1. Register rubric with default target_score and auto-generated rubric_id
|
|
47
|
+
criteria = [
|
|
48
|
+
{
|
|
49
|
+
"pointer_id": "PTR-TEST-01",
|
|
50
|
+
"description": "Unit tests pass with 100% coverage",
|
|
51
|
+
"weight": 3.0,
|
|
52
|
+
"verifier_command": "pytest tests/test_core.py"
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
"pointer_id": "PTR-TEST-02",
|
|
56
|
+
"description": "Clean linter with zero errors",
|
|
57
|
+
"weight": 1.0,
|
|
58
|
+
"verifier_command": "ruff check ."
|
|
59
|
+
},
|
|
60
|
+
"PTR-TEST-03: Memory consumption below 50MB"
|
|
61
|
+
]
|
|
62
|
+
|
|
63
|
+
res = handle_fable_session({
|
|
64
|
+
"action": "set_goal_rubric",
|
|
65
|
+
"session_name": self.session_name,
|
|
66
|
+
"task_objective": "Deliver high-reliability component",
|
|
67
|
+
"criteria": criteria
|
|
68
|
+
})
|
|
69
|
+
|
|
70
|
+
self.assertIn("Goal Rubric Initialized", res)
|
|
71
|
+
self.assertIn("95.0%", res)
|
|
72
|
+
self.assertIn("PENDING", res)
|
|
73
|
+
self.assertIn("PTR-TEST-01", res)
|
|
74
|
+
self.assertIn("PTR-TEST-02", res)
|
|
75
|
+
self.assertIn("PTR-TEST-03", res)
|
|
76
|
+
|
|
77
|
+
session = ACTIVE_SESSIONS[self.session_name]
|
|
78
|
+
self.assertEqual(len(session.goal_rubrics), 1)
|
|
79
|
+
rubric = session.goal_rubrics[0]
|
|
80
|
+
self.assertTrue(rubric["rubric_id"].startswith(f"rubric_{self.session_name}_"))
|
|
81
|
+
self.assertEqual(rubric["target_score"], 0.95)
|
|
82
|
+
self.assertEqual(rubric["status"], "pending")
|
|
83
|
+
self.assertEqual(len(rubric["items"]), 3)
|
|
84
|
+
self.assertEqual(rubric["items"][0]["weight"], 3.0)
|
|
85
|
+
self.assertEqual(rubric["items"][1]["weight"], 1.0)
|
|
86
|
+
self.assertEqual(rubric["items"][2]["weight"], 1.0)
|
|
87
|
+
|
|
88
|
+
# 2. Register custom rubric with explicit rubric_id and custom target_score
|
|
89
|
+
custom_res = handle_fable_session({
|
|
90
|
+
"action": "set_goal_rubric",
|
|
91
|
+
"session_name": self.session_name,
|
|
92
|
+
"rubric_id": "custom_rubric_v2",
|
|
93
|
+
"target_score": 0.98,
|
|
94
|
+
"criteria": [
|
|
95
|
+
{"pointer_id": "P1", "description": "Spec 1", "weight": 2.0}
|
|
96
|
+
]
|
|
97
|
+
})
|
|
98
|
+
self.assertIn("custom_rubric_v2", custom_res)
|
|
99
|
+
self.assertIn("98.0%", custom_res)
|
|
100
|
+
self.assertEqual(len(session.goal_rubrics), 2)
|
|
101
|
+
self.assertEqual(session.goal_rubrics[1]["rubric_id"], "custom_rubric_v2")
|
|
102
|
+
self.assertEqual(session.goal_rubrics[1]["target_score"], 0.98)
|
|
103
|
+
|
|
104
|
+
def test_evaluate_goal_rubric_transitions_status(self):
|
|
105
|
+
self._create_test_session()
|
|
106
|
+
|
|
107
|
+
# Rubric with 2 items: weights 1.0 and 3.0 (total weight 4.0), target 0.95
|
|
108
|
+
handle_fable_session({
|
|
109
|
+
"action": "set_goal_rubric",
|
|
110
|
+
"session_name": self.session_name,
|
|
111
|
+
"rubric_id": "eval_rubric_test",
|
|
112
|
+
"target_score": 0.95,
|
|
113
|
+
"criteria": [
|
|
114
|
+
{"pointer_id": "PTR-CORE", "description": "Core algorithm", "weight": 1.0},
|
|
115
|
+
{"pointer_id": "PTR-SAFETY", "description": "Safety proofs", "weight": 3.0}
|
|
116
|
+
]
|
|
117
|
+
})
|
|
118
|
+
|
|
119
|
+
session = ACTIVE_SESSIONS[self.session_name]
|
|
120
|
+
session.proof_receipts.extend([
|
|
121
|
+
{"receipt_id": "rcpt_core_1", "verified": True},
|
|
122
|
+
{"receipt_id": "rcpt_safety_1", "verified": True},
|
|
123
|
+
])
|
|
124
|
+
|
|
125
|
+
# Step 1: Evaluate only PTR-CORE (score: (1.0*1.0 + 3.0*0.0)/4.0 = 0.25 -> 25%)
|
|
126
|
+
res1 = handle_fable_session({
|
|
127
|
+
"action": "evaluate_goal_rubric",
|
|
128
|
+
"session_name": self.session_name,
|
|
129
|
+
"rubric_id": "eval_rubric_test",
|
|
130
|
+
"evaluations": [
|
|
131
|
+
{"pointer_id": "PTR-CORE", "satisfied": True, "score": 1.0, "evidence_receipt_id": "rcpt_core_1"}
|
|
132
|
+
]
|
|
133
|
+
})
|
|
134
|
+
|
|
135
|
+
self.assertIn("Goal Rubric Evaluation", res1)
|
|
136
|
+
self.assertIn("25.00%", res1)
|
|
137
|
+
self.assertIn("IN_PROGRESS (< 95%)", res1)
|
|
138
|
+
rubric = session.get_goal_rubric("eval_rubric_test")
|
|
139
|
+
self.assertEqual(rubric["current_score"], 0.25)
|
|
140
|
+
self.assertEqual(rubric["status"], "in_progress")
|
|
141
|
+
|
|
142
|
+
# Step 2: Evaluate PTR-SAFETY as satisfied (score: (1.0*1.0 + 3.0*1.0)/4.0 = 1.0 -> 100%)
|
|
143
|
+
res2 = handle_fable_session({
|
|
144
|
+
"action": "evaluate_goal_rubric",
|
|
145
|
+
"session_name": self.session_name,
|
|
146
|
+
"rubric_id": "eval_rubric_test",
|
|
147
|
+
"evaluations": [
|
|
148
|
+
{"pointer_id": "PTR-SAFETY", "satisfied": True, "score": 1.0, "evidence_receipt_id": "rcpt_safety_1"}
|
|
149
|
+
]
|
|
150
|
+
})
|
|
151
|
+
|
|
152
|
+
self.assertIn("100.00%", res2)
|
|
153
|
+
self.assertIn("ACHIEVED (>= 95%)", res2)
|
|
154
|
+
rubric = session.get_goal_rubric("eval_rubric_test")
|
|
155
|
+
self.assertEqual(rubric["current_score"], 1.0)
|
|
156
|
+
self.assertEqual(rubric["status"], "achieved")
|
|
157
|
+
|
|
158
|
+
# Step 3: Partial score evaluation that meets target exactly:
|
|
159
|
+
# e.g., reset and set target to 0.90, weights 1.0 and 1.0, scores 0.90 and 0.90 -> 0.90 >= 0.90 -> achieved
|
|
160
|
+
handle_fable_session({
|
|
161
|
+
"action": "set_goal_rubric",
|
|
162
|
+
"session_name": self.session_name,
|
|
163
|
+
"rubric_id": "partial_rubric",
|
|
164
|
+
"target_score": 0.90,
|
|
165
|
+
"criteria": [
|
|
166
|
+
{"pointer_id": "A", "weight": 1.0},
|
|
167
|
+
{"pointer_id": "B", "weight": 1.0}
|
|
168
|
+
]
|
|
169
|
+
})
|
|
170
|
+
session.proof_receipts.extend([
|
|
171
|
+
{"receipt_id": "rcpt_partial_a", "verified": True},
|
|
172
|
+
{"receipt_id": "rcpt_partial_b", "verified": True},
|
|
173
|
+
])
|
|
174
|
+
handle_fable_session({
|
|
175
|
+
"action": "evaluate_goal_rubric",
|
|
176
|
+
"session_name": self.session_name,
|
|
177
|
+
"rubric_id": "partial_rubric",
|
|
178
|
+
"evaluations": [
|
|
179
|
+
{"pointer_id": "A", "score": 0.90, "satisfied": True, "verifier_command": "verify-a", "evidence_receipt_id": "rcpt_partial_a"},
|
|
180
|
+
{"pointer_id": "B", "score": 0.90, "satisfied": True, "verifier_command": "verify-b", "evidence_receipt_id": "rcpt_partial_b"}
|
|
181
|
+
]
|
|
182
|
+
})
|
|
183
|
+
r_partial = session.get_goal_rubric("partial_rubric")
|
|
184
|
+
self.assertEqual(r_partial["current_score"], 0.90)
|
|
185
|
+
self.assertEqual(r_partial["status"], "achieved")
|
|
186
|
+
|
|
187
|
+
command_only = session.set_goal_rubric(
|
|
188
|
+
"Caller-controlled verifier",
|
|
189
|
+
[{"pointer_id": "CMD", "satisfied": True, "score": 1.0, "verifier_command": "true"}],
|
|
190
|
+
rubric_id="command_only",
|
|
191
|
+
)
|
|
192
|
+
self.assertNotEqual(command_only["status"], "achieved")
|
|
193
|
+
|
|
194
|
+
session.proof_receipts.append({"receipt_id": "failed_receipt", "verified": False})
|
|
195
|
+
failed_receipt = session.set_goal_rubric(
|
|
196
|
+
"Failed verifier execution",
|
|
197
|
+
[{"pointer_id": "FAIL", "satisfied": True, "score": 1.0, "evidence_receipt_id": "failed_receipt"}],
|
|
198
|
+
rubric_id="failed_receipt",
|
|
199
|
+
)
|
|
200
|
+
self.assertNotEqual(failed_receipt["status"], "achieved")
|
|
201
|
+
|
|
202
|
+
def test_get_goal_rubric(self):
|
|
203
|
+
self._create_test_session()
|
|
204
|
+
|
|
205
|
+
# 1. Query before any rubric is registered
|
|
206
|
+
empty_res = handle_fable_session({
|
|
207
|
+
"action": "get_goal_rubric",
|
|
208
|
+
"session_name": self.session_name
|
|
209
|
+
})
|
|
210
|
+
self.assertIn("No Goal Rubric Found", empty_res)
|
|
211
|
+
|
|
212
|
+
session = ACTIVE_SESSIONS[self.session_name]
|
|
213
|
+
self.assertIsNone(session.get_goal_rubric())
|
|
214
|
+
|
|
215
|
+
# 2. Register multiple rubrics
|
|
216
|
+
handle_fable_session({
|
|
217
|
+
"action": "set_goal_rubric",
|
|
218
|
+
"session_name": self.session_name,
|
|
219
|
+
"rubric_id": "rubric_alpha",
|
|
220
|
+
"criteria": ["Alpha Criterion 1"]
|
|
221
|
+
})
|
|
222
|
+
handle_fable_session({
|
|
223
|
+
"action": "set_goal_rubric",
|
|
224
|
+
"session_name": self.session_name,
|
|
225
|
+
"rubric_id": "rubric_beta",
|
|
226
|
+
"criteria": ["Beta Criterion 1"]
|
|
227
|
+
})
|
|
228
|
+
|
|
229
|
+
# 3. Query without rubric_id returns latest rubric (rubric_beta)
|
|
230
|
+
latest_res = handle_fable_session({
|
|
231
|
+
"action": "get_goal_rubric",
|
|
232
|
+
"session_name": self.session_name
|
|
233
|
+
})
|
|
234
|
+
self.assertIn("rubric_beta", latest_res)
|
|
235
|
+
self.assertEqual(session.get_goal_rubric()["rubric_id"], "rubric_beta")
|
|
236
|
+
|
|
237
|
+
# 4. Query specific rubric by ID
|
|
238
|
+
alpha_res = handle_fable_session({
|
|
239
|
+
"action": "get_goal_rubric",
|
|
240
|
+
"session_name": self.session_name,
|
|
241
|
+
"rubric_id": "rubric_alpha"
|
|
242
|
+
})
|
|
243
|
+
self.assertIn("rubric_alpha", alpha_res)
|
|
244
|
+
self.assertEqual(session.get_goal_rubric("rubric_alpha")["rubric_id"], "rubric_alpha")
|
|
245
|
+
|
|
246
|
+
# 5. Query non-existent rubric id returns None / empty
|
|
247
|
+
self.assertIsNone(session.get_goal_rubric("non_existent_id"))
|
|
248
|
+
|
|
249
|
+
def test_register_automation_pipeline(self):
|
|
250
|
+
self._create_test_session()
|
|
251
|
+
|
|
252
|
+
# 1. Register closed-loop pipeline spec with generator_cmd, evaluator_cmd, max_iterations
|
|
253
|
+
res = handle_fable_session({
|
|
254
|
+
"action": "register_automation_pipeline",
|
|
255
|
+
"session_name": self.session_name,
|
|
256
|
+
"pipeline_name": "svg_vector_perceptual_loop",
|
|
257
|
+
"pipeline_type": "closed_loop",
|
|
258
|
+
"generator_cmd": "python scratch/synthesize_svg.py",
|
|
259
|
+
"evaluator_cmd": "python scratch/compare_ssim.py",
|
|
260
|
+
"max_iterations": 8,
|
|
261
|
+
"target_threshold": 0.96
|
|
262
|
+
})
|
|
263
|
+
|
|
264
|
+
self.assertIn("Autonomous Pipeline Registered", res)
|
|
265
|
+
self.assertIn("svg_vector_perceptual_loop", res)
|
|
266
|
+
self.assertIn("python scratch/synthesize_svg.py", res)
|
|
267
|
+
self.assertIn("python scratch/compare_ssim.py", res)
|
|
268
|
+
self.assertIn("96.0%", res)
|
|
269
|
+
self.assertIn("8", res)
|
|
270
|
+
|
|
271
|
+
session = ACTIVE_SESSIONS[self.session_name]
|
|
272
|
+
self.assertEqual(len(session.automation_pipelines), 1)
|
|
273
|
+
pipe = session.automation_pipelines[0]
|
|
274
|
+
self.assertEqual(pipe["name"], "svg_vector_perceptual_loop")
|
|
275
|
+
self.assertEqual(pipe["pipeline_type"], "closed_loop")
|
|
276
|
+
self.assertEqual(pipe["generator_command"], "python scratch/synthesize_svg.py")
|
|
277
|
+
self.assertEqual(pipe["evaluator_command"], "python scratch/compare_ssim.py")
|
|
278
|
+
self.assertEqual(pipe["max_iterations"], 8)
|
|
279
|
+
self.assertEqual(pipe["target_threshold"], 0.96)
|
|
280
|
+
self.assertEqual(pipe["status"], "active")
|
|
281
|
+
|
|
282
|
+
# 2. Register pipeline using parameter aliases: name, generator_command, evaluator_command, target_score
|
|
283
|
+
res2 = handle_fable_session({
|
|
284
|
+
"action": "register_automation_pipeline",
|
|
285
|
+
"session_name": self.session_name,
|
|
286
|
+
"name": "fuzzing_pipeline",
|
|
287
|
+
"pipeline_type": "fuzzing",
|
|
288
|
+
"generator_command": "python scratch/gen_fuzz.py",
|
|
289
|
+
"evaluator_command": "python scratch/eval_fuzz.py",
|
|
290
|
+
"target_score": 0.99,
|
|
291
|
+
"max_iterations": 15
|
|
292
|
+
})
|
|
293
|
+
self.assertIn("fuzzing_pipeline", res2)
|
|
294
|
+
self.assertIn("99.0%", res2)
|
|
295
|
+
self.assertEqual(len(session.automation_pipelines), 2)
|
|
296
|
+
pipe2 = session.automation_pipelines[1]
|
|
297
|
+
self.assertEqual(pipe2["name"], "fuzzing_pipeline")
|
|
298
|
+
self.assertEqual(pipe2["pipeline_type"], "fuzzing")
|
|
299
|
+
self.assertEqual(pipe2["generator_command"], "python scratch/gen_fuzz.py")
|
|
300
|
+
self.assertEqual(pipe2["evaluator_command"], "python scratch/eval_fuzz.py")
|
|
301
|
+
self.assertEqual(pipe2["target_threshold"], 0.99)
|
|
302
|
+
self.assertEqual(pipe2["max_iterations"], 15)
|
|
303
|
+
|
|
304
|
+
def test_error_handling_and_validation(self):
|
|
305
|
+
self._create_test_session()
|
|
306
|
+
session = ACTIVE_SESSIONS[self.session_name]
|
|
307
|
+
|
|
308
|
+
# 1. Missing session_name in actions
|
|
309
|
+
for act in ["set_goal_rubric", "evaluate_goal_rubric", "get_goal_rubric", "register_automation_pipeline"]:
|
|
310
|
+
res = handle_fable_session({"action": act})
|
|
311
|
+
self.assertIn("Error:", res)
|
|
312
|
+
self.assertIn("'session_name' is required", res)
|
|
313
|
+
|
|
314
|
+
# 2. Non-existent session
|
|
315
|
+
res_nosess = handle_fable_session({
|
|
316
|
+
"action": "set_goal_rubric",
|
|
317
|
+
"session_name": "ghost_session_xyz",
|
|
318
|
+
"criteria": ["Criterion 1"]
|
|
319
|
+
})
|
|
320
|
+
self.assertIn("Error:", res_nosess)
|
|
321
|
+
self.assertIn("does not exist", res_nosess)
|
|
322
|
+
|
|
323
|
+
# 3. Missing or empty criteria in set_goal_rubric
|
|
324
|
+
res_nocrit = handle_fable_session({
|
|
325
|
+
"action": "set_goal_rubric",
|
|
326
|
+
"session_name": self.session_name
|
|
327
|
+
})
|
|
328
|
+
self.assertIn("Error:", res_nocrit)
|
|
329
|
+
self.assertIn("'criteria'", res_nocrit)
|
|
330
|
+
|
|
331
|
+
# 4. Direct session method validations:
|
|
332
|
+
# 4a. Invalid criteria type / empty list raises ValueError
|
|
333
|
+
with self.assertRaises(ValueError):
|
|
334
|
+
session.set_goal_rubric(task_objective="test", criteria=[])
|
|
335
|
+
|
|
336
|
+
# 4b. Invalid target_score outside [0.0, 1.0] raises ValueError
|
|
337
|
+
with self.assertRaises(ValueError):
|
|
338
|
+
session.set_goal_rubric(task_objective="test", criteria=["Valid"], target_score=1.5)
|
|
339
|
+
|
|
340
|
+
with self.assertRaises(ValueError):
|
|
341
|
+
session.set_goal_rubric(task_objective="test", criteria=["Valid"], target_score=-0.1)
|
|
342
|
+
|
|
343
|
+
# 4c. evaluate_goal_rubric before any rubric is registered raises ValueError
|
|
344
|
+
fresh_sess = FableSession(session_name="fresh_test_sess", objective="testing", time_budget_minutes=10.0)
|
|
345
|
+
with self.assertRaises(ValueError):
|
|
346
|
+
fresh_sess.evaluate_goal_rubric()
|
|
347
|
+
|
|
348
|
+
# 4d. evaluate_goal_rubric with non-existent rubric_id raises ValueError
|
|
349
|
+
session.set_goal_rubric(task_objective="test", criteria=["Valid criterion"])
|
|
350
|
+
with self.assertRaises(ValueError):
|
|
351
|
+
session.evaluate_goal_rubric(rubric_id="non_existent_rubric_999")
|
|
352
|
+
|
|
353
|
+
# 4e. register_automation_pipeline with empty name raises ValueError
|
|
354
|
+
with self.assertRaises(ValueError):
|
|
355
|
+
session.register_automation_pipeline(name="")
|
|
356
|
+
|
|
357
|
+
# 4f. register_automation_pipeline via handle_fable_session without name returns error
|
|
358
|
+
res_noname = handle_fable_session({
|
|
359
|
+
"action": "register_automation_pipeline",
|
|
360
|
+
"session_name": self.session_name
|
|
361
|
+
})
|
|
362
|
+
self.assertIn("Error:", res_noname)
|
|
363
|
+
self.assertIn("'name' is required", res_noname)
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
if __name__ == "__main__":
|
|
367
|
+
unittest.main()
|