eljay-ai 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +57 -0
- package/README.md +479 -0
- package/agent/__init__.py +10 -0
- package/agent/agent.py +269 -0
- package/agent/agents/__init__.py +34 -0
- package/agent/agents/registry.py +212 -0
- package/agent/builtin_tools.py +591 -0
- package/agent/chat.py +354 -0
- package/agent/config.py +67 -0
- package/agent/context.py +63 -0
- package/agent/edits.py +167 -0
- package/agent/gitaware.py +126 -0
- package/agent/hardware.py +331 -0
- package/agent/knowledge.py +133 -0
- package/agent/memory.py +98 -0
- package/agent/ollama_client.py +133 -0
- package/agent/permissions.py +93 -0
- package/agent/providers/__init__.py +51 -0
- package/agent/providers/base.py +78 -0
- package/agent/providers/image.py +110 -0
- package/agent/providers/ollama.py +111 -0
- package/agent/providers/video.py +96 -0
- package/agent/providers/web.py +155 -0
- package/agent/router.py +200 -0
- package/agent/rules.py +52 -0
- package/agent/runner.py +168 -0
- package/agent/skills.py +161 -0
- package/agent/tools.py +102 -0
- package/agent/verification.py +80 -0
- package/agent/workspace.py +487 -0
- package/bin/eljay +5 -0
- package/bin/eljay.cmd +4 -0
- package/bin/myagent +4 -0
- package/bin/myagent.cmd +4 -0
- package/eljay-ai-1.1.0.tgz +0 -0
- package/eljay.js +56 -0
- package/eljay.py +150 -0
- package/install.ps1 +28 -0
- package/knowledge/reference/diffusers.md +32 -0
- package/knowledge/reference/video-providers.md +21 -0
- package/knowledge/setup/comfyui.md +35 -0
- package/knowledge/setup/image-providers.md +16 -0
- package/knowledge/test-category/test-entry.md +5 -0
- package/myagent.py +30 -0
- package/package.json +40 -0
- package/skills/coding/code-review.md +3 -0
- package/skills/coding/fix-attempt-protocol.md +9 -0
- package/skills/debugging/root-cause-analysis.md +9 -0
- package/skills/general/communication.md +7 -0
- package/skills/laravel/authentication.md +15 -0
- package/skills/mysql/performance.md +9 -0
- package/skills/php/standards.md +8 -0
- package/skills/react/component-best-practices.md +8 -0
- package/skills/research/source-tracking.md +7 -0
- package/skills/security/input-validation.md +9 -0
- package/skills/testing/pytest-best-practices.md +7 -0
- package/tests/run_all.py +34 -0
- package/tests/test_agent_core.py +385 -0
- package/tests/test_capabilities.py +169 -0
- package/tests/test_edits.py +117 -0
- package/tests/test_eljay.py +209 -0
- package/tests/test_runner.py +114 -0
- package/tests/test_universal.py +458 -0
- package/tests/test_workspace.py +221 -0
|
@@ -0,0 +1,385 @@
|
|
|
1
|
+
"""Deterministic tests for AgentCore — no model, no network.
|
|
2
|
+
|
|
3
|
+
Run from the project root:
|
|
4
|
+
python tests/test_agent_core.py
|
|
5
|
+
|
|
6
|
+
These verify the *architecture* (tool detection, execution, fallback parsing,
|
|
7
|
+
and the unknown-tool guard) independent of how a particular model behaves.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import os
|
|
11
|
+
import sys
|
|
12
|
+
import unittest
|
|
13
|
+
|
|
14
|
+
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
|
15
|
+
|
|
16
|
+
from agent.agent import AgentCore # noqa: E402
|
|
17
|
+
from agent.permissions import ( # noqa: E402
|
|
18
|
+
DECISION_ALWAYS,
|
|
19
|
+
DECISION_YES,
|
|
20
|
+
PermissionPolicy,
|
|
21
|
+
)
|
|
22
|
+
from agent.tools import ( # noqa: E402
|
|
23
|
+
RISK_ASK,
|
|
24
|
+
RISK_HIGH,
|
|
25
|
+
RISK_SAFE,
|
|
26
|
+
Tool,
|
|
27
|
+
ToolRegistry,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class FakeClient:
|
|
32
|
+
"""Returns queued responses in order, recording the messages it was sent."""
|
|
33
|
+
|
|
34
|
+
def __init__(self, responses):
|
|
35
|
+
self._responses = list(responses)
|
|
36
|
+
self.calls = []
|
|
37
|
+
|
|
38
|
+
def chat_full(self, model, messages, tools=None, options=None):
|
|
39
|
+
self.calls.append([dict(m) for m in messages])
|
|
40
|
+
if not self._responses:
|
|
41
|
+
raise AssertionError("FakeClient ran out of queued responses")
|
|
42
|
+
return self._responses.pop(0)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def msg(content="", tool_calls=None):
|
|
46
|
+
message = {"role": "assistant", "content": content}
|
|
47
|
+
if tool_calls is not None:
|
|
48
|
+
message["tool_calls"] = tool_calls
|
|
49
|
+
return {"message": message}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def make_registry():
|
|
53
|
+
registry = ToolRegistry()
|
|
54
|
+
registry.register(
|
|
55
|
+
Tool(
|
|
56
|
+
name="add",
|
|
57
|
+
description="Add two integers.",
|
|
58
|
+
parameters={
|
|
59
|
+
"type": "object",
|
|
60
|
+
"properties": {
|
|
61
|
+
"a": {"type": "integer"},
|
|
62
|
+
"b": {"type": "integer"},
|
|
63
|
+
},
|
|
64
|
+
"required": ["a", "b"],
|
|
65
|
+
},
|
|
66
|
+
func=lambda a, b: a + b,
|
|
67
|
+
risk=RISK_SAFE,
|
|
68
|
+
)
|
|
69
|
+
)
|
|
70
|
+
return registry
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def make_ask_registry(log, risk=RISK_ASK, name="danger"):
|
|
74
|
+
"""A registry with one non-safe tool that records the calls it receives."""
|
|
75
|
+
registry = ToolRegistry()
|
|
76
|
+
|
|
77
|
+
def danger(target=""):
|
|
78
|
+
log.append(target)
|
|
79
|
+
return "did the dangerous thing"
|
|
80
|
+
|
|
81
|
+
registry.register(
|
|
82
|
+
Tool(
|
|
83
|
+
name=name,
|
|
84
|
+
description="A non-safe action.",
|
|
85
|
+
parameters={
|
|
86
|
+
"type": "object",
|
|
87
|
+
"properties": {"target": {"type": "string"}},
|
|
88
|
+
"required": [],
|
|
89
|
+
},
|
|
90
|
+
func=danger,
|
|
91
|
+
risk=risk,
|
|
92
|
+
preview=lambda target="": f"Do danger on {target}",
|
|
93
|
+
)
|
|
94
|
+
)
|
|
95
|
+
return registry
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def make_core(responses, registry=None, **kwargs):
|
|
99
|
+
client = FakeClient(responses)
|
|
100
|
+
registry = registry or make_registry()
|
|
101
|
+
core = AgentCore(client, "fake-model", registry, "system", **kwargs)
|
|
102
|
+
return core, client
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class AgentCoreTests(unittest.TestCase):
|
|
106
|
+
def test_plain_answer_passes_through(self):
|
|
107
|
+
core, _ = make_core([msg("Hello there.")])
|
|
108
|
+
out = core.run([{"role": "user", "content": "hi"}])
|
|
109
|
+
self.assertEqual(out, "Hello there.")
|
|
110
|
+
|
|
111
|
+
def test_text_json_tool_call_is_executed(self):
|
|
112
|
+
core, client = make_core(
|
|
113
|
+
[
|
|
114
|
+
msg('{"name": "add", "arguments": {"a": 2, "b": 3}}'),
|
|
115
|
+
msg("The answer is 5."),
|
|
116
|
+
]
|
|
117
|
+
)
|
|
118
|
+
out = core.run([{"role": "user", "content": "add 2 and 3"}])
|
|
119
|
+
self.assertEqual(out, "The answer is 5.")
|
|
120
|
+
# The tool result must have been fed back as a user turn.
|
|
121
|
+
second_call = client.calls[1]
|
|
122
|
+
self.assertTrue(
|
|
123
|
+
any(
|
|
124
|
+
m["role"] == "user" and "TOOL RESULT (add)" in m["content"]
|
|
125
|
+
for m in second_call
|
|
126
|
+
)
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
def test_native_tool_call_is_executed(self):
|
|
130
|
+
core, _ = make_core(
|
|
131
|
+
[
|
|
132
|
+
msg(
|
|
133
|
+
"",
|
|
134
|
+
tool_calls=[
|
|
135
|
+
{"function": {"name": "add", "arguments": {"a": 10, "b": 5}}}
|
|
136
|
+
],
|
|
137
|
+
),
|
|
138
|
+
msg("15"),
|
|
139
|
+
]
|
|
140
|
+
)
|
|
141
|
+
out = core.run([{"role": "user", "content": "10+5"}])
|
|
142
|
+
self.assertEqual(out, "15")
|
|
143
|
+
|
|
144
|
+
def test_fenced_json_tool_call_is_parsed(self):
|
|
145
|
+
core, _ = make_core(
|
|
146
|
+
[msg('```json\n{"name": "add", "arguments": {"a": 1, "b": 1}}\n```'),
|
|
147
|
+
msg("2")],
|
|
148
|
+
)
|
|
149
|
+
out = core.run([{"role": "user", "content": "1+1"}])
|
|
150
|
+
self.assertEqual(out, "2")
|
|
151
|
+
|
|
152
|
+
def test_unknown_tool_is_not_leaked_to_user(self):
|
|
153
|
+
# Model first emits a bogus tool name, then answers normally after the nudge.
|
|
154
|
+
core, client = make_core(
|
|
155
|
+
[
|
|
156
|
+
msg('{"name": "await", "arguments": {}}'),
|
|
157
|
+
msg("await pauses an async function."),
|
|
158
|
+
]
|
|
159
|
+
)
|
|
160
|
+
out = core.run([{"role": "user", "content": "what does await do?"}])
|
|
161
|
+
self.assertEqual(out, "await pauses an async function.")
|
|
162
|
+
# A nudge must have been sent telling the model the tool is unknown.
|
|
163
|
+
self.assertTrue(
|
|
164
|
+
any(
|
|
165
|
+
"not a valid tool call" in m["content"]
|
|
166
|
+
for m in client.calls[1]
|
|
167
|
+
if m["role"] == "user"
|
|
168
|
+
)
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
def test_repeated_unknown_tool_gives_up_gracefully(self):
|
|
172
|
+
core, _ = make_core(
|
|
173
|
+
[
|
|
174
|
+
msg('{"name": "await", "arguments": {}}'),
|
|
175
|
+
msg('{"name": "await", "arguments": {}}'),
|
|
176
|
+
]
|
|
177
|
+
)
|
|
178
|
+
out = core.run([{"role": "user", "content": "what does await do?"}])
|
|
179
|
+
self.assertIn("unavailable tool", out)
|
|
180
|
+
|
|
181
|
+
def test_bad_arguments_produce_error_result_not_crash(self):
|
|
182
|
+
core, _ = make_core(
|
|
183
|
+
[
|
|
184
|
+
msg('{"name": "add", "arguments": {"a": 1}}'), # missing "b"
|
|
185
|
+
msg("Could not add."),
|
|
186
|
+
]
|
|
187
|
+
)
|
|
188
|
+
out = core.run([{"role": "user", "content": "add"}])
|
|
189
|
+
self.assertEqual(out, "Could not add.")
|
|
190
|
+
|
|
191
|
+
# -- permissions -------------------------------------------------------
|
|
192
|
+
|
|
193
|
+
def test_non_safe_tool_denied_when_no_confirmer(self):
|
|
194
|
+
log = []
|
|
195
|
+
core, client = make_core(
|
|
196
|
+
[msg('{"name": "danger", "arguments": {"target": "x"}}'), msg("ok")],
|
|
197
|
+
registry=make_ask_registry(log),
|
|
198
|
+
)
|
|
199
|
+
out = core.run([{"role": "user", "content": "do it"}])
|
|
200
|
+
self.assertEqual(log, []) # the tool never ran
|
|
201
|
+
self.assertEqual(out, "ok")
|
|
202
|
+
self.assertTrue(
|
|
203
|
+
any(
|
|
204
|
+
"PERMISSION DENIED" in m["content"]
|
|
205
|
+
for m in client.calls[1]
|
|
206
|
+
if m["role"] == "user"
|
|
207
|
+
)
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
def test_non_safe_tool_denied_by_user(self):
|
|
211
|
+
log = []
|
|
212
|
+
policy = PermissionPolicy(ask=lambda *_a: "no")
|
|
213
|
+
core, _ = make_core(
|
|
214
|
+
[msg('{"name": "danger", "arguments": {}}'), msg("fine")],
|
|
215
|
+
registry=make_ask_registry(log),
|
|
216
|
+
policy=policy,
|
|
217
|
+
)
|
|
218
|
+
core.run([{"role": "user", "content": "do it"}])
|
|
219
|
+
self.assertEqual(log, [])
|
|
220
|
+
|
|
221
|
+
def test_non_safe_tool_runs_when_approved(self):
|
|
222
|
+
log = []
|
|
223
|
+
previews = []
|
|
224
|
+
|
|
225
|
+
def ask(tool, preview, risk):
|
|
226
|
+
previews.append(preview)
|
|
227
|
+
return DECISION_YES
|
|
228
|
+
|
|
229
|
+
core, _ = make_core(
|
|
230
|
+
[msg('{"name": "danger", "arguments": {"target": "x"}}'), msg("done")],
|
|
231
|
+
registry=make_ask_registry(log),
|
|
232
|
+
policy=PermissionPolicy(ask=ask),
|
|
233
|
+
)
|
|
234
|
+
out = core.run([{"role": "user", "content": "do it"}])
|
|
235
|
+
self.assertEqual(log, ["x"]) # ran exactly once
|
|
236
|
+
self.assertTrue(any("Do danger on x" in p for p in previews))
|
|
237
|
+
self.assertEqual(out, "done")
|
|
238
|
+
|
|
239
|
+
def test_safe_tool_is_never_confirmed(self):
|
|
240
|
+
asked = []
|
|
241
|
+
policy = PermissionPolicy(
|
|
242
|
+
ask=lambda tool, preview, risk: asked.append(tool.name) or DECISION_YES
|
|
243
|
+
)
|
|
244
|
+
core, _ = make_core(
|
|
245
|
+
[msg('{"name": "add", "arguments": {"a": 1, "b": 1}}'), msg("2")],
|
|
246
|
+
policy=policy,
|
|
247
|
+
)
|
|
248
|
+
out = core.run([{"role": "user", "content": "1+1"}])
|
|
249
|
+
self.assertEqual(asked, []) # safe tool -> no prompt
|
|
250
|
+
self.assertEqual(out, "2")
|
|
251
|
+
|
|
252
|
+
# -- tiers: always / auto-approve / high --------------------------------
|
|
253
|
+
|
|
254
|
+
def test_always_decision_is_remembered_for_the_session(self):
|
|
255
|
+
log = []
|
|
256
|
+
asked = []
|
|
257
|
+
|
|
258
|
+
def ask(tool, preview, risk):
|
|
259
|
+
asked.append(tool.name)
|
|
260
|
+
return DECISION_ALWAYS
|
|
261
|
+
|
|
262
|
+
policy = PermissionPolicy(ask=ask)
|
|
263
|
+
core, _ = make_core(
|
|
264
|
+
[
|
|
265
|
+
msg('{"name": "danger", "arguments": {"target": "1"}}'),
|
|
266
|
+
msg("ok1"),
|
|
267
|
+
msg('{"name": "danger", "arguments": {"target": "2"}}'),
|
|
268
|
+
msg("ok2"),
|
|
269
|
+
],
|
|
270
|
+
registry=make_ask_registry(log),
|
|
271
|
+
policy=policy,
|
|
272
|
+
)
|
|
273
|
+
messages = [{"role": "user", "content": "go"}]
|
|
274
|
+
core.run(messages)
|
|
275
|
+
core.run(messages)
|
|
276
|
+
self.assertEqual(log, ["1", "2"]) # ran both times
|
|
277
|
+
self.assertEqual(asked, ["danger"]) # asked only once
|
|
278
|
+
|
|
279
|
+
def test_always_is_not_remembered_for_high_risk(self):
|
|
280
|
+
log = []
|
|
281
|
+
policy = PermissionPolicy(ask=lambda *_a: DECISION_ALWAYS)
|
|
282
|
+
core, _ = make_core(
|
|
283
|
+
[
|
|
284
|
+
msg('{"name": "danger", "arguments": {}}'),
|
|
285
|
+
msg("a"),
|
|
286
|
+
msg('{"name": "danger", "arguments": {}}'),
|
|
287
|
+
msg("b"),
|
|
288
|
+
],
|
|
289
|
+
registry=make_ask_registry(log, risk=RISK_HIGH),
|
|
290
|
+
policy=policy,
|
|
291
|
+
)
|
|
292
|
+
messages = [{"role": "user", "content": "go"}]
|
|
293
|
+
core.run(messages)
|
|
294
|
+
core.run(messages)
|
|
295
|
+
self.assertEqual(len(log), 2) # ran each time, permission re-asked
|
|
296
|
+
self.assertFalse(policy.is_always_allowed("danger"))
|
|
297
|
+
|
|
298
|
+
def test_auto_approve_never_bypasses_high_risk(self):
|
|
299
|
+
log = []
|
|
300
|
+
asked = []
|
|
301
|
+
policy = PermissionPolicy(
|
|
302
|
+
ask=lambda tool, preview, risk: asked.append(tool.name) or DECISION_YES,
|
|
303
|
+
auto_approve=True,
|
|
304
|
+
)
|
|
305
|
+
core, _ = make_core(
|
|
306
|
+
[msg('{"name": "danger", "arguments": {}}'), msg("x")],
|
|
307
|
+
registry=make_ask_registry(log, risk=RISK_HIGH),
|
|
308
|
+
policy=policy,
|
|
309
|
+
)
|
|
310
|
+
core.run([{"role": "user", "content": "go"}])
|
|
311
|
+
# Despite auto-approve, a high-risk action must still be confirmed.
|
|
312
|
+
self.assertEqual(asked, ["danger"])
|
|
313
|
+
|
|
314
|
+
def test_session_allowance_never_bypasses_a_high_risk_call(self):
|
|
315
|
+
# One tool whose risk depends on its arguments (like run_command).
|
|
316
|
+
log = []
|
|
317
|
+
registry = ToolRegistry()
|
|
318
|
+
|
|
319
|
+
def dyn(target=""):
|
|
320
|
+
log.append(target)
|
|
321
|
+
return "done"
|
|
322
|
+
|
|
323
|
+
registry.register(
|
|
324
|
+
Tool(
|
|
325
|
+
name="dyn",
|
|
326
|
+
description="dynamic risk",
|
|
327
|
+
parameters={
|
|
328
|
+
"type": "object",
|
|
329
|
+
"properties": {"target": {"type": "string"}},
|
|
330
|
+
"required": [],
|
|
331
|
+
},
|
|
332
|
+
func=dyn,
|
|
333
|
+
risk=RISK_ASK,
|
|
334
|
+
risk_func=lambda target="": (
|
|
335
|
+
RISK_HIGH if target == "danger" else RISK_ASK
|
|
336
|
+
),
|
|
337
|
+
preview=lambda target="": f"dyn {target}",
|
|
338
|
+
)
|
|
339
|
+
)
|
|
340
|
+
asked = []
|
|
341
|
+
policy = PermissionPolicy(ask=lambda t, p, r: asked.append(r) or "always")
|
|
342
|
+
core, _ = make_core(
|
|
343
|
+
[
|
|
344
|
+
msg('{"name": "dyn", "arguments": {"target": "ok"}}'),
|
|
345
|
+
msg("a"),
|
|
346
|
+
msg('{"name": "dyn", "arguments": {"target": "danger"}}'),
|
|
347
|
+
msg("b"),
|
|
348
|
+
],
|
|
349
|
+
registry=registry,
|
|
350
|
+
policy=policy,
|
|
351
|
+
)
|
|
352
|
+
messages = [{"role": "user", "content": "go"}]
|
|
353
|
+
core.run(messages) # ask tier -> 'always' remembered
|
|
354
|
+
core.run(messages) # high tier -> must ask again despite 'always'
|
|
355
|
+
self.assertEqual(log, ["ok", "danger"])
|
|
356
|
+
self.assertEqual(asked, [RISK_ASK, RISK_HIGH])
|
|
357
|
+
|
|
358
|
+
def test_auto_approve_skips_the_ask_tier(self):
|
|
359
|
+
log = []
|
|
360
|
+
asked = []
|
|
361
|
+
policy = PermissionPolicy(
|
|
362
|
+
ask=lambda tool, preview, risk: asked.append(tool.name) or "no",
|
|
363
|
+
auto_approve=True,
|
|
364
|
+
)
|
|
365
|
+
core, _ = make_core(
|
|
366
|
+
[msg('{"name": "danger", "arguments": {"target": "x"}}'), msg("done")],
|
|
367
|
+
registry=make_ask_registry(log),
|
|
368
|
+
policy=policy,
|
|
369
|
+
)
|
|
370
|
+
core.run([{"role": "user", "content": "go"}])
|
|
371
|
+
self.assertEqual(log, ["x"]) # ran without asking
|
|
372
|
+
self.assertEqual(asked, [])
|
|
373
|
+
|
|
374
|
+
def test_max_steps_is_enforced(self):
|
|
375
|
+
# Always returns a tool call -> must stop rather than loop forever.
|
|
376
|
+
core, _ = make_core(
|
|
377
|
+
[msg('{"name": "add", "arguments": {"a": 1, "b": 1}}')] * 3,
|
|
378
|
+
max_steps=3,
|
|
379
|
+
)
|
|
380
|
+
out = core.run([{"role": "user", "content": "add"}])
|
|
381
|
+
self.assertIn("maximum number of tool steps", out)
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
if __name__ == "__main__":
|
|
385
|
+
unittest.main(verbosity=2)
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""Tests for Phases 10-14: verification, rules, context, memory, git.
|
|
2
|
+
|
|
3
|
+
Run from the project root:
|
|
4
|
+
python tests/test_capabilities.py
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import subprocess
|
|
10
|
+
import sys
|
|
11
|
+
import tempfile
|
|
12
|
+
import unittest
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
|
16
|
+
|
|
17
|
+
from agent.context import message_size, trim_messages # noqa: E402
|
|
18
|
+
from agent.gitaware import changed_files, is_git_repo # noqa: E402
|
|
19
|
+
from agent.gitaware import status as git_status # noqa: E402
|
|
20
|
+
from agent.memory import Memory # noqa: E402
|
|
21
|
+
from agent.rules import build_system_prompt, find_rules_file, load_rules # noqa: E402
|
|
22
|
+
from agent.verification import detect_checks, format_checks # noqa: E402
|
|
23
|
+
from agent.workspace import Workspace # noqa: E402
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class Base(unittest.TestCase):
|
|
27
|
+
def setUp(self):
|
|
28
|
+
self._tmp = tempfile.TemporaryDirectory()
|
|
29
|
+
self.root = Path(self._tmp.name)
|
|
30
|
+
self.ws = Workspace(self.root)
|
|
31
|
+
|
|
32
|
+
def tearDown(self):
|
|
33
|
+
self._tmp.cleanup()
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# -- Phase 11: rules -------------------------------------------------------
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class RulesTests(Base):
|
|
40
|
+
def test_loads_agents_md(self):
|
|
41
|
+
(self.root / "AGENTS.md").write_text("# Rules\nBe careful.\n", encoding="utf-8")
|
|
42
|
+
text, source = load_rules(self.ws)
|
|
43
|
+
self.assertIn("Be careful", text)
|
|
44
|
+
self.assertEqual(source, "AGENTS.md")
|
|
45
|
+
|
|
46
|
+
def test_no_rules_returns_empty(self):
|
|
47
|
+
self.assertEqual(load_rules(self.ws), ("", None))
|
|
48
|
+
self.assertIsNone(find_rules_file(self.ws))
|
|
49
|
+
|
|
50
|
+
def test_build_system_prompt_injects_rules(self):
|
|
51
|
+
out = build_system_prompt("BASE", "Do not refactor.", "AGENTS.md")
|
|
52
|
+
self.assertIn("BASE", out)
|
|
53
|
+
self.assertIn("PROJECT RULES", out)
|
|
54
|
+
self.assertIn("Do not refactor.", out)
|
|
55
|
+
|
|
56
|
+
def test_build_system_prompt_without_rules_is_unchanged(self):
|
|
57
|
+
self.assertEqual(build_system_prompt("BASE", "", None), "BASE")
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
# -- Phase 12: context -----------------------------------------------------
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class ContextTests(unittest.TestCase):
|
|
64
|
+
def test_no_trim_when_under_budget(self):
|
|
65
|
+
messages = [{"role": "system", "content": "s"}, {"role": "user", "content": "hi"}]
|
|
66
|
+
self.assertEqual(trim_messages(messages, 10_000), messages)
|
|
67
|
+
|
|
68
|
+
def test_trims_oldest_and_notes_it(self):
|
|
69
|
+
messages = [{"role": "system", "content": "sys"}]
|
|
70
|
+
for i in range(20):
|
|
71
|
+
messages.append({"role": "user", "content": f"message-{i}-" + "x" * 100})
|
|
72
|
+
out = trim_messages(messages, 500)
|
|
73
|
+
self.assertLess(message_size(out), 500 + 200)
|
|
74
|
+
# newest kept, oldest dropped, and a note explaining it
|
|
75
|
+
self.assertTrue(any("message-19" in str(m.get("content")) for m in out))
|
|
76
|
+
self.assertTrue(any("omitted" in str(m.get("content")) for m in out))
|
|
77
|
+
# the system prompt is preserved
|
|
78
|
+
self.assertEqual(out[0]["role"], "system")
|
|
79
|
+
self.assertIn("sys", out[0]["content"])
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
# -- Phase 13: memory ------------------------------------------------------
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class MemoryTests(Base):
|
|
86
|
+
def test_history_roundtrip(self):
|
|
87
|
+
mem = Memory(self.ws)
|
|
88
|
+
mem.save_history([{"role": "user", "content": "hello"}])
|
|
89
|
+
self.assertEqual(mem.load_history(), [{"role": "user", "content": "hello"}])
|
|
90
|
+
self.assertTrue((self.root / ".agent" / "history.json").exists())
|
|
91
|
+
|
|
92
|
+
def test_notes_roundtrip(self):
|
|
93
|
+
mem = Memory(self.ws)
|
|
94
|
+
mem.add_note("Uses Laravel 11")
|
|
95
|
+
self.assertEqual(mem.load_notes(), ["Uses Laravel 11"])
|
|
96
|
+
|
|
97
|
+
def test_clear_removes_files(self):
|
|
98
|
+
mem = Memory(self.ws)
|
|
99
|
+
mem.save_history([{"role": "user", "content": "x"}])
|
|
100
|
+
mem.add_note("y")
|
|
101
|
+
mem.clear()
|
|
102
|
+
self.assertEqual(mem.load_history(), [])
|
|
103
|
+
self.assertEqual(mem.load_notes(), [])
|
|
104
|
+
|
|
105
|
+
def test_summary_mentions_paths(self):
|
|
106
|
+
mem = Memory(self.ws)
|
|
107
|
+
self.assertIn("history", mem.summary())
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
# -- Phase 10: verification ------------------------------------------------
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class VerificationTests(Base):
|
|
114
|
+
def test_detects_npm_scripts_smallest_first(self):
|
|
115
|
+
(self.root / "package.json").write_text(
|
|
116
|
+
json.dumps({"scripts": {"build": "vite build", "test": "jest", "lint": "eslint"}}),
|
|
117
|
+
encoding="utf-8",
|
|
118
|
+
)
|
|
119
|
+
checks = detect_checks(self.ws)
|
|
120
|
+
commands = [c.command for c in checks]
|
|
121
|
+
self.assertIn("npm run lint", commands)
|
|
122
|
+
self.assertIn("npm run test", commands)
|
|
123
|
+
self.assertIn("npm run build", commands)
|
|
124
|
+
self.assertEqual(commands[0], "npm run lint") # smallest/fastest first
|
|
125
|
+
|
|
126
|
+
def test_detects_python_tests(self):
|
|
127
|
+
(self.root / "tests").mkdir()
|
|
128
|
+
self.assertIn("python -m pytest -q", [c.command for c in detect_checks(self.ws)])
|
|
129
|
+
|
|
130
|
+
def test_detects_laravel(self):
|
|
131
|
+
(self.root / "composer.json").write_text("{}", encoding="utf-8")
|
|
132
|
+
(self.root / "artisan").write_text("", encoding="utf-8")
|
|
133
|
+
self.assertIn("php artisan test", [c.command for c in detect_checks(self.ws)])
|
|
134
|
+
|
|
135
|
+
def test_no_checks_message(self):
|
|
136
|
+
self.assertIn("No standard", format_checks(self.ws))
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
# -- Phase 14: git ---------------------------------------------------------
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
class GitTests(Base):
|
|
143
|
+
def test_non_repo(self):
|
|
144
|
+
self.assertFalse(is_git_repo(self.ws))
|
|
145
|
+
self.assertIn("not a git repository", git_status(self.ws).lower())
|
|
146
|
+
self.assertEqual(changed_files(self.ws), [])
|
|
147
|
+
|
|
148
|
+
def test_real_repo_status(self):
|
|
149
|
+
try:
|
|
150
|
+
subprocess.run(
|
|
151
|
+
["git", "init"],
|
|
152
|
+
cwd=str(self.root),
|
|
153
|
+
capture_output=True,
|
|
154
|
+
timeout=20,
|
|
155
|
+
check=False,
|
|
156
|
+
)
|
|
157
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
158
|
+
self.skipTest("git not available")
|
|
159
|
+
if not is_git_repo(self.ws):
|
|
160
|
+
self.skipTest("git init did not create a repo")
|
|
161
|
+
|
|
162
|
+
(self.root / "new_file.txt").write_text("hi", encoding="utf-8")
|
|
163
|
+
report = git_status(self.ws)
|
|
164
|
+
self.assertIn("new_file.txt", report)
|
|
165
|
+
self.assertIn("new_file.txt", changed_files(self.ws))
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
if __name__ == "__main__":
|
|
169
|
+
unittest.main(verbosity=2)
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""Deterministic tests for the file-modification module — no model, no network.
|
|
2
|
+
|
|
3
|
+
Run from the project root:
|
|
4
|
+
python tests/test_edits.py
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
import sys
|
|
9
|
+
import tempfile
|
|
10
|
+
import unittest
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
|
14
|
+
|
|
15
|
+
from agent.edits import ( # noqa: E402
|
|
16
|
+
create_file,
|
|
17
|
+
edit_file,
|
|
18
|
+
preview_create,
|
|
19
|
+
preview_edit,
|
|
20
|
+
)
|
|
21
|
+
from agent.workspace import Workspace # noqa: E402
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class EditTests(unittest.TestCase):
|
|
25
|
+
def setUp(self):
|
|
26
|
+
self._tmp = tempfile.TemporaryDirectory()
|
|
27
|
+
self.root = Path(self._tmp.name)
|
|
28
|
+
(self.root / "existing.txt").write_text("hello world\n", encoding="utf-8")
|
|
29
|
+
(self.root / "dup.txt").write_text("cat\ncat\n", encoding="utf-8")
|
|
30
|
+
self.ws = Workspace(self.root)
|
|
31
|
+
|
|
32
|
+
def tearDown(self):
|
|
33
|
+
self._tmp.cleanup()
|
|
34
|
+
|
|
35
|
+
# -- create ------------------------------------------------------------
|
|
36
|
+
|
|
37
|
+
def test_create_new_file(self):
|
|
38
|
+
out = create_file(self.ws, "new.txt", "content here\n")
|
|
39
|
+
self.assertIn("Created", out)
|
|
40
|
+
self.assertEqual((self.root / "new.txt").read_text(encoding="utf-8"), "content here\n")
|
|
41
|
+
|
|
42
|
+
def test_create_refuses_to_overwrite(self):
|
|
43
|
+
out = create_file(self.ws, "existing.txt", "CLOBBERED")
|
|
44
|
+
self.assertIn("already exists", out)
|
|
45
|
+
# original file untouched
|
|
46
|
+
self.assertEqual((self.root / "existing.txt").read_text(encoding="utf-8"), "hello world\n")
|
|
47
|
+
|
|
48
|
+
def test_create_can_make_parent_dirs(self):
|
|
49
|
+
out = create_file(self.ws, "pkg/sub/new.txt", "x")
|
|
50
|
+
self.assertIn("Created", out)
|
|
51
|
+
self.assertTrue((self.root / "pkg" / "sub" / "new.txt").exists())
|
|
52
|
+
|
|
53
|
+
def test_create_outside_root_rejected(self):
|
|
54
|
+
out = create_file(self.ws, "../evil.txt", "x")
|
|
55
|
+
self.assertIn("ERROR", out)
|
|
56
|
+
self.assertFalse((self.root.parent / "evil.txt").exists())
|
|
57
|
+
|
|
58
|
+
def test_create_requires_path(self):
|
|
59
|
+
self.assertIn("ERROR", create_file(self.ws, "", "x"))
|
|
60
|
+
|
|
61
|
+
def test_create_rejects_non_string_content(self):
|
|
62
|
+
self.assertIn("ERROR", create_file(self.ws, "n.txt", None)) # type: ignore[arg-type]
|
|
63
|
+
|
|
64
|
+
# -- edit --------------------------------------------------------------
|
|
65
|
+
|
|
66
|
+
def test_edit_replaces_unique_text(self):
|
|
67
|
+
out = edit_file(self.ws, "existing.txt", "hello", "goodbye")
|
|
68
|
+
self.assertIn("Updated", out)
|
|
69
|
+
self.assertEqual((self.root / "existing.txt").read_text(encoding="utf-8"), "goodbye world\n")
|
|
70
|
+
|
|
71
|
+
def test_edit_reports_when_text_missing(self):
|
|
72
|
+
out = edit_file(self.ws, "existing.txt", "not-there", "x")
|
|
73
|
+
self.assertIn("was not found", out)
|
|
74
|
+
self.assertEqual((self.root / "existing.txt").read_text(encoding="utf-8"), "hello world\n")
|
|
75
|
+
|
|
76
|
+
def test_edit_ambiguous_requires_replace_all(self):
|
|
77
|
+
out = edit_file(self.ws, "dup.txt", "cat", "dog")
|
|
78
|
+
self.assertIn("occurrences", out)
|
|
79
|
+
# nothing changed
|
|
80
|
+
self.assertEqual((self.root / "dup.txt").read_text(encoding="utf-8"), "cat\ncat\n")
|
|
81
|
+
|
|
82
|
+
def test_edit_replace_all(self):
|
|
83
|
+
out = edit_file(self.ws, "dup.txt", "cat", "dog", replace_all=True)
|
|
84
|
+
self.assertIn("2 occurrence", out)
|
|
85
|
+
self.assertEqual((self.root / "dup.txt").read_text(encoding="utf-8"), "dog\ndog\n")
|
|
86
|
+
|
|
87
|
+
def test_edit_missing_file(self):
|
|
88
|
+
out = edit_file(self.ws, "nope.txt", "a", "b")
|
|
89
|
+
self.assertIn("does not exist", out)
|
|
90
|
+
|
|
91
|
+
def test_edit_rejects_empty_old_text(self):
|
|
92
|
+
self.assertIn("ERROR", edit_file(self.ws, "existing.txt", "", "x"))
|
|
93
|
+
|
|
94
|
+
def test_edit_binary_file_refused(self):
|
|
95
|
+
(self.root / "blob.bin").write_bytes(b"\x00\x01hello")
|
|
96
|
+
out = edit_file(self.ws, "blob.bin", "hello", "bye")
|
|
97
|
+
self.assertIn("binary", out.lower())
|
|
98
|
+
|
|
99
|
+
def test_edit_outside_root_rejected(self):
|
|
100
|
+
out = edit_file(self.ws, "../evil.txt", "a", "b")
|
|
101
|
+
self.assertIn("ERROR", out)
|
|
102
|
+
|
|
103
|
+
# -- previews ----------------------------------------------------------
|
|
104
|
+
|
|
105
|
+
def test_preview_create_shows_additions(self):
|
|
106
|
+
out = preview_create("new.txt", "line one\nline two")
|
|
107
|
+
self.assertIn("CREATE new.txt", out)
|
|
108
|
+
self.assertIn("+ line one", out)
|
|
109
|
+
|
|
110
|
+
def test_preview_edit_shows_both_sides(self):
|
|
111
|
+
out = preview_edit("existing.txt", "hello", "goodbye")
|
|
112
|
+
self.assertIn("- hello", out)
|
|
113
|
+
self.assertIn("+ goodbye", out)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
if __name__ == "__main__":
|
|
117
|
+
unittest.main(verbosity=2)
|