pwn 0.5.680 → 0.5.682
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.gitignore +1 -0
- data/documentation/AI-Integration.md +1 -1
- data/documentation/Agent-Tool-Registry.md +1 -1
- data/documentation/Configuration.md +3 -6
- data/documentation/How-PWN-Works.md +2 -2
- data/documentation/Reinforcement-Learning.md +1 -1
- data/documentation/diagrams/dot/task-summarizer.dot +3 -3
- data/documentation/pwn-ai-Agent.md +16 -35
- data/lib/pwn/ai/agent/loop.rb +230 -191
- data/lib/pwn/ai/agent/mistakes.rb +17 -0
- data/lib/pwn/ai/agent/policy.rb +55 -5
- data/lib/pwn/ai/agent/prompt_builder.rb +38 -12
- data/lib/pwn/ai/agent/registry.rb +17 -9
- data/lib/pwn/ai/agent/task_summarizer.rb +374 -503
- data/lib/pwn/config.rb +1 -1
- data/lib/pwn/version.rb +1 -1
- data/spec/integration/prompt_builder_spec.rb +6 -4
- data/spec/lib/pwn/ai/agent/loop_spec.rb +251 -33
- data/spec/lib/pwn/ai/agent/mistakes_spec.rb +14 -0
- data/spec/lib/pwn/ai/agent/policy_spec.rb +52 -1
- data/spec/lib/pwn/ai/agent/prompt_builder_spec.rb +4 -9
- data/spec/lib/pwn/ai/agent/registry_spec.rb +22 -2
- data/spec/lib/pwn/ai/agent/signal_hygiene_spec.rb +4 -5
- data/spec/lib/pwn/ai/agent/task_summarizer_spec.rb +321 -90
- data/third_party/pwn_rdoc.jsonl +24 -9
- metadata +1 -1
data/lib/pwn/config.rb
CHANGED
|
@@ -160,7 +160,7 @@ module PWN
|
|
|
160
160
|
# LLM tangible-task decomposition for autonomous goals (default on).
|
|
161
161
|
task_summary_llm: nil,
|
|
162
162
|
tool_router: nil, # nil = auto (true when :active is local :ollama/:openwebui) — cuts ~11k→~3k schema tokens
|
|
163
|
-
tool_preference: %w[memory_recall
|
|
163
|
+
tool_preference: %w[memory_recall pwn_eval shell mistakes_record mistakes_resolve learning_note_outcome memory_remember],
|
|
164
164
|
escalation_persona: 'escalator', # Swarm persona for frontier corrective hints when a local model is stuck
|
|
165
165
|
# sample E3 verify_as_reward: true|false|nil(auto: ~10% local / always frontier when CLAIM_RX hits)
|
|
166
166
|
verify_as_reward: nil,
|
data/lib/pwn/version.rb
CHANGED
|
@@ -29,14 +29,16 @@ RSpec.describe 'PWN::AI::Agent::PromptBuilder', :aggregate_failures do
|
|
|
29
29
|
PWN::Env[:ai] = { active: :anthropic, module_reflection: false, agent: @agent_cfg }
|
|
30
30
|
end
|
|
31
31
|
|
|
32
|
-
it '
|
|
33
|
-
prompt = builder.build(session_id: 'sess_abc')
|
|
34
|
-
['ENVIRONMENT', '
|
|
32
|
+
it 'mid-turn autonomous prompt keeps request + CORE_TOOLS + known-fix, not the full harness' do
|
|
33
|
+
prompt = builder.build(session_id: 'sess_abc', request: 'Write hello into /tmp/x and verify it')
|
|
34
|
+
['ENVIRONMENT', 'KNOWN MISTAKES', 'TOOL USE'].each do |hdr|
|
|
35
35
|
expect(prompt).to include(hdr), "missing section: #{hdr}"
|
|
36
36
|
end
|
|
37
37
|
expect(prompt).to include('session_id : sess_abc')
|
|
38
38
|
expect(prompt).to include(PWN::VERSION)
|
|
39
|
-
|
|
39
|
+
['MEMORY', 'SKILLS', 'LEARNING', 'TOOL EFFECTIVENESS', 'EXTROSPECTION', 'POLICY'].each do |hdr|
|
|
40
|
+
expect(prompt).not_to include(hdr), "mid-turn prompt still injects #{hdr}"
|
|
41
|
+
end
|
|
40
42
|
end
|
|
41
43
|
|
|
42
44
|
it 'never leaks a raw nil or an unrendered #{...} interpolation' do
|
|
@@ -54,6 +54,44 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
|
|
|
54
54
|
expect(src.scan('inject_task_focus!').length).to be >= 3
|
|
55
55
|
end
|
|
56
56
|
|
|
57
|
+
it 'does not keep injecting English focus after the plan is covered' do
|
|
58
|
+
src = File.read(described_class.method(:inject_task_focus!).source_location.first)
|
|
59
|
+
focus = src[/private_class_method def self\.inject_task_focus!.*?private_class_method def self\.\w+/m]
|
|
60
|
+
focus ||= src
|
|
61
|
+
expect(focus).to match(/plan_open\?/)
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
it 'does not tell an open English plan to stop after 3 tools just because budget is hot' do
|
|
65
|
+
src = File.read(described_class.method(:run).source_location.first)
|
|
66
|
+
expect(src).to match(/budget_exhaustion_hot\?/)
|
|
67
|
+
expect(src).to match(/plan_open\?/)
|
|
68
|
+
# local ≤3-tool abort is only for a closed/short plan, not mid-goal.
|
|
69
|
+
expect(src).to match(/english_open|plan_open\?/)
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
it 'parks stale extra budget scars even while the host is hot' do
|
|
73
|
+
tmp = Dir.mktmpdir
|
|
74
|
+
stub_const('PWN::AI::Agent::Mistakes::MISTAKES_FILE', File.join(tmp, 'mistakes.json'))
|
|
75
|
+
PWN::AI::Agent::Mistakes.reset if PWN::AI::Agent::Mistakes.respond_to?(:reset)
|
|
76
|
+
a = PWN::AI::Agent::Mistakes.record(
|
|
77
|
+
tool: 'agent_loop',
|
|
78
|
+
error: '[pwn-ai] iteration budget exhausted A',
|
|
79
|
+
shape: 'budget_exhausted'
|
|
80
|
+
)
|
|
81
|
+
b = PWN::AI::Agent::Mistakes.record(
|
|
82
|
+
tool: 'agent_loop',
|
|
83
|
+
error: '[pwn-ai] iteration budget exhausted B',
|
|
84
|
+
shape: 'budget_exhausted'
|
|
85
|
+
)
|
|
86
|
+
described_class.send(:maybe_park_budget_scars!)
|
|
87
|
+
parked_n = [a, b].count do |m|
|
|
88
|
+
PWN::AI::Agent::Mistakes.find(signature: m[:signature])[:parked]
|
|
89
|
+
end
|
|
90
|
+
expect(parked_n).to be >= 1
|
|
91
|
+
ensure
|
|
92
|
+
FileUtils.remove_entry(tmp) if tmp && Dir.exist?(tmp)
|
|
93
|
+
end
|
|
94
|
+
|
|
57
95
|
it 're-ranks Registry tools from English tangible tasks after plan (sole driver)' do
|
|
58
96
|
src = File.read(described_class.method(:run).source_location.first)
|
|
59
97
|
expect(src).to match(/TaskSummarizer\.relevance_query/)
|
|
@@ -82,7 +120,7 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
|
|
|
82
120
|
|
|
83
121
|
last_plan = { plan: %w[identify implement verify], plan_idx: 2 }
|
|
84
122
|
msgs_mut = msgs + [
|
|
85
|
-
{ role: 'tool', content: '{"success":true,"result":{"stdout":"
|
|
123
|
+
{ role: 'tool', content: '{"success":true,"result":{"stdout":"patched loop.rb\n0 offenses detected"}}' }
|
|
86
124
|
]
|
|
87
125
|
r_mut = loop_mod.send(
|
|
88
126
|
:evidence_enough_to_finalize?,
|
|
@@ -98,10 +136,199 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
|
|
|
98
136
|
)
|
|
99
137
|
expect(r_bare).to be false
|
|
100
138
|
|
|
139
|
+
# plan_idx on last item is not enough if mutate/verify English tasks lack evidence
|
|
140
|
+
r_idx_only = loop_mod.send(
|
|
141
|
+
:evidence_enough_to_finalize?,
|
|
142
|
+
messages: msgs, turn_fails: {}, i: 5, max_iters: 40,
|
|
143
|
+
request: 'fix p17', plan_steps: 3, ts_state: last_plan
|
|
144
|
+
)
|
|
145
|
+
expect(r_idx_only).to be false
|
|
146
|
+
|
|
101
147
|
# call site must pass ts_state
|
|
102
148
|
src = File.read(loop_mod.method(:run).source_location.first)
|
|
103
149
|
expect(src).to match(/evidence_enough_to_finalize\?\([\s\S]*?ts_state: ts_state/)
|
|
104
|
-
expect(src).to match(/
|
|
150
|
+
expect(src).to match(/original request|request is the completion/i)
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
it 'P17 does not inject finalize while implement/verify English tasks remain open' do
|
|
154
|
+
loop_mod = described_class
|
|
155
|
+
ls_find = [
|
|
156
|
+
{ role: 'user', content: 'map then implement then verify' },
|
|
157
|
+
{ role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"ls: loop.rb","exit":0}}' },
|
|
158
|
+
{ role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"find: task_summarizer.rb","exit":0}}' },
|
|
159
|
+
{ role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"resolved file listing","exit":0}}' }
|
|
160
|
+
]
|
|
161
|
+
mid = {
|
|
162
|
+
plan: [
|
|
163
|
+
'Map how tasks complete',
|
|
164
|
+
'Implement the completion fix',
|
|
165
|
+
'Verify full task completion end-to-end'
|
|
166
|
+
],
|
|
167
|
+
plan_idx: 0
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
r_active = loop_mod.send(
|
|
171
|
+
:evidence_enough_to_finalize?,
|
|
172
|
+
messages: ls_find, turn_fails: {}, i: 4, max_iters: 40,
|
|
173
|
+
request: 'map then implement then verify', plan_steps: 3, ts_state: mid
|
|
174
|
+
)
|
|
175
|
+
expect(r_active).to be false
|
|
176
|
+
|
|
177
|
+
jumped = mid.merge(plan_idx: 2)
|
|
178
|
+
r_jumped = loop_mod.send(
|
|
179
|
+
:evidence_enough_to_finalize?,
|
|
180
|
+
messages: ls_find, turn_fails: {}, i: 4, max_iters: 40,
|
|
181
|
+
request: 'map then implement then verify', plan_steps: 3, ts_state: jumped
|
|
182
|
+
)
|
|
183
|
+
expect(r_jumped).to be false
|
|
184
|
+
expect(
|
|
185
|
+
PWN::AI::Agent::TaskSummarizer.plan_open?(state: jumped, messages: ls_find)
|
|
186
|
+
).to eq true
|
|
187
|
+
left = PWN::AI::Agent::TaskSummarizer.unfinished_tasks(state: jumped, messages: ls_find)
|
|
188
|
+
expect(left.map { |t| t[:item] }.join(' ')).to match(/Implement|Verify/i)
|
|
189
|
+
|
|
190
|
+
src = File.read(loop_mod.method(:run).source_location.first)
|
|
191
|
+
expect(src).to match(/Write the complete final answer now/)
|
|
192
|
+
expect(src).to match(/evidence_enough_to_finalize\?/)
|
|
193
|
+
expect(src).to match(/Do NOT call more tools/)
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
it 'P17 can early-final when English tasks are covered even if plan_idx is still 0' do
|
|
197
|
+
loop_mod = described_class
|
|
198
|
+
done = {
|
|
199
|
+
plan: [
|
|
200
|
+
'Determine the local hostname',
|
|
201
|
+
'Present the result and report completion'
|
|
202
|
+
],
|
|
203
|
+
plan_idx: 0
|
|
204
|
+
}
|
|
205
|
+
msgs = [
|
|
206
|
+
{ role: 'user', content: 'what is my hostname?' },
|
|
207
|
+
{ role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"kali-box","exit":0}}' },
|
|
208
|
+
{ role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"kali-box","exit":0}}' }
|
|
209
|
+
]
|
|
210
|
+
expect(
|
|
211
|
+
PWN::AI::Agent::TaskSummarizer.plan_open?(state: done, messages: msgs)
|
|
212
|
+
).to eq false
|
|
213
|
+
r = loop_mod.send(
|
|
214
|
+
:evidence_enough_to_finalize?,
|
|
215
|
+
messages: msgs, turn_fails: {}, i: 4, max_iters: 40,
|
|
216
|
+
request: 'what is my hostname?', plan_steps: 2, ts_state: done
|
|
217
|
+
)
|
|
218
|
+
expect(r).to be true
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
it 'P17 can early-final a finished request even while advisory English tasks remain' do
|
|
222
|
+
loop_mod = described_class
|
|
223
|
+
leftover_plan = {
|
|
224
|
+
plan: [
|
|
225
|
+
'Determine the local hostname',
|
|
226
|
+
'run rspec to verify'
|
|
227
|
+
],
|
|
228
|
+
plan_idx: 0
|
|
229
|
+
}
|
|
230
|
+
msgs = [
|
|
231
|
+
{ role: 'user', content: 'what is my hostname?' },
|
|
232
|
+
{ role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"kali-box","exit":0}}' },
|
|
233
|
+
{ role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"kali-box","exit":0}}' }
|
|
234
|
+
]
|
|
235
|
+
expect(
|
|
236
|
+
PWN::AI::Agent::TaskSummarizer.plan_open?(state: leftover_plan, messages: msgs)
|
|
237
|
+
).to eq true
|
|
238
|
+
r = loop_mod.send(
|
|
239
|
+
:evidence_enough_to_finalize?,
|
|
240
|
+
messages: msgs, turn_fails: {}, i: 4, max_iters: 40,
|
|
241
|
+
request: 'what is my hostname?', plan_steps: 2, ts_state: leftover_plan
|
|
242
|
+
)
|
|
243
|
+
expect(r).to be true
|
|
244
|
+
end
|
|
245
|
+
|
|
246
|
+
it 'open_plan_blocks_final? does not govern completion — the original request does' do
|
|
247
|
+
src = File.read(described_class.method(:run).source_location.first)
|
|
248
|
+
expect(src).to match(/original request is the completion signal/i)
|
|
249
|
+
expect(src).not_to match(/if open_plan_blocks_final\?/)
|
|
250
|
+
end
|
|
251
|
+
|
|
252
|
+
it 'request_unsatisfied? uses request-shaped evidence, not just mutation regex' do
|
|
253
|
+
ls_only = [
|
|
254
|
+
{ role: 'user', content: 'write complete into /tmp/x.txt and verify it' },
|
|
255
|
+
{
|
|
256
|
+
role: 'assistant',
|
|
257
|
+
tool_calls: [
|
|
258
|
+
{ function: { name: 'shell', arguments: '{"command":"ls /tmp"}' } }
|
|
259
|
+
]
|
|
260
|
+
},
|
|
261
|
+
{ role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"x","exit":0}}' }
|
|
262
|
+
]
|
|
263
|
+
expect(
|
|
264
|
+
described_class.send(
|
|
265
|
+
:request_unsatisfied?,
|
|
266
|
+
request: 'Write complete into /tmp/x.txt and verify it reads complete',
|
|
267
|
+
messages: ls_only,
|
|
268
|
+
last_iter: false
|
|
269
|
+
)
|
|
270
|
+
).to eq true
|
|
271
|
+
# Path-shaped write without the old printf/sed regex still satisfies the request.
|
|
272
|
+
written = ls_only + [
|
|
273
|
+
{
|
|
274
|
+
role: 'assistant',
|
|
275
|
+
tool_calls: [
|
|
276
|
+
{ function: { name: 'shell', arguments: '{"command":"python3 -c \\"open(\'/tmp/x.txt\',\'w\').write(\'complete\')\\""}' } }
|
|
277
|
+
]
|
|
278
|
+
},
|
|
279
|
+
{
|
|
280
|
+
role: 'tool',
|
|
281
|
+
name: 'shell',
|
|
282
|
+
content: '{"success":true,"result":{"stdout":"","stderr":"","exit":0}}'
|
|
283
|
+
}
|
|
284
|
+
]
|
|
285
|
+
expect(
|
|
286
|
+
described_class.send(
|
|
287
|
+
:request_unsatisfied?,
|
|
288
|
+
request: 'Write complete into /tmp/x.txt and verify it reads complete',
|
|
289
|
+
messages: written,
|
|
290
|
+
last_iter: false
|
|
291
|
+
)
|
|
292
|
+
).to eq false
|
|
293
|
+
expect(
|
|
294
|
+
described_class.send(
|
|
295
|
+
:request_unsatisfied?,
|
|
296
|
+
request: 'what is my hostname?',
|
|
297
|
+
messages: [
|
|
298
|
+
{ role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"kali","exit":0}}' }
|
|
299
|
+
],
|
|
300
|
+
last_iter: false
|
|
301
|
+
)
|
|
302
|
+
).to eq false
|
|
303
|
+
end
|
|
304
|
+
|
|
305
|
+
it 'does not bounce world-knowledge asks that need no host work' do
|
|
306
|
+
expect(described_class.needs_host_work?(request: 'what color is a cherry')).to eq false
|
|
307
|
+
expect(described_class.world_knowledge?(request: 'what color is a cherry')).to eq true
|
|
308
|
+
expect(
|
|
309
|
+
described_class.send(
|
|
310
|
+
:request_unsatisfied?,
|
|
311
|
+
request: 'what color is a cherry',
|
|
312
|
+
messages: [{ role: 'assistant', content: 'Red.', tool_calls: [] }],
|
|
313
|
+
last_iter: false
|
|
314
|
+
)
|
|
315
|
+
).to eq false
|
|
316
|
+
expect(described_class.needs_host_work?(request: 'Write hello into /tmp/x.txt')).to eq true
|
|
317
|
+
end
|
|
318
|
+
|
|
319
|
+
it 'last-iter strips tools only on the true last slot; leftover English tasks do not force required' do
|
|
320
|
+
src = File.read(described_class.method(:run).source_location.first)
|
|
321
|
+
expect(src).to match(/text_only_iters = 1/)
|
|
322
|
+
expect(src).not_to match(/english_open && hot/)
|
|
323
|
+
expect(src).to match(/has_tool_result \|\| !need_tools \? 'auto' : 'required'/)
|
|
324
|
+
expect(src).not_to match(/still_acting/)
|
|
325
|
+
end
|
|
326
|
+
|
|
327
|
+
it 'Loop.run default tool pool is CORE_TOOLS, not the full registry' do
|
|
328
|
+
src = File.read(described_class.method(:run).source_location.first)
|
|
329
|
+
expect(src).to match(/fetch\(:core_only,\s*true\)/)
|
|
330
|
+
expect(src).to match(/core_only:/)
|
|
331
|
+
expect(src).to match(/CORE_TOOLS/)
|
|
105
332
|
end
|
|
106
333
|
|
|
107
334
|
it 'does not put TaskSummarizer into Reward credit paths' do
|
|
@@ -120,9 +347,8 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
|
|
|
120
347
|
expect(src).to match(/tool_calls_from_text/)
|
|
121
348
|
expect(src).to match(/_text_tool_coerced/)
|
|
122
349
|
expect(src).to match(/tool_choice/)
|
|
123
|
-
#
|
|
124
|
-
expect(src).to match(/
|
|
125
|
-
expect(src).to match(/has_tool_result && !still_acting/)
|
|
350
|
+
# After the first tool result, auto. English leftovers do not keep required.
|
|
351
|
+
expect(src).to match(/has_tool_result \|\| !need_tools \? 'auto' : 'required'/)
|
|
126
352
|
# normalize_llm must promote text tool forms
|
|
127
353
|
expect(src).to match(/normalize_llm[\s\S]*tool_calls_from_text/m)
|
|
128
354
|
expect(src).to match(/MONOLOGUE_TOOL_INTENT_RX/)
|
|
@@ -154,26 +380,16 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
|
|
|
154
380
|
expect(described_class.request_intent(request: 'refactor Loop.run and run rubocop')).to eq(:act)
|
|
155
381
|
end
|
|
156
382
|
|
|
157
|
-
it '
|
|
158
|
-
expect(described_class).
|
|
159
|
-
expect(described_class.request_kind(request: 'FYI the build is green.')).to eq(:statement)
|
|
160
|
-
expect(described_class.request_kind(request: 'how to do a ping sweep with hping3?')).to eq(:question)
|
|
161
|
-
expect(described_class.request_kind(request: 'what did I just say?')).to eq(:question)
|
|
162
|
-
expect(described_class.request_kind(request: 'hi')).to eq(:statement)
|
|
163
|
-
expect(described_class.request_kind(request: 'refactor Loop.run and run rubocop')).to eq(:autonomous_goal)
|
|
164
|
-
expect(described_class.request_kind(request: 'using hping3 what live hosts can you find in this subnet?')).to eq(:autonomous_goal)
|
|
165
|
-
# Live host-fact Qs need tools — not text-only question path.
|
|
166
|
-
expect(described_class.request_kind(request: 'what is my hostname?')).to eq(:autonomous_goal)
|
|
167
|
-
expect(described_class.request_kind(request: 'excellent - what is my hostname?')).to eq(:autonomous_goal)
|
|
383
|
+
it 'does not classify request types — Loop has no request_kind' do
|
|
384
|
+
expect(described_class).not_to respond_to :request_kind
|
|
168
385
|
end
|
|
169
386
|
|
|
170
|
-
it 'run short-
|
|
387
|
+
it 'run does not short-circuit statement/question — they take the full tool loop' do
|
|
171
388
|
src = File.read(described_class.method(:run).source_location.first)
|
|
172
|
-
expect(src).
|
|
173
|
-
expect(src).
|
|
174
|
-
expect(src).
|
|
175
|
-
expect(src).to match(/
|
|
176
|
-
expect(src).to match(/answer_question/)
|
|
389
|
+
expect(src).not_to match(/kind\.to_sym == :statement/)
|
|
390
|
+
expect(src).not_to match(/kind\.to_sym == :question/)
|
|
391
|
+
expect(src).not_to match(/Request type/)
|
|
392
|
+
expect(src).to match(/Every request gets a task compass/)
|
|
177
393
|
end
|
|
178
394
|
|
|
179
395
|
it 'classifies pure prior-turn recall and vague memory cues as :recall' do
|
|
@@ -214,13 +430,10 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
|
|
|
214
430
|
it 'run short-circuits how-to without plan_first or tools' do
|
|
215
431
|
src = File.read(described_class.method(:run).source_location.first)
|
|
216
432
|
expect(src).to match(/request_intent/)
|
|
217
|
-
expect(src).
|
|
433
|
+
expect(src).not_to match(/request_kind/)
|
|
218
434
|
expect(src).to match(/answer_howto/)
|
|
219
|
-
expect(src).to match(/answer_statement/)
|
|
220
|
-
expect(src).to match(/answer_question/)
|
|
221
435
|
expect(src).to match(/intent == :howto/)
|
|
222
436
|
expect(src).to match(/skip_plan/)
|
|
223
|
-
expect(src).to match(/needs_breakdown/)
|
|
224
437
|
end
|
|
225
438
|
|
|
226
439
|
it 'answer_howto does not register shell tool use in source path' do
|
|
@@ -235,8 +448,7 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
|
|
|
235
448
|
expect(src).to match(/answer_recall/)
|
|
236
449
|
expect(src).to match(/intent == :recall.*?return answer_recall/m)
|
|
237
450
|
expect(src).to match(/skip_plan/)
|
|
238
|
-
expect(src).
|
|
239
|
-
expect(src).to match(/needs_breakdown|autonomous_goal|statement|question/)
|
|
451
|
+
expect(src).not_to match(/request_kind/)
|
|
240
452
|
end
|
|
241
453
|
|
|
242
454
|
it 'run short-circuits greetings via answer_greeting before plan_first' do
|
|
@@ -245,8 +457,7 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
|
|
|
245
457
|
expect(src).to match(/answer_greeting/)
|
|
246
458
|
expect(src).to match(/intent == :greeting.*?return answer_greeting/m)
|
|
247
459
|
expect(src).to match(/skip_plan/)
|
|
248
|
-
expect(src).
|
|
249
|
-
expect(src).to match(/needs_breakdown|autonomous_goal|statement|question/)
|
|
460
|
+
expect(src).not_to match(/request_kind/)
|
|
250
461
|
end
|
|
251
462
|
|
|
252
463
|
it 'answer_greeting returns fixed ack without weather echo or tools' do
|
|
@@ -474,16 +685,23 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
|
|
|
474
685
|
end
|
|
475
686
|
end
|
|
476
687
|
|
|
688
|
+
it 'Loop.run finishes the original request; English tasks are advisory only' do
|
|
689
|
+
src = File.read(described_class.method(:run).source_location.first)
|
|
690
|
+
expect(src).to match(/original request is the completion signal/i)
|
|
691
|
+
expect(src).not_to match(/do NOT finalize/i)
|
|
692
|
+
expect(src).not_to match(/if open_plan_blocks_final\?/)
|
|
693
|
+
end
|
|
694
|
+
|
|
477
695
|
it 'Loop.run marks the Hermes user-path so TurnFinalizer can defer' do
|
|
478
696
|
src = File.read(described_class.method(:run).source_location.first)
|
|
479
697
|
expect(src).to match(/TurnFinalizer\.enter_user_path!/)
|
|
480
698
|
expect(src).to match(/TurnFinalizer\.leave_user_path!/)
|
|
481
699
|
end
|
|
482
700
|
|
|
483
|
-
it 'should_auto_introspect skips cheap greeting/
|
|
701
|
+
it 'should_auto_introspect skips cheap greeting/howto/recall answers' do
|
|
484
702
|
src = File.read(described_class.method(:run).source_location.first)
|
|
485
703
|
expect(src).to include('return false if %i[greeting howto recall].include?(intent)')
|
|
486
|
-
expect(src).
|
|
487
|
-
expect(src).
|
|
704
|
+
expect(src).not_to include('return false if kind == :statement')
|
|
705
|
+
expect(src).not_to include('return false if kind == :question && fails.zero?')
|
|
488
706
|
end
|
|
489
707
|
end # rubocop:enable Metrics/BlockLength
|
|
@@ -170,4 +170,18 @@ describe PWN::AI::Agent::Mistakes do
|
|
|
170
170
|
expect(out[:extinguished]).to be >= 1
|
|
171
171
|
expect(described_class.find(signature: m[:signature])[:resolved]).to be true
|
|
172
172
|
end
|
|
173
|
+
|
|
174
|
+
it 'to_context downranks budget-exhaustion scars on unrelated requests' do
|
|
175
|
+
stub_const('PWN::AI::Agent::Mistakes::MISTAKES_FILE', File.join(Dir.mktmpdir, 'mistakes.json'))
|
|
176
|
+
described_class.reset if described_class.respond_to?(:reset)
|
|
177
|
+
described_class.record(
|
|
178
|
+
tool: 'agent_loop',
|
|
179
|
+
error: '[pwn-ai] iteration budget exhausted',
|
|
180
|
+
shape: 'budget_exhausted'
|
|
181
|
+
)
|
|
182
|
+
described_class.record(tool: 'shell', error: 'nmpa: command not found unique-host')
|
|
183
|
+
ctx = described_class.to_context(request: 'what is my hostname?', limit: 2)
|
|
184
|
+
expect(ctx).to include('shell')
|
|
185
|
+
expect(ctx).not_to match(/iteration budget exhausted/)
|
|
186
|
+
end
|
|
173
187
|
end
|
|
@@ -80,7 +80,11 @@ describe PWN::AI::Agent::Policy do
|
|
|
80
80
|
kind: :autonomous_goal,
|
|
81
81
|
request: 'fix the reward judge',
|
|
82
82
|
engine: :grok,
|
|
83
|
-
ts_state: {
|
|
83
|
+
ts_state: {
|
|
84
|
+
plan: %w[inspect tighten verify],
|
|
85
|
+
plan_idx: 2,
|
|
86
|
+
evidence_blob: 'inspected reward.rb patched judge 12 examples, 0 failures'
|
|
87
|
+
},
|
|
84
88
|
final: 'The cheap ORM now grades the last tools and usable result.',
|
|
85
89
|
score: 0.82
|
|
86
90
|
)
|
|
@@ -91,6 +95,53 @@ describe PWN::AI::Agent::Policy do
|
|
|
91
95
|
expect(open_s).not_to eq(done_s)
|
|
92
96
|
end
|
|
93
97
|
|
|
98
|
+
it 'quality bin is not high while English tasks remain even if plan_idx is last' do
|
|
99
|
+
last_but_open = described_class.state(
|
|
100
|
+
kind: :autonomous_goal,
|
|
101
|
+
request: 'fix the reward judge',
|
|
102
|
+
engine: :grok,
|
|
103
|
+
ts_state: { plan: %w[inspect tighten verify], plan_idx: 2 }
|
|
104
|
+
)
|
|
105
|
+
expect(last_but_open).to include('|p')
|
|
106
|
+
expect(last_but_open).not_to include('|ph')
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
it 'observe_step credits English-task progress above bare tool-ok hygiene' do
|
|
110
|
+
tmp = Dir.mktmpdir
|
|
111
|
+
stub_const('PWN::AI::Agent::Policy::POLICY_FILE', File.join(tmp, 'policy.json'))
|
|
112
|
+
stub_const('PWN::AI::Agent::Policy::TRAJECTORY_FILE', File.join(tmp, 'policy_traj.jsonl'))
|
|
113
|
+
described_class.reset
|
|
114
|
+
allow(described_class).to receive(:enabled?).and_return(true)
|
|
115
|
+
PWN::Env[:ai] ||= {}
|
|
116
|
+
PWN::Env[:ai][:agent] ||= {}
|
|
117
|
+
PWN::Env[:ai][:agent][:policy] = true
|
|
118
|
+
|
|
119
|
+
ts = {
|
|
120
|
+
plan: ['locate the source', 'fix the truncation bug', 'run rspec to verify'],
|
|
121
|
+
plan_idx: 0,
|
|
122
|
+
evidence_blob: ''
|
|
123
|
+
}
|
|
124
|
+
described_class.begin_episode(
|
|
125
|
+
session_id: 'en_credit',
|
|
126
|
+
request: 'locate then fix',
|
|
127
|
+
kind: :autonomous_goal,
|
|
128
|
+
engine: :grok,
|
|
129
|
+
ts_state: ts
|
|
130
|
+
)
|
|
131
|
+
grind = described_class.observe_step(
|
|
132
|
+
action: 'shell', ok: true, session_id: 'en_credit', ts_state: ts
|
|
133
|
+
)
|
|
134
|
+
ts[:plan_idx] = 1
|
|
135
|
+
ts[:evidence_blob] = 'rg hit locate the source lib/task_summarizer.rb patched later'
|
|
136
|
+
advance = described_class.observe_step(
|
|
137
|
+
action: 'shell', ok: true, session_id: 'en_credit', ts_state: ts
|
|
138
|
+
)
|
|
139
|
+
expect(advance[:reward].to_f).to be > grind[:reward].to_f
|
|
140
|
+
ensure
|
|
141
|
+
described_class.reset
|
|
142
|
+
FileUtils.remove_entry(tmp) if tmp && Dir.exist?(tmp)
|
|
143
|
+
end
|
|
144
|
+
|
|
94
145
|
it 'warmup! meets the episode budget so greedy suggestions appear' do
|
|
95
146
|
tmp = Dir.mktmpdir
|
|
96
147
|
stub_const('PWN::AI::Agent::Policy::POLICY_FILE', File.join(tmp, 'policy.json'))
|
|
@@ -64,16 +64,11 @@ describe PWN::AI::Agent::PromptBuilder do
|
|
|
64
64
|
end
|
|
65
65
|
end
|
|
66
66
|
|
|
67
|
-
describe '
|
|
68
|
-
it '
|
|
67
|
+
describe 'mid-turn lean prompt' do
|
|
68
|
+
it 'does not inject MEMORY/SKILLS/LEARNING/POLICY/EXTRO on a live goal' do
|
|
69
69
|
src = File.read(described_class.method(:build).source_location.first)
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
skills_at = src.index(skills_needle)
|
|
73
|
-
memory_at = src.index(memory_needle)
|
|
74
|
-
expect(skills_at).not_to be_nil
|
|
75
|
-
expect(memory_at).not_to be_nil
|
|
76
|
-
expect(skills_at).to be < memory_at
|
|
70
|
+
expect(src).to match(/KNOWN MISTAKES|mistakes_block/)
|
|
71
|
+
expect(src).to match(/stuck|expand_harness|lean/)
|
|
77
72
|
end
|
|
78
73
|
end
|
|
79
74
|
end
|
|
@@ -13,9 +13,29 @@ describe PWN::AI::Agent::Registry do
|
|
|
13
13
|
expect(help_response).to respond_to :help
|
|
14
14
|
end
|
|
15
15
|
|
|
16
|
-
it 'DEFAULT_PREFERENCE
|
|
16
|
+
it 'DEFAULT_PREFERENCE names only CORE_TOOLS in a recall-first fallback' do
|
|
17
17
|
expect(described_class::DEFAULT_PREFERENCE).to eq(
|
|
18
|
-
%w[memory_recall
|
|
18
|
+
%w[memory_recall pwn_eval shell mistakes_record mistakes_resolve learning_note_outcome memory_remember]
|
|
19
19
|
)
|
|
20
|
+
expect(described_class::DEFAULT_PREFERENCE - described_class::CORE_TOOLS).to be_empty
|
|
21
|
+
expect(described_class::DEFAULT_PREFERENCE).not_to include('sessions_view')
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
it 'preference_order defaults to shell / pwn_eval — there is no request type' do
|
|
25
|
+
expect(described_class.preference_order.first(2)).to eq(%w[shell pwn_eval])
|
|
26
|
+
expect(described_class.preference_order(kind: :question).first(2)).to eq(%w[shell pwn_eval])
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
it 'preference_order still honors explicit empty list and Env override' do
|
|
30
|
+
expect(described_class.preference_order(order: [])).to eq([])
|
|
31
|
+
expect(described_class.preference_order(preference: %w[shell])).to eq(%w[shell])
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
it 'definitions(core_only: true) ships CORE_TOOLS, not the full ~85 schema set' do
|
|
35
|
+
described_class.discover
|
|
36
|
+
names = described_class.definitions(core_only: true).map { |t| t.dig(:function, :name) }
|
|
37
|
+
expect(names).to match_array(described_class::CORE_TOOLS)
|
|
38
|
+
expect(names).not_to include('sessions_view')
|
|
39
|
+
expect(names.length).to be <= described_class::CORE_TOOLS.length
|
|
20
40
|
end
|
|
21
41
|
end
|
|
@@ -107,12 +107,11 @@ describe 'P0 signal hygiene (handler + inbox + policy-cold + calibrate)' do
|
|
|
107
107
|
expect(one[:n]).to eq(1)
|
|
108
108
|
end
|
|
109
109
|
|
|
110
|
-
it '
|
|
110
|
+
it 'treats review/opinion asks as goals with no request type' do
|
|
111
111
|
allow(PWN::Env).to receive(:dig).and_call_original
|
|
112
|
-
allow(PWN::Env).to receive(:dig).with(:ai, :agent, :request_kind_llm).and_return(false)
|
|
113
112
|
allow(PWN::Env).to receive(:dig).with(:ai, :agent, :task_summary_llm).and_return(false)
|
|
114
|
-
expect(PWN::AI::Agent::TaskSummarizer
|
|
115
|
-
expect(PWN::AI::Agent::TaskSummarizer.needs_task_breakdown?(request: 'suggest areas for improvement')).to eq(
|
|
116
|
-
expect(PWN::AI::Agent::TaskSummarizer.
|
|
113
|
+
expect(PWN::AI::Agent::TaskSummarizer).not_to respond_to(:request_kind)
|
|
114
|
+
expect(PWN::AI::Agent::TaskSummarizer.needs_task_breakdown?(request: 'suggest areas for improvement')).to eq(true)
|
|
115
|
+
expect(PWN::AI::Agent::TaskSummarizer.needs_task_breakdown?(request: 'fix all the issues you just described')).to eq(true)
|
|
117
116
|
end
|
|
118
117
|
end
|