pwn 0.5.680 → 0.5.682

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/lib/pwn/config.rb CHANGED
@@ -160,7 +160,7 @@ module PWN
160
160
  # LLM tangible-task decomposition for autonomous goals (default on).
161
161
  task_summary_llm: nil,
162
162
  tool_router: nil, # nil = auto (true when :active is local :ollama/:openwebui) — cuts ~11k→~3k schema tokens
163
- tool_preference: %w[memory_recall sessions_view pwn_eval shell mistakes_record mistakes_resolve learning_note_outcome memory_remember],
163
+ tool_preference: %w[memory_recall pwn_eval shell mistakes_record mistakes_resolve learning_note_outcome memory_remember],
164
164
  escalation_persona: 'escalator', # Swarm persona for frontier corrective hints when a local model is stuck
165
165
  # sample E3 verify_as_reward: true|false|nil(auto: ~10% local / always frontier when CLAIM_RX hits)
166
166
  verify_as_reward: nil,
data/lib/pwn/version.rb CHANGED
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module PWN
4
- VERSION = '0.5.680'
4
+ VERSION = '0.5.682'
5
5
  end
@@ -29,14 +29,16 @@ RSpec.describe 'PWN::AI::Agent::PromptBuilder', :aggregate_failures do
29
29
  PWN::Env[:ai] = { active: :anthropic, module_reflection: false, agent: @agent_cfg }
30
30
  end
31
31
 
32
- it 'contains every section header, session_id and PWN::VERSION' do
33
- prompt = builder.build(session_id: 'sess_abc')
34
- ['ENVIRONMENT', 'MEMORY', 'SKILLS', 'LEARNING', 'KNOWN MISTAKES', 'TOOL EFFECTIVENESS', 'EXTROSPECTION', 'TOOL USE'].each do |hdr|
32
+ it 'mid-turn autonomous prompt keeps request + CORE_TOOLS + known-fix, not the full harness' do
33
+ prompt = builder.build(session_id: 'sess_abc', request: 'Write hello into /tmp/x and verify it')
34
+ ['ENVIRONMENT', 'KNOWN MISTAKES', 'TOOL USE'].each do |hdr|
35
35
  expect(prompt).to include(hdr), "missing section: #{hdr}"
36
36
  end
37
37
  expect(prompt).to include('session_id : sess_abc')
38
38
  expect(prompt).to include(PWN::VERSION)
39
- expect(prompt).to include('prompt builder marker')
39
+ ['MEMORY', 'SKILLS', 'LEARNING', 'TOOL EFFECTIVENESS', 'EXTROSPECTION', 'POLICY'].each do |hdr|
40
+ expect(prompt).not_to include(hdr), "mid-turn prompt still injects #{hdr}"
41
+ end
40
42
  end
41
43
 
42
44
  it 'never leaks a raw nil or an unrendered #{...} interpolation' do
@@ -54,6 +54,44 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
54
54
  expect(src.scan('inject_task_focus!').length).to be >= 3
55
55
  end
56
56
 
57
+ it 'does not keep injecting English focus after the plan is covered' do
58
+ src = File.read(described_class.method(:inject_task_focus!).source_location.first)
59
+ focus = src[/private_class_method def self\.inject_task_focus!.*?private_class_method def self\.\w+/m]
60
+ focus ||= src
61
+ expect(focus).to match(/plan_open\?/)
62
+ end
63
+
64
+ it 'does not tell an open English plan to stop after 3 tools just because budget is hot' do
65
+ src = File.read(described_class.method(:run).source_location.first)
66
+ expect(src).to match(/budget_exhaustion_hot\?/)
67
+ expect(src).to match(/plan_open\?/)
68
+ # local ≤3-tool abort is only for a closed/short plan, not mid-goal.
69
+ expect(src).to match(/english_open|plan_open\?/)
70
+ end
71
+
72
+ it 'parks stale extra budget scars even while the host is hot' do
73
+ tmp = Dir.mktmpdir
74
+ stub_const('PWN::AI::Agent::Mistakes::MISTAKES_FILE', File.join(tmp, 'mistakes.json'))
75
+ PWN::AI::Agent::Mistakes.reset if PWN::AI::Agent::Mistakes.respond_to?(:reset)
76
+ a = PWN::AI::Agent::Mistakes.record(
77
+ tool: 'agent_loop',
78
+ error: '[pwn-ai] iteration budget exhausted A',
79
+ shape: 'budget_exhausted'
80
+ )
81
+ b = PWN::AI::Agent::Mistakes.record(
82
+ tool: 'agent_loop',
83
+ error: '[pwn-ai] iteration budget exhausted B',
84
+ shape: 'budget_exhausted'
85
+ )
86
+ described_class.send(:maybe_park_budget_scars!)
87
+ parked_n = [a, b].count do |m|
88
+ PWN::AI::Agent::Mistakes.find(signature: m[:signature])[:parked]
89
+ end
90
+ expect(parked_n).to be >= 1
91
+ ensure
92
+ FileUtils.remove_entry(tmp) if tmp && Dir.exist?(tmp)
93
+ end
94
+
57
95
  it 're-ranks Registry tools from English tangible tasks after plan (sole driver)' do
58
96
  src = File.read(described_class.method(:run).source_location.first)
59
97
  expect(src).to match(/TaskSummarizer\.relevance_query/)
@@ -82,7 +120,7 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
82
120
 
83
121
  last_plan = { plan: %w[identify implement verify], plan_idx: 2 }
84
122
  msgs_mut = msgs + [
85
- { role: 'tool', content: '{"success":true,"result":{"stdout":"0 offenses detected"}}' }
123
+ { role: 'tool', content: '{"success":true,"result":{"stdout":"patched loop.rb\n0 offenses detected"}}' }
86
124
  ]
87
125
  r_mut = loop_mod.send(
88
126
  :evidence_enough_to_finalize?,
@@ -98,10 +136,199 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
98
136
  )
99
137
  expect(r_bare).to be false
100
138
 
139
+ # plan_idx on last item is not enough if mutate/verify English tasks lack evidence
140
+ r_idx_only = loop_mod.send(
141
+ :evidence_enough_to_finalize?,
142
+ messages: msgs, turn_fails: {}, i: 5, max_iters: 40,
143
+ request: 'fix p17', plan_steps: 3, ts_state: last_plan
144
+ )
145
+ expect(r_idx_only).to be false
146
+
101
147
  # call site must pass ts_state
102
148
  src = File.read(loop_mod.method(:run).source_location.first)
103
149
  expect(src).to match(/evidence_enough_to_finalize\?\([\s\S]*?ts_state: ts_state/)
104
- expect(src).to match(/English-task gate/)
150
+ expect(src).to match(/original request|request is the completion/i)
151
+ end
152
+
153
+ it 'P17 does not inject finalize while implement/verify English tasks remain open' do
154
+ loop_mod = described_class
155
+ ls_find = [
156
+ { role: 'user', content: 'map then implement then verify' },
157
+ { role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"ls: loop.rb","exit":0}}' },
158
+ { role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"find: task_summarizer.rb","exit":0}}' },
159
+ { role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"resolved file listing","exit":0}}' }
160
+ ]
161
+ mid = {
162
+ plan: [
163
+ 'Map how tasks complete',
164
+ 'Implement the completion fix',
165
+ 'Verify full task completion end-to-end'
166
+ ],
167
+ plan_idx: 0
168
+ }
169
+
170
+ r_active = loop_mod.send(
171
+ :evidence_enough_to_finalize?,
172
+ messages: ls_find, turn_fails: {}, i: 4, max_iters: 40,
173
+ request: 'map then implement then verify', plan_steps: 3, ts_state: mid
174
+ )
175
+ expect(r_active).to be false
176
+
177
+ jumped = mid.merge(plan_idx: 2)
178
+ r_jumped = loop_mod.send(
179
+ :evidence_enough_to_finalize?,
180
+ messages: ls_find, turn_fails: {}, i: 4, max_iters: 40,
181
+ request: 'map then implement then verify', plan_steps: 3, ts_state: jumped
182
+ )
183
+ expect(r_jumped).to be false
184
+ expect(
185
+ PWN::AI::Agent::TaskSummarizer.plan_open?(state: jumped, messages: ls_find)
186
+ ).to eq true
187
+ left = PWN::AI::Agent::TaskSummarizer.unfinished_tasks(state: jumped, messages: ls_find)
188
+ expect(left.map { |t| t[:item] }.join(' ')).to match(/Implement|Verify/i)
189
+
190
+ src = File.read(loop_mod.method(:run).source_location.first)
191
+ expect(src).to match(/Write the complete final answer now/)
192
+ expect(src).to match(/evidence_enough_to_finalize\?/)
193
+ expect(src).to match(/Do NOT call more tools/)
194
+ end
195
+
196
+ it 'P17 can early-final when English tasks are covered even if plan_idx is still 0' do
197
+ loop_mod = described_class
198
+ done = {
199
+ plan: [
200
+ 'Determine the local hostname',
201
+ 'Present the result and report completion'
202
+ ],
203
+ plan_idx: 0
204
+ }
205
+ msgs = [
206
+ { role: 'user', content: 'what is my hostname?' },
207
+ { role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"kali-box","exit":0}}' },
208
+ { role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"kali-box","exit":0}}' }
209
+ ]
210
+ expect(
211
+ PWN::AI::Agent::TaskSummarizer.plan_open?(state: done, messages: msgs)
212
+ ).to eq false
213
+ r = loop_mod.send(
214
+ :evidence_enough_to_finalize?,
215
+ messages: msgs, turn_fails: {}, i: 4, max_iters: 40,
216
+ request: 'what is my hostname?', plan_steps: 2, ts_state: done
217
+ )
218
+ expect(r).to be true
219
+ end
220
+
221
+ it 'P17 can early-final a finished request even while advisory English tasks remain' do
222
+ loop_mod = described_class
223
+ leftover_plan = {
224
+ plan: [
225
+ 'Determine the local hostname',
226
+ 'run rspec to verify'
227
+ ],
228
+ plan_idx: 0
229
+ }
230
+ msgs = [
231
+ { role: 'user', content: 'what is my hostname?' },
232
+ { role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"kali-box","exit":0}}' },
233
+ { role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"kali-box","exit":0}}' }
234
+ ]
235
+ expect(
236
+ PWN::AI::Agent::TaskSummarizer.plan_open?(state: leftover_plan, messages: msgs)
237
+ ).to eq true
238
+ r = loop_mod.send(
239
+ :evidence_enough_to_finalize?,
240
+ messages: msgs, turn_fails: {}, i: 4, max_iters: 40,
241
+ request: 'what is my hostname?', plan_steps: 2, ts_state: leftover_plan
242
+ )
243
+ expect(r).to be true
244
+ end
245
+
246
+ it 'open_plan_blocks_final? does not govern completion — the original request does' do
247
+ src = File.read(described_class.method(:run).source_location.first)
248
+ expect(src).to match(/original request is the completion signal/i)
249
+ expect(src).not_to match(/if open_plan_blocks_final\?/)
250
+ end
251
+
252
+ it 'request_unsatisfied? uses request-shaped evidence, not just mutation regex' do
253
+ ls_only = [
254
+ { role: 'user', content: 'write complete into /tmp/x.txt and verify it' },
255
+ {
256
+ role: 'assistant',
257
+ tool_calls: [
258
+ { function: { name: 'shell', arguments: '{"command":"ls /tmp"}' } }
259
+ ]
260
+ },
261
+ { role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"x","exit":0}}' }
262
+ ]
263
+ expect(
264
+ described_class.send(
265
+ :request_unsatisfied?,
266
+ request: 'Write complete into /tmp/x.txt and verify it reads complete',
267
+ messages: ls_only,
268
+ last_iter: false
269
+ )
270
+ ).to eq true
271
+ # Path-shaped write without the old printf/sed regex still satisfies the request.
272
+ written = ls_only + [
273
+ {
274
+ role: 'assistant',
275
+ tool_calls: [
276
+ { function: { name: 'shell', arguments: '{"command":"python3 -c \\"open(\'/tmp/x.txt\',\'w\').write(\'complete\')\\""}' } }
277
+ ]
278
+ },
279
+ {
280
+ role: 'tool',
281
+ name: 'shell',
282
+ content: '{"success":true,"result":{"stdout":"","stderr":"","exit":0}}'
283
+ }
284
+ ]
285
+ expect(
286
+ described_class.send(
287
+ :request_unsatisfied?,
288
+ request: 'Write complete into /tmp/x.txt and verify it reads complete',
289
+ messages: written,
290
+ last_iter: false
291
+ )
292
+ ).to eq false
293
+ expect(
294
+ described_class.send(
295
+ :request_unsatisfied?,
296
+ request: 'what is my hostname?',
297
+ messages: [
298
+ { role: 'tool', name: 'shell', content: '{"success":true,"result":{"stdout":"kali","exit":0}}' }
299
+ ],
300
+ last_iter: false
301
+ )
302
+ ).to eq false
303
+ end
304
+
305
+ it 'does not bounce world-knowledge asks that need no host work' do
306
+ expect(described_class.needs_host_work?(request: 'what color is a cherry')).to eq false
307
+ expect(described_class.world_knowledge?(request: 'what color is a cherry')).to eq true
308
+ expect(
309
+ described_class.send(
310
+ :request_unsatisfied?,
311
+ request: 'what color is a cherry',
312
+ messages: [{ role: 'assistant', content: 'Red.', tool_calls: [] }],
313
+ last_iter: false
314
+ )
315
+ ).to eq false
316
+ expect(described_class.needs_host_work?(request: 'Write hello into /tmp/x.txt')).to eq true
317
+ end
318
+
319
+ it 'last-iter strips tools only on the true last slot; leftover English tasks do not force required' do
320
+ src = File.read(described_class.method(:run).source_location.first)
321
+ expect(src).to match(/text_only_iters = 1/)
322
+ expect(src).not_to match(/english_open && hot/)
323
+ expect(src).to match(/has_tool_result \|\| !need_tools \? 'auto' : 'required'/)
324
+ expect(src).not_to match(/still_acting/)
325
+ end
326
+
327
+ it 'Loop.run default tool pool is CORE_TOOLS, not the full registry' do
328
+ src = File.read(described_class.method(:run).source_location.first)
329
+ expect(src).to match(/fetch\(:core_only,\s*true\)/)
330
+ expect(src).to match(/core_only:/)
331
+ expect(src).to match(/CORE_TOOLS/)
105
332
  end
106
333
 
107
334
  it 'does not put TaskSummarizer into Reward credit paths' do
@@ -120,9 +347,8 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
120
347
  expect(src).to match(/tool_calls_from_text/)
121
348
  expect(src).to match(/_text_tool_coerced/)
122
349
  expect(src).to match(/tool_choice/)
123
- # stay required while monologue / unfinished; only auto when settled
124
- expect(src).to match(/still_acting/)
125
- expect(src).to match(/has_tool_result && !still_acting/)
350
+ # After the first tool result, auto. English leftovers do not keep required.
351
+ expect(src).to match(/has_tool_result \|\| !need_tools \? 'auto' : 'required'/)
126
352
  # normalize_llm must promote text tool forms
127
353
  expect(src).to match(/normalize_llm[\s\S]*tool_calls_from_text/m)
128
354
  expect(src).to match(/MONOLOGUE_TOOL_INTENT_RX/)
@@ -154,26 +380,16 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
154
380
  expect(described_class.request_intent(request: 'refactor Loop.run and run rubocop')).to eq(:act)
155
381
  end
156
382
 
157
- it 'request_kind maps statement | question | autonomous_goal' do
158
- expect(described_class).to respond_to :request_kind
159
- expect(described_class.request_kind(request: 'FYI the build is green.')).to eq(:statement)
160
- expect(described_class.request_kind(request: 'how to do a ping sweep with hping3?')).to eq(:question)
161
- expect(described_class.request_kind(request: 'what did I just say?')).to eq(:question)
162
- expect(described_class.request_kind(request: 'hi')).to eq(:statement)
163
- expect(described_class.request_kind(request: 'refactor Loop.run and run rubocop')).to eq(:autonomous_goal)
164
- expect(described_class.request_kind(request: 'using hping3 what live hosts can you find in this subnet?')).to eq(:autonomous_goal)
165
- # Live host-fact Qs need tools — not text-only question path.
166
- expect(described_class.request_kind(request: 'what is my hostname?')).to eq(:autonomous_goal)
167
- expect(described_class.request_kind(request: 'excellent - what is my hostname?')).to eq(:autonomous_goal)
383
+ it 'does not classify request types — Loop has no request_kind' do
384
+ expect(described_class).not_to respond_to :request_kind
168
385
  end
169
386
 
170
- it 'run short-circuits statement/question but not host-evidence goals' do
387
+ it 'run does not short-circuit statement/question — they take the full tool loop' do
171
388
  src = File.read(described_class.method(:run).source_location.first)
172
- expect(src).to match(/kind\.to_sym == :statement/)
173
- expect(src).to match(/kind\.to_sym == :question/)
174
- expect(src).to match(/needs_breakdown/)
175
- expect(src).to match(/answer_statement/)
176
- expect(src).to match(/answer_question/)
389
+ expect(src).not_to match(/kind\.to_sym == :statement/)
390
+ expect(src).not_to match(/kind\.to_sym == :question/)
391
+ expect(src).not_to match(/Request type/)
392
+ expect(src).to match(/Every request gets a task compass/)
177
393
  end
178
394
 
179
395
  it 'classifies pure prior-turn recall and vague memory cues as :recall' do
@@ -214,13 +430,10 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
214
430
  it 'run short-circuits how-to without plan_first or tools' do
215
431
  src = File.read(described_class.method(:run).source_location.first)
216
432
  expect(src).to match(/request_intent/)
217
- expect(src).to match(/request_kind/)
433
+ expect(src).not_to match(/request_kind/)
218
434
  expect(src).to match(/answer_howto/)
219
- expect(src).to match(/answer_statement/)
220
- expect(src).to match(/answer_question/)
221
435
  expect(src).to match(/intent == :howto/)
222
436
  expect(src).to match(/skip_plan/)
223
- expect(src).to match(/needs_breakdown/)
224
437
  end
225
438
 
226
439
  it 'answer_howto does not register shell tool use in source path' do
@@ -235,8 +448,7 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
235
448
  expect(src).to match(/answer_recall/)
236
449
  expect(src).to match(/intent == :recall.*?return answer_recall/m)
237
450
  expect(src).to match(/skip_plan/)
238
- expect(src).to match(/request_kind/)
239
- expect(src).to match(/needs_breakdown|autonomous_goal|statement|question/)
451
+ expect(src).not_to match(/request_kind/)
240
452
  end
241
453
 
242
454
  it 'run short-circuits greetings via answer_greeting before plan_first' do
@@ -245,8 +457,7 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
245
457
  expect(src).to match(/answer_greeting/)
246
458
  expect(src).to match(/intent == :greeting.*?return answer_greeting/m)
247
459
  expect(src).to match(/skip_plan/)
248
- expect(src).to match(/request_kind/)
249
- expect(src).to match(/needs_breakdown|autonomous_goal|statement|question/)
460
+ expect(src).not_to match(/request_kind/)
250
461
  end
251
462
 
252
463
  it 'answer_greeting returns fixed ack without weather echo or tools' do
@@ -474,16 +685,23 @@ describe PWN::AI::Agent::Loop do # rubocop:disable Metrics/BlockLength
474
685
  end
475
686
  end
476
687
 
688
+ it 'Loop.run finishes the original request; English tasks are advisory only' do
689
+ src = File.read(described_class.method(:run).source_location.first)
690
+ expect(src).to match(/original request is the completion signal/i)
691
+ expect(src).not_to match(/do NOT finalize/i)
692
+ expect(src).not_to match(/if open_plan_blocks_final\?/)
693
+ end
694
+
477
695
  it 'Loop.run marks the Hermes user-path so TurnFinalizer can defer' do
478
696
  src = File.read(described_class.method(:run).source_location.first)
479
697
  expect(src).to match(/TurnFinalizer\.enter_user_path!/)
480
698
  expect(src).to match(/TurnFinalizer\.leave_user_path!/)
481
699
  end
482
700
 
483
- it 'should_auto_introspect skips cheap greeting/question/statement answers' do
701
+ it 'should_auto_introspect skips cheap greeting/howto/recall answers' do
484
702
  src = File.read(described_class.method(:run).source_location.first)
485
703
  expect(src).to include('return false if %i[greeting howto recall].include?(intent)')
486
- expect(src).to include('return false if kind == :statement')
487
- expect(src).to include('return false if kind == :question && fails.zero?')
704
+ expect(src).not_to include('return false if kind == :statement')
705
+ expect(src).not_to include('return false if kind == :question && fails.zero?')
488
706
  end
489
707
  end # rubocop:enable Metrics/BlockLength
@@ -170,4 +170,18 @@ describe PWN::AI::Agent::Mistakes do
170
170
  expect(out[:extinguished]).to be >= 1
171
171
  expect(described_class.find(signature: m[:signature])[:resolved]).to be true
172
172
  end
173
+
174
+ it 'to_context downranks budget-exhaustion scars on unrelated requests' do
175
+ stub_const('PWN::AI::Agent::Mistakes::MISTAKES_FILE', File.join(Dir.mktmpdir, 'mistakes.json'))
176
+ described_class.reset if described_class.respond_to?(:reset)
177
+ described_class.record(
178
+ tool: 'agent_loop',
179
+ error: '[pwn-ai] iteration budget exhausted',
180
+ shape: 'budget_exhausted'
181
+ )
182
+ described_class.record(tool: 'shell', error: 'nmpa: command not found unique-host')
183
+ ctx = described_class.to_context(request: 'what is my hostname?', limit: 2)
184
+ expect(ctx).to include('shell')
185
+ expect(ctx).not_to match(/iteration budget exhausted/)
186
+ end
173
187
  end
@@ -80,7 +80,11 @@ describe PWN::AI::Agent::Policy do
80
80
  kind: :autonomous_goal,
81
81
  request: 'fix the reward judge',
82
82
  engine: :grok,
83
- ts_state: { plan: %w[inspect tighten verify], plan_idx: 2 },
83
+ ts_state: {
84
+ plan: %w[inspect tighten verify],
85
+ plan_idx: 2,
86
+ evidence_blob: 'inspected reward.rb patched judge 12 examples, 0 failures'
87
+ },
84
88
  final: 'The cheap ORM now grades the last tools and usable result.',
85
89
  score: 0.82
86
90
  )
@@ -91,6 +95,53 @@ describe PWN::AI::Agent::Policy do
91
95
  expect(open_s).not_to eq(done_s)
92
96
  end
93
97
 
98
+ it 'quality bin is not high while English tasks remain even if plan_idx is last' do
99
+ last_but_open = described_class.state(
100
+ kind: :autonomous_goal,
101
+ request: 'fix the reward judge',
102
+ engine: :grok,
103
+ ts_state: { plan: %w[inspect tighten verify], plan_idx: 2 }
104
+ )
105
+ expect(last_but_open).to include('|p')
106
+ expect(last_but_open).not_to include('|ph')
107
+ end
108
+
109
+ it 'observe_step credits English-task progress above bare tool-ok hygiene' do
110
+ tmp = Dir.mktmpdir
111
+ stub_const('PWN::AI::Agent::Policy::POLICY_FILE', File.join(tmp, 'policy.json'))
112
+ stub_const('PWN::AI::Agent::Policy::TRAJECTORY_FILE', File.join(tmp, 'policy_traj.jsonl'))
113
+ described_class.reset
114
+ allow(described_class).to receive(:enabled?).and_return(true)
115
+ PWN::Env[:ai] ||= {}
116
+ PWN::Env[:ai][:agent] ||= {}
117
+ PWN::Env[:ai][:agent][:policy] = true
118
+
119
+ ts = {
120
+ plan: ['locate the source', 'fix the truncation bug', 'run rspec to verify'],
121
+ plan_idx: 0,
122
+ evidence_blob: ''
123
+ }
124
+ described_class.begin_episode(
125
+ session_id: 'en_credit',
126
+ request: 'locate then fix',
127
+ kind: :autonomous_goal,
128
+ engine: :grok,
129
+ ts_state: ts
130
+ )
131
+ grind = described_class.observe_step(
132
+ action: 'shell', ok: true, session_id: 'en_credit', ts_state: ts
133
+ )
134
+ ts[:plan_idx] = 1
135
+ ts[:evidence_blob] = 'rg hit locate the source lib/task_summarizer.rb patched later'
136
+ advance = described_class.observe_step(
137
+ action: 'shell', ok: true, session_id: 'en_credit', ts_state: ts
138
+ )
139
+ expect(advance[:reward].to_f).to be > grind[:reward].to_f
140
+ ensure
141
+ described_class.reset
142
+ FileUtils.remove_entry(tmp) if tmp && Dir.exist?(tmp)
143
+ end
144
+
94
145
  it 'warmup! meets the episode budget so greedy suggestions appear' do
95
146
  tmp = Dir.mktmpdir
96
147
  stub_const('PWN::AI::Agent::Policy::POLICY_FILE', File.join(tmp, 'policy.json'))
@@ -64,16 +64,11 @@ describe PWN::AI::Agent::PromptBuilder do
64
64
  end
65
65
  end
66
66
 
67
- describe 'Hermes skill-index prefix' do
68
- it 'emits SKILLS before MEMORY so prompt-cache can pin the index' do
67
+ describe 'mid-turn lean prompt' do
68
+ it 'does not inject MEMORY/SKILLS/LEARNING/POLICY/EXTRO on a live goal' do
69
69
  src = File.read(described_class.method(:build).source_location.first)
70
- skills_needle = ['#', '{skills_block}'].join
71
- memory_needle = ['#', '{memory_block'].join
72
- skills_at = src.index(skills_needle)
73
- memory_at = src.index(memory_needle)
74
- expect(skills_at).not_to be_nil
75
- expect(memory_at).not_to be_nil
76
- expect(skills_at).to be < memory_at
70
+ expect(src).to match(/KNOWN MISTAKES|mistakes_block/)
71
+ expect(src).to match(/stuck|expand_harness|lean/)
77
72
  end
78
73
  end
79
74
  end
@@ -13,9 +13,29 @@ describe PWN::AI::Agent::Registry do
13
13
  expect(help_response).to respond_to :help
14
14
  end
15
15
 
16
- it 'DEFAULT_PREFERENCE is memory_recall, sessions_view, pwn_eval, shell, mistakes_record, mistakes_resolve, learning_note_outcome, memory_remember' do
16
+ it 'DEFAULT_PREFERENCE names only CORE_TOOLS in a recall-first fallback' do
17
17
  expect(described_class::DEFAULT_PREFERENCE).to eq(
18
- %w[memory_recall sessions_view pwn_eval shell mistakes_record mistakes_resolve learning_note_outcome memory_remember]
18
+ %w[memory_recall pwn_eval shell mistakes_record mistakes_resolve learning_note_outcome memory_remember]
19
19
  )
20
+ expect(described_class::DEFAULT_PREFERENCE - described_class::CORE_TOOLS).to be_empty
21
+ expect(described_class::DEFAULT_PREFERENCE).not_to include('sessions_view')
22
+ end
23
+
24
+ it 'preference_order defaults to shell / pwn_eval — there is no request type' do
25
+ expect(described_class.preference_order.first(2)).to eq(%w[shell pwn_eval])
26
+ expect(described_class.preference_order(kind: :question).first(2)).to eq(%w[shell pwn_eval])
27
+ end
28
+
29
+ it 'preference_order still honors explicit empty list and Env override' do
30
+ expect(described_class.preference_order(order: [])).to eq([])
31
+ expect(described_class.preference_order(preference: %w[shell])).to eq(%w[shell])
32
+ end
33
+
34
+ it 'definitions(core_only: true) ships CORE_TOOLS, not the full ~85 schema set' do
35
+ described_class.discover
36
+ names = described_class.definitions(core_only: true).map { |t| t.dig(:function, :name) }
37
+ expect(names).to match_array(described_class::CORE_TOOLS)
38
+ expect(names).not_to include('sessions_view')
39
+ expect(names.length).to be <= described_class::CORE_TOOLS.length
20
40
  end
21
41
  end
@@ -107,12 +107,11 @@ describe 'P0 signal hygiene (handler + inbox + policy-cold + calibrate)' do
107
107
  expect(one[:n]).to eq(1)
108
108
  end
109
109
 
110
- it 'classifies review/opinion asks as questions' do
110
+ it 'treats review/opinion asks as goals with no request type' do
111
111
  allow(PWN::Env).to receive(:dig).and_call_original
112
- allow(PWN::Env).to receive(:dig).with(:ai, :agent, :request_kind_llm).and_return(false)
113
112
  allow(PWN::Env).to receive(:dig).with(:ai, :agent, :task_summary_llm).and_return(false)
114
- expect(PWN::AI::Agent::TaskSummarizer.request_kind(request: 'suggest areas for improvement')).to eq(:question)
115
- expect(PWN::AI::Agent::TaskSummarizer.needs_task_breakdown?(request: 'suggest areas for improvement')).to eq(false)
116
- expect(PWN::AI::Agent::TaskSummarizer.request_kind(request: 'fix all the issues you just described')).to eq(:autonomous_goal)
113
+ expect(PWN::AI::Agent::TaskSummarizer).not_to respond_to(:request_kind)
114
+ expect(PWN::AI::Agent::TaskSummarizer.needs_task_breakdown?(request: 'suggest areas for improvement')).to eq(true)
115
+ expect(PWN::AI::Agent::TaskSummarizer.needs_task_breakdown?(request: 'fix all the issues you just described')).to eq(true)
117
116
  end
118
117
  end