pwn 0.5.680 → 0.5.683

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. checksums.yaml +4 -4
  2. data/.gitignore +1 -0
  3. data/documentation/AI-Integration.md +1 -1
  4. data/documentation/Agent-Tool-Registry.md +11 -7
  5. data/documentation/Configuration.md +6 -7
  6. data/documentation/How-PWN-Works.md +2 -2
  7. data/documentation/Reinforcement-Learning.md +1 -1
  8. data/documentation/diagrams/agent-tool-registry.svg +1 -1
  9. data/documentation/diagrams/dot/agent-tool-registry.dot +1 -1
  10. data/documentation/diagrams/dot/task-summarizer.dot +3 -3
  11. data/documentation/pwn-ai-Agent.md +16 -35
  12. data/lib/pwn/ai/agent/loop.rb +265 -192
  13. data/lib/pwn/ai/agent/mistakes.rb +17 -0
  14. data/lib/pwn/ai/agent/policy.rb +53 -4
  15. data/lib/pwn/ai/agent/prompt_builder.rb +46 -19
  16. data/lib/pwn/ai/agent/registry.rb +18 -13
  17. data/lib/pwn/ai/agent/task_summarizer.rb +374 -503
  18. data/lib/pwn/ai/agent/tools/sessions.rb +32 -0
  19. data/lib/pwn/ai/agent/tools/skills.rb +56 -0
  20. data/lib/pwn/ai/anthropic.rb +0 -1
  21. data/lib/pwn/ai/gemini.rb +0 -1
  22. data/lib/pwn/ai/grok.rb +0 -1
  23. data/lib/pwn/ai/ollama.rb +0 -1
  24. data/lib/pwn/ai/open_ai.rb +0 -1
  25. data/lib/pwn/ai/open_web_ui.rb +0 -1
  26. data/lib/pwn/config.rb +1 -1
  27. data/lib/pwn/plugins/repl.rb +37 -0
  28. data/lib/pwn/plugins/tty_spinner.rb +55 -6
  29. data/lib/pwn/sessions.rb +82 -0
  30. data/lib/pwn/version.rb +1 -1
  31. data/spec/integration/prompt_builder_spec.rb +6 -4
  32. data/spec/lib/pwn/ai/agent/loop_spec.rb +314 -34
  33. data/spec/lib/pwn/ai/agent/mistakes_spec.rb +14 -0
  34. data/spec/lib/pwn/ai/agent/policy_spec.rb +52 -1
  35. data/spec/lib/pwn/ai/agent/prompt_builder_spec.rb +8 -9
  36. data/spec/lib/pwn/ai/agent/registry_spec.rb +30 -3
  37. data/spec/lib/pwn/ai/agent/signal_hygiene_spec.rb +4 -5
  38. data/spec/lib/pwn/ai/agent/task_summarizer_spec.rb +321 -90
  39. data/spec/lib/pwn/ai/agent/tools/sessions_spec.rb +5 -0
  40. data/spec/lib/pwn/ai/agent/tools/skills_spec.rb +16 -0
  41. data/spec/lib/pwn/ai/red_team/test_case_engine_spec.rb +20 -0
  42. data/spec/lib/pwn/plugins/repl_spec.rb +9 -0
  43. data/spec/lib/pwn/plugins/tty_spinner_spec.rb +30 -0
  44. data/spec/lib/pwn/sessions_spec.rb +29 -0
  45. data/spec/spec_helper.rb +6 -0
  46. data/third_party/pwn_rdoc.jsonl +30 -9
  47. metadata +1 -1
@@ -34,6 +34,12 @@ module PWN
34
34
  # PromptBuilder.mistakes_block re-injects the top open mistakes and
35
35
  # top known fixes into the system prompt of every future turn.
36
36
  #
37
+ # COMPLETION
38
+ # ----------
39
+ # The original request is the completion signal. TaskSummarizer and
40
+ # Policy are advisory (compass / rank). Loop keeps calling CORE_TOOLS
41
+ # until that request is done or truly blocked, then stops.
42
+ #
37
43
  # LOCAL-MODEL SCAFFOLDING
38
44
  # -----------------------
39
45
  # When the active engine is :ollama (or the corresponding :agent flags
@@ -160,22 +166,31 @@ module PWN
160
166
 
161
167
  private_class_method def self.maybe_park_budget_scars!
162
168
  return unless defined?(Mistakes)
163
- return if budget_exhaustion_hot?
164
169
  return unless Mistakes.respond_to?(:park)
165
170
 
166
- top = Mistakes.top(limit: 12, unresolved_only: true)
171
+ top = Mistakes.top(limit: 24, unresolved_only: true)
167
172
  now = Time.now
168
- top.each do |mistake|
169
- next unless budget_hit?(mistake: mistake)
170
- next if mistake[:parked]
171
-
173
+ budget = top.select { |mistake| budget_hit?(mistake: mistake) && !mistake[:parked] }
174
+ budget.each do |mistake|
172
175
  stamp = mistake_ts(mistake: mistake)
173
- # cool detector + scar older than PARK_COOL_SECS → park
174
176
  next if stamp && (now - stamp) <= PARK_COOL_SECS
175
177
 
176
178
  Mistakes.park(
177
179
  signature: mistake[:signature].to_s,
178
- reason: 'p17 rate-cool: outside PARK_COOL_SECS while hot?=false'
180
+ reason: 'p17 rate-cool: outside PARK_COOL_SECS'
181
+ )
182
+ end
183
+ live = budget.reject do |mistake|
184
+ row = Mistakes.find(signature: mistake[:signature].to_s)
185
+ row.nil? || row[:parked]
186
+ end
187
+ return if live.length <= 1
188
+
189
+ # Never let 2+ budget scars latch hot forever. Keep only the newest.
190
+ live.sort_by { |mistake| mistake_ts(mistake: mistake) || Time.at(0) }[0...-1].each do |mistake|
191
+ Mistakes.park(
192
+ signature: mistake[:signature].to_s,
193
+ reason: 'p17 keep-newest budget scar; extras parked so tomorrow is not hot'
179
194
  )
180
195
  end
181
196
  rescue StandardError
@@ -215,10 +230,9 @@ module PWN
215
230
  end
216
231
 
217
232
  # P17 — evidence-enough early final: latest tool rounds already answer
218
- # the ask → force synthesis instead of burning iters into text-only tail.
219
- # Must NOT fire on routine tool JSON {"success":true} while a multi-step
220
- # English plan still has open tasks — that blocks legitimate completion
221
- # (mid-fix "write the complete final answer now" thrash).
233
+ # the original request → force synthesis. English tasks are an advisory
234
+ # compass only — an open verify tail must not block a finished ask.
235
+ # The original request is the completion signal.
222
236
  private_class_method def self.evidence_enough_to_finalize?(opts = {})
223
237
  messages = Array(opts[:messages])
224
238
  turn_fails = opts[:turn_fails] || {}
@@ -226,30 +240,15 @@ module PWN
226
240
  max_i = opts[:max_iters].to_i
227
241
  request = opts[:request].to_s
228
242
  return false if max_i <= 0 || iter < 2
229
- # Need runway before the text-only strip, and no thrash.
230
243
  return false if turn_fails['empty_final'].to_i.positive?
231
244
  return false if turn_fails['incomplete_final'].to_i > 1
232
245
 
233
246
  fail_n = turn_fails.values.sum
234
247
  return false if fail_n >= 3
235
248
 
236
- # English-task gate: multi-step plans only early-final on/after the
237
- # last tangible task. plan_idx is 0-based; open work => not enough.
238
- ts_state = opts[:ts_state]
239
- if ts_state.is_a?(Hash)
240
- plan = Array(ts_state[:plan])
241
- if plan.length >= 2
242
- idx = ts_state[:plan_idx].to_i
243
- return false if idx < (plan.length - 1)
244
- end
245
- end
246
-
247
249
  tools_ok = messages.select { |msg| msg[:role].to_s == 'tool' }
248
250
  return false if tools_ok.size < 2
249
251
 
250
- # Last two tool payloads should look like successful evidence, not errors.
251
- # Agent tool wrappers always emit "success":true on ok — that alone is
252
- # NOT proof the user goal is done (do not match bare success JSON).
253
252
  last2 = tools_ok.last(2)
254
253
  return false if last2.any? do |msg|
255
254
  content = msg[:content].to_s
@@ -257,20 +256,19 @@ module PWN
257
256
  !content.match?(/"success"\s*:\s*true/i)
258
257
  end
259
258
 
260
- # Prefer when plan was short / we already spent half the budget usefully.
261
259
  plan_steps = opts[:plan_steps].to_i
262
260
  short_plan = plan_steps.positive? && plan_steps <= 3
263
261
  deep_enough = tools_ok.size >= 3 || (short_plan && tools_ok.size >= plan_steps)
264
262
  return false unless deep_enough
265
263
 
266
264
  recent_txt = last2.map { |msg| msg[:content].to_s[0, 500] }.join(' ')
267
- # Goal-shaped completion only — write/patch/verify, not shell success wrappers.
268
265
  mutation_done = recent_txt.match?(
269
- /syntax ok|wrote |patched|resolved|File\.write|ruby -c|0 offenses|examples?,\s*0 failures/i
266
+ /syntax ok|wrote |patched|File\.write|ruby -c|0 offenses|examples?,\s*0 failures/i
270
267
  )
271
268
  return true if mutation_done
269
+ return true if request_path_evidenced?(request: request, blob: recent_txt)
272
270
  return true if short_plan && tools_ok.size >= plan_steps && fail_n.zero? &&
273
- request.match?(/\b(what|who|when|where|which|how many|status|list|show|print|uname|cwd|version)\b/i)
271
+ request.match?(/\b(what|who|when|where|which|how many|status|list|show|print|uname|cwd|version|hostname)\b/i)
274
272
 
275
273
  false
276
274
  rescue StandardError
@@ -312,6 +310,107 @@ module PWN
312
310
  )
313
311
  /ix
314
312
 
313
+ ACT_REQUEST_RX = /
314
+ \b(write|create|implement|fix|patch|replace|refactor|overwrite|
315
+ add (?:a |the )?|update|install|delete|remove|rename)\b
316
+ /ix
317
+ MUTATION_EVIDENCE_RX = /
318
+ printf\s|tee\s|sed\s+-i|ruby\s+-i|>\s|>>\s|file\.write|binwrite|
319
+ patched|wrote\s|syntax\sok|0\s+offenses|examples?,\s*0\s+failures
320
+ /ix
321
+ LOOKUP_REQUEST_RX = /
322
+ \b(what\s+is\s+my|hostname|uname|cwd|whoami|status|version|how\s+many)\b
323
+ /ix
324
+ # Real filesystem paths only — not https://host.tld (that was matching //host.tld).
325
+ HOST_PATH_RX = %r{(?:(?<![.:/])/(?!/)|\./)[\w./-]+\.\w+}
326
+ BROWSER_REQUEST_RX = /
327
+ TransparentBrowser|browser_obj|\bdevtools\b|
328
+ \b(navigate|dump_links|headless_?chrome|watir)\b
329
+ /ix
330
+ BROWSER_EVIDENCE_RX = %r{https?://|dump_links|TransparentBrowser|\.close\b|browser_obj}i
331
+
332
+ # True only when the ask needs a live host/file/browser effect. World-knowledge
333
+ # questions ("what color is a cherry") do not.
334
+ public_class_method def self.needs_host_work?(opts = {})
335
+ request = opts[:request].to_s
336
+ return false if request.strip.empty?
337
+ return true if request.match?(ACT_REQUEST_RX)
338
+ return true if request.match?(LOOKUP_REQUEST_RX)
339
+ return true if request.match?(HOST_PATH_RX)
340
+ return true if request.match?(BROWSER_REQUEST_RX)
341
+
342
+ false
343
+ rescue StandardError
344
+ false
345
+ end
346
+
347
+ # Short world-knowledge asks (no host/file work). Skip the planner LLM
348
+ # and do not bounce a text-only answer.
349
+ public_class_method def self.world_knowledge?(opts = {})
350
+ request = opts[:request].to_s.strip
351
+ return false if request.empty?
352
+ return false if needs_host_work?(request: request)
353
+ return false if request.length > 120
354
+ return false if request.match?(%r{\b(this\s+(?:host|machine|box|system|subnet|file|repo)|/opt/|implement|scan|hosts?)\b}i)
355
+
356
+ request.match?(/\A(?:what|why|who|when|where|which|how)\b/i)
357
+ rescue StandardError
358
+ false
359
+ end
360
+
361
+ # True when a text-only reply cannot yet be the original request.
362
+ # Distinct from English-task leftovers (advisory) and polite handoffs.
363
+ # Never bounce world-knowledge / no-host-work asks.
364
+ private_class_method def self.request_unsatisfied?(opts = {})
365
+ return false if opts[:last_iter]
366
+ return false if world_knowledge?(request: opts[:request])
367
+ return false unless needs_host_work?(request: opts[:request])
368
+
369
+ request = opts[:request].to_s
370
+ messages = Array(opts[:messages])
371
+ tools = messages.select { |msg| msg.is_a?(Hash) && msg[:role].to_s == 'tool' }
372
+ blob = +''
373
+ tools.each do |msg|
374
+ blob << msg[:name].to_s << ' ' << msg[:content].to_s << "\n"
375
+ end
376
+ messages.each do |msg|
377
+ next unless msg.is_a?(Hash) && msg[:role].to_s == 'assistant'
378
+
379
+ Array(msg[:tool_calls]).each do |tc|
380
+ blob << tc.dig(:function, :name).to_s << ' '
381
+ blob << tc.dig(:function, :arguments).to_s << "\n"
382
+ end
383
+ end
384
+
385
+ return false if request.match?(LOOKUP_REQUEST_RX) && tools.any? && blob.length >= 20
386
+ return true if tools.empty?
387
+
388
+ if request.match?(ACT_REQUEST_RX) || request.match?(HOST_PATH_RX)
389
+ return false if blob.match?(MUTATION_EVIDENCE_RX)
390
+ return false if request_path_evidenced?(request: request, blob: blob)
391
+
392
+ return true
393
+ end
394
+ return false if request.match?(BROWSER_REQUEST_RX) && blob.match?(BROWSER_EVIDENCE_RX)
395
+
396
+ blob.length < 20
397
+ rescue StandardError
398
+ false
399
+ end
400
+
401
+ private_class_method def self.request_path_evidenced?(opts = {})
402
+ request = opts[:request].to_s
403
+ blob = opts[:blob].to_s
404
+ paths = request.scan(%r{(?:/|\./)[\w./-]+\.\w+})
405
+ return false if paths.empty?
406
+
407
+ paths.any? do |path|
408
+ blob.include?(path) && blob.match?(
409
+ /open\(|File\.(?:write|open|binwrite)|write\(|puts\s|print\s|>\s|>>\s|tee\s|sed\s+-i|ruby\s+-i|patched|wrote/i
410
+ )
411
+ end
412
+ end
413
+
315
414
  private_class_method def self.incomplete_final?(opts = {})
316
415
  text = opts[:text].to_s
317
416
  return false if text.strip.empty?
@@ -999,18 +1098,11 @@ module PWN
999
1098
  if env_tc && !env_tc.to_s.empty?
1000
1099
  cwt_opts[:tool_choice] = env_tc
1001
1100
  else
1002
- # Weak chat templates (TEMPLATE {{ .Prompt }} on abliterated
1003
- # Gemma etc.) ignore tools: under tool_choice=auto and dump
1004
- # monologue as content. Stay on required until a tool result
1005
- # exists AND the last assistant turn already looks like a
1006
- # genuine final (no monologue / handoff markers). That keeps
1007
- # pressure on native tool_calls through the mid-loop thrash
1008
- # that previously returned "Wait, let's try hping3…" as FINAL.
1101
+ # After the first tool result, auto so the model can emit a
1102
+ # real final. Leftover English tasks do not keep required.
1009
1103
  has_tool_result = Array(messages).any? { |m| m[:role].to_s == 'tool' }
1010
- last_asst = Array(messages).reverse.find { |m| m[:role].to_s == 'assistant' }
1011
- last_txt = last_asst.is_a?(Hash) ? last_asst[:content].to_s : ''
1012
- still_acting = last_txt.strip.empty? || incomplete_final?(text: last_txt, last_iter: false)
1013
- cwt_opts[:tool_choice] = has_tool_result && !still_acting ? 'auto' : 'required'
1104
+ need_tools = needs_host_work?(request: Array(messages).find { |m| m[:role].to_s == 'user' }&.[](:content))
1105
+ cwt_opts[:tool_choice] = has_tool_result || !need_tools ? 'auto' : 'required'
1014
1106
  end
1015
1107
  end
1016
1108
  response = mod.chat_with_tools(cwt_opts)
@@ -1063,15 +1155,9 @@ module PWN
1063
1155
  # 3.2 — local models cannot afford auto_introspect (judge+prm+critic+
1064
1156
  # sentinel+extro) on every success. Default :failure_only when local.
1065
1157
  private_class_method def self.should_auto_introspect?(opts = {})
1066
- kind = (opts[:kind] || Thread.current[:pwn_request_kind]).to_s.to_sym
1067
1158
  intent = (opts[:intent] || Thread.current[:pwn_request_intent]).to_s.to_sym
1068
- fails = opts[:turn_fails].is_a?(Hash) ? opts[:turn_fails].values.sum : 0
1069
- # Cheap answers already returned user-visible text. The post-answer
1070
- # critic + 12s ORM printed ERROR: Timed out reading data from server
1071
- # after greetings / takes / questions.
1159
+ # Cheap answers already returned user-visible text.
1072
1160
  return false if %i[greeting howto recall].include?(intent)
1073
- return false if kind == :statement
1074
- return false if kind == :question && fails.zero?
1075
1161
 
1076
1162
  return true unless opts[:local]
1077
1163
 
@@ -1116,9 +1202,13 @@ module PWN
1116
1202
  messages = opts[:messages]
1117
1203
  return nil unless state.is_a?(Hash) && messages.is_a?(Array)
1118
1204
  return nil unless defined?(TaskSummarizer) && TaskSummarizer.enabled?
1205
+ return nil if respond_to?(:needs_host_work?) && !needs_host_work?(request: opts[:request] || state[:original_request] || state[:request])
1206
+ return nil unless TaskSummarizer.plan_open?(state: state, messages: messages)
1119
1207
 
1208
+ req = opts[:request]
1209
+ req = state[:original_request] || state[:request] if req.to_s.strip.empty? && state.is_a?(Hash)
1120
1210
  text =
1121
- (TaskSummarizer.active_task_prompt(state: state, force: opts[:force]) if TaskSummarizer.respond_to?(:active_task_prompt))
1211
+ (TaskSummarizer.active_task_prompt(state: state, force: opts[:force], request: req) if TaskSummarizer.respond_to?(:active_task_prompt))
1122
1212
  return nil if text.to_s.strip.empty?
1123
1213
 
1124
1214
  messages << { role: 'user', content: text }
@@ -1249,6 +1339,7 @@ module PWN
1249
1339
  last\s+thing\s+(?:i|you)\s+said
1250
1340
  )\b
1251
1341
  /ix
1342
+ LAST_SESSION_RX = /\b(?:in|from|of)\s+(?:the\s+)?(?:last|previous|prior)\s+session\b|\blast\s+session\b/i
1252
1343
 
1253
1344
  # Pure greeting / light smalltalk — never full :act tool loop.
1254
1345
  # Anchored short forms only so "hi, please scan X" stays :act/:recon_act.
@@ -1316,6 +1407,14 @@ module PWN
1316
1407
  return :recall unless doing
1317
1408
  end
1318
1409
 
1410
+ if req.match?(LAST_SESSION_RX) && !req.match?(HOWTO_RX) && !req.match?(LIVE_RECON_RX)
1411
+ doing = req.match?(
1412
+ /\b(implement|fix|patch|refactor|run|execute|scan|write|edit|
1413
+ change|deploy|install|build|compile|commit|push)\b/ix
1414
+ )
1415
+ return :recall unless doing
1416
+ end
1417
+
1319
1418
  # Live-action recon takes precedence over bare "how to" when both appear
1320
1419
  # only if the user clearly asks the agent to do the sweep here.
1321
1420
  live = req.match?(LIVE_RECON_RX) && req.match?(
@@ -1335,55 +1434,6 @@ module PWN
1335
1434
  :act
1336
1435
  end
1337
1436
 
1338
- # Top-level request kind for task planning (statement | question | autonomous_goal).
1339
- # Single source of truth: TaskSummarizer.request_kind (LLM + heuristics).
1340
- # Mirrors intent/heuristics only when TaskSummarizer is unavailable.
1341
- #
1342
- # Supported Method Parameters::
1343
- # kind = PWN::AI::Agent::Loop.request_kind(
1344
- # request: 'required - user text',
1345
- # kind: 'optional - precomputed',
1346
- # llm_kind: 'optional - injected LLM label',
1347
- # heuristic_only: 'optional - skip LLM'
1348
- # )
1349
- public_class_method def self.request_kind(opts = {})
1350
- req = opts[:request].to_s
1351
- if defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:request_kind)
1352
- return TaskSummarizer.request_kind(
1353
- request: req,
1354
- kind: opts[:kind],
1355
- llm_kind: opts[:llm_kind],
1356
- heuristic_only: opts[:heuristic_only]
1357
- )
1358
- end
1359
-
1360
- case request_intent(request: req)
1361
- when :greeting, :empty
1362
- :statement
1363
- when :howto, :recall
1364
- :question
1365
- when :recon_act
1366
- :autonomous_goal
1367
- else
1368
- # :act — distinguish bare questions from work the agent must do.
1369
- # Host-local facts need tools → autonomous_goal.
1370
- if defined?(TaskSummarizer) && TaskSummarizer.const_defined?(:NEEDS_LOCAL_EVIDENCE_RX)
1371
- return :autonomous_goal if req.match?(TaskSummarizer::NEEDS_LOCAL_EVIDENCE_RX)
1372
- elsif req.match?(/\b(?:hostname|whoami|\bcwd\b|\bpwd\b|my\s+ip)\b/i)
1373
- return :autonomous_goal
1374
- end
1375
- return :question if req.match?(/\?\s*\z/) && !req.match?(
1376
- /\b(please|implement|fix|patch|refactor|run|scan|find|write|change)\b/i
1377
- )
1378
- return :question if req.match?(/\A\s*(?:what|why|when|where|who|which|how)\b/i) &&
1379
- !req.match?(/\b(please|implement|fix|patch|run|scan)\b/i)
1380
-
1381
- :autonomous_goal
1382
- end
1383
- rescue StandardError
1384
- :autonomous_goal
1385
- end
1386
-
1387
1437
  public_class_method def self.recon_authorized?(opts = {})
1388
1438
  req = opts[:request].to_s
1389
1439
  return true if req.match?(AUTH_SCOPE_RX)
@@ -1687,6 +1737,18 @@ module PWN
1687
1737
  session_id = opts[:session_id]
1688
1738
  system_role_content = opts[:system_role_content].to_s
1689
1739
  target = recall_target(request: request)
1740
+ last_session = request.match?(LAST_SESSION_RX)
1741
+ if last_session && defined?(PWN::Sessions) && PWN::Sessions.respond_to?(:previous_id)
1742
+ prev = PWN::Sessions.previous_id(exclude_session_id: session_id)
1743
+ if prev.to_s.empty?
1744
+ txt = 'I do not have a previous session transcript yet.'
1745
+ append_session(session_id: opts[:session_id], role: 'user', content: request)
1746
+ append_session(session_id: opts[:session_id], role: 'assistant', content: txt)
1747
+ return txt
1748
+ end
1749
+
1750
+ session_id = prev
1751
+ end
1690
1752
 
1691
1753
  prior_user = nil
1692
1754
  prior_asst = nil
@@ -1729,7 +1791,7 @@ module PWN
1729
1791
  # User-target / fallback: skip meta intermediate recall asks so
1730
1792
  # "what did I just say?" after a nested chain still surfaces the
1731
1793
  # original utterance when appropriate; default stays newest.
1732
- skip_meta = target == :assistant
1794
+ skip_meta = target == :assistant || last_session
1733
1795
  prior_user = PWN::Memory.prior_user_message(
1734
1796
  session_id: session_id,
1735
1797
  max_chars: 4_000,
@@ -1913,73 +1975,76 @@ module PWN
1913
1975
  session_id = opts[:session_id]
1914
1976
  on_tool = opts[:on_tool]
1915
1977
  TurnFinalizer.enter_user_path! if defined?(TurnFinalizer)
1916
- # Live coalesced "what am I doing" lines for the TUI (not a model tool).
1917
- ts_state = (TaskSummarizer.fresh(request: request) if defined?(TaskSummarizer) && TaskSummarizer.enabled? && Thread.current[:pwn_reflect_depth].to_i.zero?)
1918
1978
  engine = active_engine
1919
1979
  local = local_engine?(engine: engine)
1920
- system_role_content = opts[:system_role_content] ||= PWN::AI::Agent::PromptBuilder.build(session_id: session_id, request: request)
1921
-
1922
- Registry.discover
1923
- maybe_refresh_extro_snapshot!
1924
- opts[:enabled_toolsets] = default_interactive_toolsets(request: request) unless opts.key?(:enabled_toolsets)
1925
- expose_current_session(session_id: session_id)
1926
- Mistakes.check_user_correction(request: request, session_id: session_id) if defined?(Mistakes)
1927
1980
 
1981
+ # Cheap intent/kind FIRST - before PromptBuilder / Registry / TaskSummarizer
1982
+ # so greetings, FYIs, how-tos, recall, and simple Qs never pay the fat path.
1928
1983
  intent = request_intent(request: request)
1929
- kind = request_kind(request: request)
1930
1984
  Thread.current[:pwn_request_intent] = intent
1931
- Thread.current[:pwn_request_kind] = kind
1932
1985
  Thread.current[:pwn_recon_authorized] = recon_authorized?(request: request)
1933
1986
  Thread.current[:pwn_extinguished] = {}
1934
- # Greeting / light smalltalk: deterministic ack — no weather echo, no tools.
1987
+ expose_current_session(session_id: session_id)
1988
+ Mistakes.check_user_correction(request: request, session_id: session_id) if defined?(Mistakes)
1989
+
1990
+ cheap = opts[:force_tools] != true && %i[greeting howto recall].include?(intent)
1991
+
1992
+ # Greeting / light smalltalk: deterministic ack - no weather echo, no tools,
1993
+ # no PromptBuilder, no Registry.
1935
1994
  if intent == :greeting && opts[:force_tools] != true
1936
1995
  return answer_greeting(
1937
1996
  request: request,
1938
1997
  session_id: session_id
1939
1998
  )
1940
1999
  end
1941
- # How-to: never enter plan_first / task recon / tool thrash (ollama/openwebui).
1942
- if intent == :howto && opts[:force_tools] != true
1943
- return answer_howto(
1944
- request: request,
1945
- session_id: session_id,
1946
- system_role_content: system_role_content
1947
- )
1948
- end
1949
- # Pure prior-turn / vague memory recall: one cheap path, no plan_first.
1950
- if intent == :recall && opts[:force_tools] != true
1951
- return answer_recall(
1952
- request: request,
1953
- session_id: session_id,
1954
- system_role_content: system_role_content
1955
- )
1956
- end
1957
- # General statements: acknowledge briefly — no multi-step task plan.
1958
- # Kind is source of truth (LLM+heuristic). Never short-circuit goals.
1959
- if kind.to_sym == :statement && intent != :recon_act && opts[:force_tools] != true
1960
- return answer_statement(
1961
- request: request,
1962
- session_id: session_id
1963
- )
1964
- end
1965
- # Pure questions that are not how-to/recall: concise answer, no multi-step plan.
1966
- # Host-evidence interrogatives classify as autonomous_goal above so they
1967
- # keep tools (e.g. "what is my hostname?"). force_tools bypasses for tests.
1968
- if kind.to_sym == :question && !%i[recon_act].include?(intent) && opts[:force_tools] != true
1969
- return answer_question(
1970
- request: request,
1971
- session_id: session_id,
1972
- system_role_content: system_role_content
1973
- )
2000
+
2001
+ # Thin system prompt only for remaining cheap paths (howto/recall/statement/question).
2002
+ if cheap
2003
+ system_role_content = opts[:system_role_content]
2004
+ if system_role_content.nil? || system_role_content.to_s.empty?
2005
+ system_role_content = PWN::AI::Agent::PromptBuilder.build(
2006
+ session_id: session_id,
2007
+ request: request,
2008
+ thin: true
2009
+ )
2010
+ opts[:system_role_content] = system_role_content
2011
+ end
2012
+ # How-to: never enter plan_first / task recon / tool thrash.
2013
+ if intent == :howto
2014
+ return answer_howto(
2015
+ request: request,
2016
+ session_id: session_id,
2017
+ system_role_content: system_role_content
2018
+ )
2019
+ end
2020
+ # Pure prior-turn / vague memory recall: one cheap path, no plan_first.
2021
+ if intent == :recall
2022
+ return answer_recall(
2023
+ request: request,
2024
+ session_id: session_id,
2025
+ system_role_content: system_role_content
2026
+ )
2027
+ end
1974
2028
  end
1975
2029
 
2030
+ # --- act / recon / autonomous_goal: full context + tools ---
2031
+ # Reuse precomputed kind so TaskSummarizer.fresh does not classify twice.
2032
+ ts_state = (TaskSummarizer.fresh(request: request) if defined?(TaskSummarizer) && TaskSummarizer.enabled? && Thread.current[:pwn_reflect_depth].to_i.zero?)
2033
+ system_role_content = opts[:system_role_content] ||= PWN::AI::Agent::PromptBuilder.build(
2034
+ session_id: session_id,
2035
+ request: request
2036
+ )
2037
+
2038
+ Registry.discover
2039
+ maybe_refresh_extro_snapshot!
2040
+ opts[:enabled_toolsets] = default_interactive_toolsets(request: request) unless opts.key?(:enabled_toolsets)
2041
+
1976
2042
  # R5 — open the live MDP episode BEFORE the first Registry.rank so
1977
2043
  # Q(s,a) can advise this turn. Planning still owns the task list.
1978
2044
  if defined?(PWN::AI::Agent::Policy) && Policy.respond_to?(:begin_episode)
1979
2045
  Policy.begin_episode(
1980
2046
  session_id: session_id,
1981
2047
  request: request,
1982
- kind: kind,
1983
2048
  intent: intent,
1984
2049
  engine: engine,
1985
2050
  ts_state: ts_state
@@ -1990,53 +2055,43 @@ module PWN
1990
2055
  # TaskSummarizer.emit_plan! we re-rank using English tangible tasks
1991
2056
  # so generated tasks — not the bare request — drive which tools
1992
2057
  # the model may call.
1993
- tools = Registry.definitions(enabled: opts[:enabled_toolsets], relevance: request)
2058
+ # CORE_TOOLS is the default action space. Extra schemas are
2059
+ # opt-in via enabled_toolsets + core_only: false.
2060
+ core_only = opts.fetch(:core_only, true)
2061
+ tools = Registry.definitions(
2062
+ enabled: opts[:enabled_toolsets],
2063
+ relevance: request,
2064
+ core_only: core_only,
2065
+ intent: intent
2066
+ )
1994
2067
  messages = [{ role: 'system', content: system_role_content }]
1995
2068
  messages.concat(Learning.exemplars_for(request: request)) if local && defined?(Learning) && Learning.respond_to?(:exemplars_for)
1996
2069
  messages << { role: 'user', content: request }
1997
2070
  append_session(session_id: session_id, role: 'user', content: request)
1998
2071
 
1999
- # Tangible-task breakdown ONLY for autonomous goals.
2000
- # General statements and questions stay without multi-step plans.
2001
- needs_breakdown =
2002
- if defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:needs_task_breakdown?)
2003
- TaskSummarizer.needs_task_breakdown?(kind: kind, request: request)
2004
- else
2005
- kind.to_sym == :autonomous_goal
2006
- end
2007
- ts_state[:request_kind] = kind if ts_state.is_a?(Hash)
2008
- if needs_breakdown
2009
- task_summary_plan!(state: ts_state, request: request, on_tool: on_tool)
2010
- elsif ts_state.is_a?(Hash) && defined?(TaskSummarizer)
2011
- # Record kind on state; optional one-line kind banner (no task list).
2012
- ts_state[:plan] = []
2013
- ts_state[:request_kind] = kind
2014
- if TaskSummarizer.respond_to?(:format_plan)
2015
- banner = TaskSummarizer.format_plan(tasks: [], request: request, request_kind: kind)
2016
- if banner && !banner.to_s.empty?
2017
- ts_state[:plan_text] = banner
2018
- ts_state[:plan_emitted] = true
2019
- emit_task_summary(line: banner, on_tool: on_tool)
2020
- end
2021
- end
2022
- end
2072
+ # Every request gets a task compass.
2073
+ task_summary_plan!(state: ts_state, request: request, on_tool: on_tool) if defined?(TaskSummarizer)
2023
2074
  # Re-bind tools from English plan so task list is the sole driver of
2024
2075
  # tool exposure/ranking (Registry keyword router + CORE).
2025
2076
  if ts_state.is_a?(Hash) && defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:relevance_query)
2026
2077
  rq = TaskSummarizer.relevance_query(state: ts_state, request: request)
2027
- tools = Registry.definitions(enabled: opts[:enabled_toolsets], relevance: rq) unless rq.to_s.strip.empty?
2078
+ unless rq.to_s.strip.empty?
2079
+ tools = Registry.definitions(
2080
+ enabled: opts[:enabled_toolsets],
2081
+ relevance: rq,
2082
+ core_only: core_only,
2083
+ intent: intent
2084
+ )
2085
+ end
2028
2086
  end
2029
2087
  # English-task-as-primary: inject tangible tasks only for autonomous goals.
2030
- inject_task_focus!(messages: messages, state: ts_state, force: true) if needs_breakdown
2031
-
2088
+ inject_task_focus!(messages: messages, state: ts_state, force: true, request: request)
2032
2089
  predicted = nil
2033
2090
  Thread.current[:pwn_plan_predicted] = nil
2034
2091
  cal_state = calibration_state
2035
2092
  force_plan = cal_state[:force_plan]
2036
- # Skip plan_first for non-goals (statements/questions) and cheap intents.
2037
- skip_plan = %i[howto recall greeting].include?(intent) ||
2038
- %i[statement question].include?(kind.to_sym) ||
2039
- !needs_breakdown
2093
+ # Skip plan_first only for remaining cheap intents.
2094
+ skip_plan = %i[howto recall greeting].include?(intent)
2040
2095
  if !skip_plan && (force_plan || agent_flag(key: :plan_first, default: local) || budget_exhaustion_hot?) && !Array(tools).empty?
2041
2096
  predicted = plan_first(messages: messages, request: request, ts_state: ts_state)
2042
2097
  # P22 — prefer explicit return; fall back to thread stash
@@ -2046,12 +2101,21 @@ module PWN
2046
2101
  # PLAN: tool-call scaffold jargon (unify_plan! refuses that).
2047
2102
  if ts_state.is_a?(Hash) && defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:relevance_query)
2048
2103
  rq = TaskSummarizer.relevance_query(state: ts_state, request: request)
2049
- tools = Registry.definitions(enabled: opts[:enabled_toolsets], relevance: rq) unless rq.to_s.strip.empty?
2104
+ unless rq.to_s.strip.empty?
2105
+ tools = Registry.definitions(
2106
+ enabled: opts[:enabled_toolsets],
2107
+ relevance: rq,
2108
+ core_only: core_only,
2109
+ intent: intent
2110
+ )
2111
+ end
2050
2112
  end
2051
- inject_task_focus!(messages: messages, state: ts_state, force: true)
2113
+ inject_task_focus!(messages: messages, state: ts_state, force: true, request: request)
2052
2114
  end
2053
2115
  if budget_exhaustion_hot?
2054
- hot_hint = if local_engine?
2116
+ english_open = defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:plan_open?) &&
2117
+ TaskSummarizer.plan_open?(state: ts_state, messages: messages)
2118
+ hot_hint = if local_engine? && !english_open
2055
2119
  '[pwn-ai/p17] Budget-exhaustion is the top open failure on this host. ' \
2056
2120
  'Prefer the SHORTEST plan that finishes the ask (≤3 tool calls). ' \
2057
2121
  'Emit a final answer as soon as you have evidence — do not explore.'
@@ -2083,7 +2147,7 @@ module PWN
2083
2147
  compact_history!(messages: messages) if local
2084
2148
  # English-task-as-primary: when plan_idx advanced, tell the model
2085
2149
  # which plain-English task is active before the next tool batch.
2086
- inject_task_focus!(messages: messages, state: ts_state)
2150
+ inject_task_focus!(messages: messages, state: ts_state, request: request)
2087
2151
 
2088
2152
  # P17 — on the final iteration, strip tools and demand a plain-text
2089
2153
  # answer. Without this the model happily emits one more tool_calls
@@ -2116,15 +2180,9 @@ module PWN
2116
2180
  plan_faithful = hot && plan_steps.positive? && plan_steps <= plan_step_limit &&
2117
2181
  turn_fails['empty_final'].to_i.zero? &&
2118
2182
  turn_fails.values.sum < 2
2119
- text_only_iters = if hot
2120
- if plan_faithful
2121
- 1
2122
- else
2123
- (local_engine? ? 3 : 2)
2124
- end
2125
- else
2126
- 1
2127
- end
2183
+ # Last-iter strips tools only on the true last slot. English
2184
+ # leftovers and budget-hot must not steal runway from a live goal.
2185
+ text_only_iters = 1
2128
2186
  last_iter = (i >= max_iters - text_only_iters)
2129
2187
  if last_iter
2130
2188
  tag = i >= max_iters - 1 ? 'FINAL ITERATION' : 'PENULTIMATE — wrap up'
@@ -2138,7 +2196,7 @@ module PWN
2138
2196
  }
2139
2197
  end
2140
2198
 
2141
- msg = call_engine(messages: messages, tools: last_iter ? nil : tools)
2199
+ msg = call_engine(messages: messages, tools: last_iter ? nil : tools, ts_state: ts_state)
2142
2200
  if msg.nil?
2143
2201
  task_summary_flush!(state: ts_state, on_tool: on_tool)
2144
2202
  return '[pwn-ai] engine returned no message'
@@ -2195,6 +2253,21 @@ module PWN
2195
2253
  }
2196
2254
  next
2197
2255
  end
2256
+ if request_unsatisfied?(
2257
+ request: request,
2258
+ messages: messages,
2259
+ last_iter: last_iter
2260
+ ) && turn_fails['unsatisfied'].to_i < 4
2261
+ turn_fails['unsatisfied'] += 1
2262
+ warn "[pwn-ai/loop] original request not evidenced on iter=#{i}; continuing"
2263
+ messages << {
2264
+ role: 'user',
2265
+ content: '[pwn-ai] The original request is not evidenced yet. ' \
2266
+ 'Keep calling CORE_TOOLS (shell, pwn_eval) until that request is ' \
2267
+ 'done or truly blocked. Do not declare completion from a listing alone.'
2268
+ }
2269
+ next
2270
+ end
2198
2271
  append_session(session_id: session_id, role: 'assistant', content: text)
2199
2272
  Learning.auto_introspect(session_id: session_id, request: request, final: text, predicted: predicted, plan: ts_state && ts_state[:plan], ts_state: ts_state) if defined?(Learning) && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: i)
2200
2273
  maybe_finish_policy(session_id: session_id, proxy_ok: true, ts_state: ts_state)