pwn 0.5.685 → 0.5.688

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. checksums.yaml +4 -4
  2. data/.gitignore +2 -0
  3. data/.ruby-version +1 -1
  4. data/Gemfile +3 -3
  5. data/README.md +3 -4
  6. data/documentation/Agent-Tool-Registry.md +2 -2
  7. data/documentation/Configuration.md +5 -4
  8. data/documentation/General-PWN-Usage.md +1 -1
  9. data/documentation/Home.md +1 -1
  10. data/documentation/How-PWN-Works.md +1 -1
  11. data/documentation/Mistakes.md +1 -1
  12. data/documentation/Reinforcement-Learning.md +4 -6
  13. data/documentation/Troubleshooting.md +1 -1
  14. data/documentation/pwn-REPL.md +1 -1
  15. data/documentation/pwn-ai-Agent.md +2 -3
  16. data/lib/pwn/ai/agent/curriculum.rb +2 -2
  17. data/lib/pwn/ai/agent/dispatch.rb +1 -1
  18. data/lib/pwn/ai/agent/learning.rb +150 -32
  19. data/lib/pwn/ai/agent/loop.rb +352 -127
  20. data/lib/pwn/ai/agent/prompt_builder.rb +50 -29
  21. data/lib/pwn/ai/agent/registry.rb +2 -1
  22. data/lib/pwn/ai/agent/reward.rb +9 -0
  23. data/lib/pwn/ai/agent/swarm.rb +3 -1
  24. data/lib/pwn/ai/agent/task_summarizer.rb +23 -6
  25. data/lib/pwn/ai/agent/tool_guard.rb +109 -0
  26. data/lib/pwn/ai/agent/tools/ruby_eval.rb +44 -20
  27. data/lib/pwn/ai/agent/tools/shell.rb +17 -4
  28. data/lib/pwn/ai/agent/tools/skills.rb +63 -2
  29. data/lib/pwn/ai/anthropic.rb +25 -123
  30. data/lib/pwn/ai/gemini.rb +25 -20
  31. data/lib/pwn/ai/grok.rb +30 -171
  32. data/lib/pwn/ai/http_retry.rb +78 -0
  33. data/lib/pwn/ai/ollama.rb +25 -17
  34. data/lib/pwn/ai/open_ai.rb +26 -123
  35. data/lib/pwn/ai/open_web_ui.rb +25 -17
  36. data/lib/pwn/ai.rb +1 -86
  37. data/lib/pwn/config.rb +1 -1
  38. data/lib/pwn/plugins/log.rb +444 -0
  39. data/lib/pwn/plugins/repl.rb +75 -20
  40. data/lib/pwn/plugins/transparent_browser.rb +16 -2
  41. data/lib/pwn/sessions.rb +50 -1
  42. data/lib/pwn/version.rb +1 -1
  43. data/spec/integration/reinforced_feedback_loop_spec.rb +9 -11
  44. data/spec/lib/pwn/ai/agent/loop_spec.rb +223 -25
  45. data/spec/lib/pwn/ai/agent/prompt_builder_spec.rb +3 -0
  46. data/spec/lib/pwn/ai/agent/registry_spec.rb +1 -1
  47. data/spec/lib/pwn/ai/agent/reward_spec.rb +13 -0
  48. data/spec/lib/pwn/ai/agent/swarm_spec.rb +6 -0
  49. data/spec/lib/pwn/ai/agent/task_summarizer_spec.rb +36 -0
  50. data/spec/lib/pwn/ai/agent/tool_guard_spec.rb +46 -0
  51. data/spec/lib/pwn/ai/agent/tools/ruby_eval_spec.rb +24 -0
  52. data/spec/lib/pwn/ai/agent/tools/shell_spec.rb +10 -0
  53. data/spec/lib/pwn/ai/agent/tools/skills_spec.rb +29 -0
  54. data/spec/lib/pwn/ai/anthropic_spec.rb +2 -2
  55. data/spec/lib/pwn/ai/gemini_spec.rb +2 -2
  56. data/spec/lib/pwn/ai/grok_spec.rb +24 -2
  57. data/spec/lib/pwn/ai/http_retry_spec.rb +55 -0
  58. data/spec/lib/pwn/ai/ollama_spec.rb +2 -7
  59. data/spec/lib/pwn/ai/open_ai_spec.rb +2 -23
  60. data/spec/lib/pwn/ai/open_web_ui_spec.rb +2 -7
  61. data/spec/lib/pwn/ai_spec.rb +4 -21
  62. data/spec/lib/pwn/plugins/log_spec.rb +290 -0
  63. data/spec/lib/pwn/plugins/repl_pwn_vault_spec.rb +42 -1
  64. data/spec/lib/pwn/plugins/repl_spec.rb +4 -3
  65. data/spec/lib/pwn/plugins/transparent_browser_spec.rb +8 -0
  66. data/spec/lib/pwn/sessions_spec.rb +24 -0
  67. data/third_party/pwn_rdoc.jsonl +65 -23
  68. data/tmp_pwn_critic_skills_flaw.txt +9 -0
  69. metadata +11 -8
@@ -62,6 +62,140 @@ module PWN
62
62
  unsatisfied incomplete_final empty_final evidence_final
63
63
  ].freeze
64
64
 
65
+ public_class_method def self.debug_on?(opts = {})
66
+ return true if opts[:debug]
67
+ return true if defined?(PWN::Plugins::Log) && PWN::Plugins::Log.debug_enabled?
68
+
69
+ pry_on = defined?(Pry) && Pry.respond_to?(:config) &&
70
+ Pry.config.respond_to?(:pwn_ai_debug) && Pry.config.pwn_ai_debug
71
+ return true if pry_on
72
+
73
+ false
74
+ end
75
+
76
+ private_class_method def self.debug_progress(opts = {})
77
+ return false unless debug_on?(opts)
78
+ return false unless defined?(PWN::Plugins::Log)
79
+
80
+ payload = {
81
+ msg: opts[:msg],
82
+ which_self: opts[:which_self] || self,
83
+ keep_newlines: opts[:keep_newlines]
84
+ }
85
+ payload[:cap] = opts[:cap] if opts.key?(:cap)
86
+ payload[:tee] = opts[:tee] if opts.key?(:tee)
87
+ PWN::Plugins::Log.progress(payload)
88
+ end
89
+
90
+ private_class_method def self.debug_tools_line(opts = {})
91
+ tools = opts[:tools]
92
+ return 'tools=none' if tools.nil?
93
+
94
+ names = Array(tools).filter_map do |tool|
95
+ next unless tool.is_a?(Hash)
96
+
97
+ tool.dig(:function, :name) ||
98
+ tool.dig('function', 'name') ||
99
+ tool[:name] ||
100
+ tool['name']
101
+ end.map(&:to_s).reject(&:empty?)
102
+ "tools=#{names.length}[#{names.join(',')}]"
103
+ end
104
+
105
+ private_class_method def self.debug_msgs_line(opts = {})
106
+ bits = Array(opts[:messages]).map do |msg|
107
+ role = (msg[:role] || msg['role'] || '?').to_s
108
+ len = (msg[:content] || msg['content']).to_s.length
109
+ "#{role}:#{len}"
110
+ end
111
+ "msgs=#{bits.length}[#{bits.join(',')}]"
112
+ end
113
+
114
+ private_class_method def self.debug_snippet(opts = {})
115
+ text = opts[:text].to_s.tr("\n", ' ').strip
116
+ max = opts[:max].to_i
117
+ max = 400 unless max.positive?
118
+ text.length > max ? "#{text[0, max]}…" : text
119
+ end
120
+
121
+ private_class_method def self.debug_final_text!(opts = {})
122
+ return unless debug_on?(opts)
123
+
124
+ debug_progress(
125
+ msg: "final text:\n#{opts[:text]}",
126
+ keep_newlines: true,
127
+ cap: 65_536,
128
+ debug: opts[:debug]
129
+ )
130
+ end
131
+
132
+ private_class_method def self.debug_tool_io!(opts = {})
133
+ return unless debug_on?(opts)
134
+
135
+ name = opts[:name].to_s
136
+ argv = opts[:args].is_a?(String) ? opts[:args].to_s : opts[:args].inspect
137
+ result = opts[:result].to_s
138
+ debug_progress(
139
+ msg: "tool #{name} request:\n#{argv}\nresult:\n#{result}",
140
+ keep_newlines: true,
141
+ cap: 0,
142
+ tee: nil,
143
+ debug: opts[:debug]
144
+ )
145
+ end
146
+
147
+ private_class_method def self.start_debug_session(opts = {})
148
+ return unless debug_on?(opts)
149
+ return unless defined?(PWN::Plugins::Log)
150
+
151
+ if defined?(TurnFinalizer) && TurnFinalizer.user_path?
152
+ debug_progress(msg: 'nested Loop.run skip_roll', debug: opts[:debug])
153
+ return
154
+ end
155
+
156
+ unless PWN::Plugins::Log.debug_enabled?
157
+ want_trace = opts[:trace] == true
158
+ begin
159
+ want_trace ||= PWN::Env.dig(:ai, :agent, :debug_trace) == true
160
+ rescue StandardError
161
+ nil
162
+ end
163
+ PWN::Plugins::Log.start_debug(
164
+ tee: opts[:debug_tee] || $stdout,
165
+ session_id: opts[:session_id],
166
+ trace: want_trace
167
+ )
168
+ end
169
+ PWN::Plugins::Log.next_request_log!(session_id: opts[:session_id])
170
+ end
171
+
172
+ private_class_method def self.finish_debug_request!(opts = {})
173
+ return unless debug_on?(opts)
174
+ return unless defined?(PWN::Plugins::Log)
175
+
176
+ PWN::Plugins::Log.finish_request_log!(
177
+ iter: opts[:iter],
178
+ tools_called: opts[:tools_called],
179
+ engine_s: opts[:engine_s],
180
+ final_chars: opts[:final_chars],
181
+ nested: opts[:nested]
182
+ )
183
+ end
184
+
185
+ private_class_method def self.quiet_debug_tui!(opts = {})
186
+ return unless debug_on?(opts)
187
+ return unless defined?(PWN::Plugins::Log)
188
+
189
+ PWN::Plugins::Log.quiet_tui!(reason: opts[:reason])
190
+ end
191
+
192
+ private_class_method def self.loud_debug_tui!(opts = {})
193
+ return unless debug_on?(opts)
194
+ return unless defined?(PWN::Plugins::Log)
195
+
196
+ PWN::Plugins::Log.loud_tui!(reason: opts[:reason])
197
+ end
198
+
65
199
  private_class_method def self.dispatch_fail_n(opts = {})
66
200
  fails = opts[:turn_fails] || {}
67
201
  fails.sum do |key, count|
@@ -326,6 +460,10 @@ module PWN
326
460
  LOOKUP_REQUEST_RX = /
327
461
  \b(what\s+is\s+my|hostname|uname|cwd|whoami|status|version|how\s+many)\b
328
462
  /ix
463
+ SKILLS_CATALOG_RX = /
464
+ \bskills?\b.{0,40}\b(available|installed|loaded|catalog|list)\b |
465
+ \b(what|which|list)\b.{0,40}\bskills?\b
466
+ /ix
329
467
  # Real filesystem paths only — not https://host.tld (that was matching //host.tld).
330
468
  HOST_PATH_RX = %r{(?:(?<![.:/])/(?!/)|\./)[\w./-]+\.\w+}
331
469
  BROWSER_REQUEST_RX = /
@@ -335,14 +473,28 @@ module PWN
335
473
 
336
474
  # True only when the ask needs a live host/file/browser effect. World-knowledge
337
475
  # questions ("what color is a cherry") do not.
476
+ public_class_method def self.catalog_lookup?(opts = {})
477
+ request = opts[:request].to_s.strip
478
+ return false if request.empty?
479
+ return false if request.length > 120
480
+ return false if request.match?(ACT_REQUEST_RX)
481
+ return false if request.match?(HOWTO_RX)
482
+
483
+ request.match?(SKILLS_CATALOG_RX)
484
+ rescue StandardError
485
+ false
486
+ end
487
+
338
488
  public_class_method def self.world_knowledge?(opts = {})
339
489
  request = opts[:request].to_s.strip
340
490
  return false if request.empty?
341
491
  return false if request.length > 120
492
+ return false if catalog_lookup?(request: request)
342
493
  return false if request.match?(ACT_REQUEST_RX)
343
494
  return false if request.match?(LOOKUP_REQUEST_RX)
344
495
  return false if request.match?(HOST_PATH_RX)
345
496
  return false if request.match?(BROWSER_REQUEST_RX)
497
+ return false if request.match?(HOWTO_RX)
346
498
  return false if request.match?(%r{\b(this\s+(?:host|machine|box|system|subnet|file|repo)|/opt/|implement|scan|hosts?)\b}i)
347
499
 
348
500
  request.match?(/\A(?:what|why|who|when|where|which|how)\b/i)
@@ -354,6 +506,7 @@ module PWN
354
506
  request = opts[:request].to_s
355
507
  return false if request.strip.empty?
356
508
  return false if world_knowledge?(request: request)
509
+ return false if catalog_lookup?(request: request)
357
510
 
358
511
  true
359
512
  rescue StandardError
@@ -363,6 +516,7 @@ module PWN
363
516
  private_class_method def self.request_need(opts = {})
364
517
  request = opts[:request].to_s
365
518
  return :none if world_knowledge?(request: request)
519
+ return :read if catalog_lookup?(request: request)
366
520
  return :read if request.match?(LOOKUP_REQUEST_RX)
367
521
  return :browse if request.match?(BROWSER_REQUEST_RX)
368
522
  return :write if request.match?(ACT_REQUEST_RX) || request.match?(HOST_PATH_RX)
@@ -410,9 +564,12 @@ module PWN
410
564
  request = opts[:request].to_s
411
565
  need = request_need(request: request)
412
566
  return false if need == :none
413
- return false unless needs_host_work?(request: request)
567
+ return false if Thread.current[:pwn_loop_no_tools]
414
568
 
415
569
  effects = tool_effects(messages: opts[:messages])
570
+ return !effects.intersect?(%i[read recall eval]) if need == :read && catalog_lookup?(request: request)
571
+ return false unless needs_host_work?(request: request)
572
+
416
573
  live = effects.reject { |fx| %i[recall store].include?(fx) }
417
574
  return true if live.empty?
418
575
  return true if need == :write && !write_verified?(effects: effects)
@@ -614,11 +771,19 @@ module PWN
614
771
  end
615
772
  m = nil
616
773
  if !sem[:semantic_ok] && defined?(Mistakes) && sem[:shape].to_s != 'invalid_payload' && !raw.include?('extinguished_repeat')
617
- # E1 — automatic blame attribution: if this tool just tripped a
618
- # CUSUM changepoint AND extro drift is present, tag the mistake
619
- # cause: :env_drift so it does NOT count toward [REPEATING].
620
774
  cause = attribute_cause(name: name)
621
- m = Mistakes.record(tool: name, error: sem[:err] || raw[0, 300], args: opts[:args], session_id: opts[:session_id], source: :tool, cause: cause, shape: sem[:shape])
775
+ err = sem[:err] || raw[0, 300]
776
+ shape = sem[:shape]
777
+ if sem[:shape].to_s == 'timeout' && defined?(ToolGuard) && ToolGuard.respond_to?(:timeout_lesson)
778
+ lesson = ToolGuard.timeout_lesson(
779
+ tool: name,
780
+ payload: opts[:args].to_s,
781
+ timeout: err.to_s[/timeout after (\d+)/, 1].to_i
782
+ )
783
+ err = lesson[:error] if lesson[:error].to_s.strip.length.positive?
784
+ shape = :timeout
785
+ end
786
+ m = Mistakes.record(tool: name, error: err, args: opts[:args], session_id: opts[:session_id], source: :tool, cause: cause, shape: shape)
622
787
  m = Mistakes.extinguish!(signature: m[:signature], args: opts[:args], shape: sem[:shape]) || m if m && defined?(Mistakes) && Mistakes.respond_to?(:extinguish!)
623
788
  end
624
789
  { ok: sem[:semantic_ok], err: sem[:err], mistake: m, benign: sem[:benign] }
@@ -1095,6 +1260,9 @@ module PWN
1095
1260
  private_class_method def self.call_engine(opts = {})
1096
1261
  messages = opts[:messages]
1097
1262
  tools = opts[:tools]
1263
+ debug_progress(
1264
+ msg: "call_engine #{debug_tools_line(tools: tools)} #{debug_msgs_line(messages: messages)}"
1265
+ )
1098
1266
 
1099
1267
  engine = active_engine
1100
1268
  mod_name = ENGINE_MODS[engine]
@@ -1115,7 +1283,7 @@ module PWN
1115
1283
  cwt_opts = {
1116
1284
  messages: wire_msgs,
1117
1285
  tools: tools,
1118
- spinner: false
1286
+ spinner: true
1119
1287
  }
1120
1288
  # Ollama + abliterated / weak chat-templates often ignore tools: and
1121
1289
  # answer in prose (or print shell(...) as text). Force native
@@ -1150,6 +1318,40 @@ module PWN
1150
1318
  # Keep: system, original user, PLAN assistant (if any), last K tool
1151
1319
  # pairs (assistant+tool), and the most recent assistant. Stale tool
1152
1320
  # bodies are truncated to history_tool_max_chars.
1321
+ private_class_method def self.session_chat_history(opts = {})
1322
+ return [] unless defined?(PWN::Sessions) && PWN::Sessions.respond_to?(:to_llm_messages)
1323
+
1324
+ cap = opts[:max_chars].to_i
1325
+ if cap <= 0
1326
+ cap = local_engine? ? 8_000 : 48_000
1327
+ begin
1328
+ n = PWN::Env.dig(:ai, active_engine, :max_prompt_length).to_i
1329
+ cap = [cap, (n / 4)].min if n.positive?
1330
+ rescue StandardError
1331
+ nil
1332
+ end
1333
+ end
1334
+ PWN::Sessions.to_llm_messages(
1335
+ session_id: opts[:session_id],
1336
+ max_chars: cap,
1337
+ skip_request: opts[:skip_request]
1338
+ )
1339
+ rescue StandardError
1340
+ []
1341
+ end
1342
+
1343
+ private_class_method def self.chat_response_history(opts = {})
1344
+ hist = session_chat_history(
1345
+ session_id: opts[:session_id],
1346
+ skip_request: opts[:skip_request] || opts[:request]
1347
+ )
1348
+ return if hist.empty?
1349
+
1350
+ {
1351
+ choices: [{ role: 'system', content: opts[:system_role_content].to_s }] + hist
1352
+ }
1353
+ end
1354
+
1153
1355
  private_class_method def self.compact_history!(opts = {})
1154
1356
  messages = opts[:messages]
1155
1357
  return messages unless messages.is_a?(Array) && messages.length > 12
@@ -1159,17 +1361,21 @@ module PWN
1159
1361
 
1160
1362
  head = []
1161
1363
  rest = messages.dup
1162
- # always keep leading system + first user + optional PLAN
1163
- while rest.any? && %w[system user].include?(rest.first[:role].to_s)
1364
+ # Keep system + this-session user/assistant conversation. Only
1365
+ # compact this-run tool dumps after that.
1366
+ head << rest.shift while rest.any? && rest.first[:role].to_s == 'system'
1367
+ while rest.any?
1368
+ msg = rest.first
1369
+ role = msg[:role].to_s
1370
+ break if role == 'tool'
1371
+ break if role == 'assistant' && Array(msg[:tool_calls]).any?
1372
+
1164
1373
  head << rest.shift
1165
- break if head.any? { |m| m[:role].to_s == 'user' }
1166
1374
  end
1167
1375
  head << rest.shift if rest.any? && rest.first[:role].to_s == 'assistant' && rest.first[:content].to_s.start_with?('PLAN:')
1168
1376
 
1169
- # find indices of tool messages in rest; keep only last keep_pairs tool groups
1170
1377
  tool_idxs = rest.each_index.select { |i| rest[i][:role].to_s == 'tool' }
1171
1378
  drop_before = tool_idxs.length > keep_pairs ? tool_idxs[-keep_pairs] : 0
1172
- # include the assistant tool_call message immediately before first kept tool
1173
1379
  start = drop_before
1174
1380
  start -= 1 if start.positive? && rest[start - 1] && rest[start - 1][:role].to_s == 'assistant'
1175
1381
  kept = rest[start..] || []
@@ -1192,6 +1398,12 @@ module PWN
1192
1398
  # Cheap answers already returned user-visible text.
1193
1399
  return false if %i[greeting howto recall].include?(intent)
1194
1400
 
1401
+ # Advisor / nested Loop.run (red-team, critic) must not start
1402
+ # another tool-armed review of the same goal.
1403
+ return false if Thread.current[:pwn_loop_no_tools]
1404
+ return false if defined?(TurnFinalizer) && TurnFinalizer.user_path? &&
1405
+ Thread.current[:pwn_turn_finalizer_depth].to_i > 1
1406
+
1195
1407
  return true unless opts[:local]
1196
1408
 
1197
1409
  policy = agent_flag(key: :local_introspect, default: :failure_only).to_s.to_sym
@@ -1487,6 +1699,11 @@ module PWN
1487
1699
  r = mod.chat(
1488
1700
  request: request,
1489
1701
  system_role_content: q_sys,
1702
+ response_history: chat_response_history(
1703
+ session_id: session_id,
1704
+ request: request,
1705
+ system_role_content: q_sys
1706
+ ),
1490
1707
  spinner: false
1491
1708
  )
1492
1709
  if r.is_a?(Hash)
@@ -1550,6 +1767,11 @@ module PWN
1550
1767
  r = mod.chat(
1551
1768
  request: request,
1552
1769
  system_role_content: q_sys,
1770
+ response_history: chat_response_history(
1771
+ session_id: session_id,
1772
+ request: request,
1773
+ system_role_content: q_sys
1774
+ ),
1553
1775
  spinner: false
1554
1776
  )
1555
1777
  if r.is_a?(Hash)
@@ -1569,6 +1791,9 @@ module PWN
1569
1791
  txt = txt.to_s.strip
1570
1792
  txt = 'I do not have enough context to answer that yet.' if txt.empty?
1571
1793
 
1794
+ debug_progress(msg: "final accepted chars=#{txt.length}")
1795
+ quiet_debug_tui!(reason: 'question')
1796
+ debug_final_text!(text: txt)
1572
1797
  append_session(session_id: session_id, role: 'user', content: request)
1573
1798
  append_session(session_id: session_id, role: 'assistant', content: txt)
1574
1799
  if defined?(Learning) && should_auto_introspect?(local: local_engine?, turn_fails: {}, iter: 0)
@@ -1583,6 +1808,7 @@ module PWN
1583
1808
  end
1584
1809
  txt
1585
1810
  rescue StandardError => e
1811
+ quiet_debug_tui!(reason: 'question_error')
1586
1812
  warn "[pwn-ai/loop] answer_question swallowed: #{e.class}: #{e.message}"
1587
1813
  "Could not answer the question (#{e.class}: #{e.message})."
1588
1814
  end
@@ -1637,6 +1863,11 @@ module PWN
1637
1863
  r = mod.chat(
1638
1864
  request: request,
1639
1865
  system_role_content: howto_sys,
1866
+ response_history: chat_response_history(
1867
+ session_id: session_id,
1868
+ request: request,
1869
+ system_role_content: howto_sys
1870
+ ),
1640
1871
  spinner: false
1641
1872
  )
1642
1873
  if r.is_a?(Hash)
@@ -1647,9 +1878,10 @@ module PWN
1647
1878
  else
1648
1879
  # chat_with_tools without tools
1649
1880
  messages = [
1650
- { role: 'system', content: howto_sys },
1651
- { role: 'user', content: request }
1881
+ { role: 'system', content: howto_sys }
1652
1882
  ]
1883
+ messages.concat(session_chat_history(session_id: session_id, skip_request: request))
1884
+ messages << { role: 'user', content: request }
1653
1885
  msg = call_engine(messages: messages, tools: nil)
1654
1886
  msg.is_a?(Hash) ? msg[:content].to_s : msg.to_s
1655
1887
  end
@@ -1933,6 +2165,11 @@ module PWN
1933
2165
  r = mod.chat(
1934
2166
  request: request,
1935
2167
  system_role_content: recall_sys,
2168
+ response_history: chat_response_history(
2169
+ session_id: session_id,
2170
+ request: request,
2171
+ system_role_content: recall_sys
2172
+ ),
1936
2173
  spinner: false
1937
2174
  )
1938
2175
  if r.is_a?(Hash)
@@ -1983,6 +2220,14 @@ module PWN
1983
2220
  request = opts[:request].to_s
1984
2221
  session_id = opts[:session_id]
1985
2222
  on_tool = opts[:on_tool]
2223
+ i = 0
2224
+ tools_called = 0
2225
+ engine_s = 0.0
2226
+ final_chars = 0
2227
+ start_debug_session(opts)
2228
+ loud_debug_tui!(debug: opts[:debug])
2229
+ debug_progress(msg: "Loop.run start request=#{request[0, 240]}", debug: opts[:debug])
2230
+ nested = defined?(TurnFinalizer) && TurnFinalizer.user_path?
1986
2231
  TurnFinalizer.enter_user_path! if defined?(TurnFinalizer)
1987
2232
  engine = active_engine
1988
2233
  local = local_engine?(engine: engine)
@@ -1997,27 +2242,31 @@ module PWN
1997
2242
  opts[:request] = request
1998
2243
  intent = request_intent(request: request)
1999
2244
  end
2000
- elsif defined?(OpenGoal) && needs_host_work?(request: request) &&
2245
+ elsif !nested && defined?(OpenGoal) && needs_host_work?(request: request) &&
2001
2246
  !%i[greeting howto recall].include?(intent)
2002
2247
  OpenGoal.begin!(request: request, session_id: session_id)
2003
2248
  end
2004
2249
  Thread.current[:pwn_request_intent] = intent
2005
2250
  Thread.current[:pwn_extinguished] = {}
2251
+ debug_progress(msg: "intent=#{intent} engine=#{engine}", debug: opts[:debug])
2006
2252
  expose_current_session(session_id: session_id)
2007
2253
  Mistakes.check_user_correction(request: request, session_id: session_id) if defined?(Mistakes)
2008
2254
 
2009
2255
  cheap = opts[:force_tools] != true && %i[greeting howto recall].include?(intent)
2010
2256
 
2011
- # Greeting / light smalltalk: deterministic ack - no weather echo, no tools,
2012
- # no PromptBuilder, no Registry.
2013
2257
  if intent == :greeting && opts[:force_tools] != true
2014
- return answer_greeting(
2258
+ debug_progress(msg: 'path=greeting', debug: opts[:debug])
2259
+ quiet_debug_tui!(debug: opts[:debug], reason: 'greeting')
2260
+ txt = answer_greeting(
2015
2261
  request: request,
2016
2262
  session_id: session_id
2017
2263
  )
2264
+ final_chars = txt.to_s.length
2265
+ debug_final_text!(text: txt, debug: opts[:debug])
2266
+ return txt
2018
2267
  end
2019
2268
 
2020
- # Thin system prompt only for remaining cheap paths (howto/recall/statement/question).
2269
+ # Thin system prompt only for remaining cheap paths (howto/recall).
2021
2270
  if cheap
2022
2271
  system_role_content = opts[:system_role_content]
2023
2272
  if system_role_content.nil? || system_role_content.to_s.empty?
@@ -2028,21 +2277,29 @@ module PWN
2028
2277
  )
2029
2278
  opts[:system_role_content] = system_role_content
2030
2279
  end
2031
- # How-to: never enter plan_first / task recon / tool thrash.
2032
2280
  if intent == :howto
2033
- return answer_howto(
2281
+ debug_progress(msg: 'path=howto', debug: opts[:debug])
2282
+ quiet_debug_tui!(debug: opts[:debug], reason: 'howto')
2283
+ txt = answer_howto(
2034
2284
  request: request,
2035
2285
  session_id: session_id,
2036
2286
  system_role_content: system_role_content
2037
2287
  )
2288
+ final_chars = txt.to_s.length
2289
+ debug_final_text!(text: txt, debug: opts[:debug])
2290
+ return txt
2038
2291
  end
2039
- # Pure prior-turn / vague memory recall: one cheap path, no plan_first.
2040
2292
  if intent == :recall
2041
- return answer_recall(
2293
+ debug_progress(msg: 'path=recall', debug: opts[:debug])
2294
+ quiet_debug_tui!(debug: opts[:debug], reason: 'recall')
2295
+ txt = answer_recall(
2042
2296
  request: request,
2043
2297
  session_id: session_id,
2044
2298
  system_role_content: system_role_content
2045
2299
  )
2300
+ final_chars = txt.to_s.length
2301
+ debug_final_text!(text: txt, debug: opts[:debug])
2302
+ return txt
2046
2303
  end
2047
2304
  end
2048
2305
 
@@ -2083,13 +2340,24 @@ module PWN
2083
2340
  core_only: core_only,
2084
2341
  intent: intent
2085
2342
  )
2343
+ no_tools = Array(tools).empty?
2344
+ Thread.current[:pwn_loop_no_tools] = no_tools
2086
2345
  messages = [{ role: 'system', content: system_role_content }]
2087
2346
  messages.concat(Learning.exemplars_for(request: request)) if local && defined?(Learning) && Learning.respond_to?(:exemplars_for)
2347
+ messages.concat(
2348
+ session_chat_history(session_id: session_id, skip_request: request)
2349
+ )
2088
2350
  messages << { role: 'user', content: request }
2089
2351
  append_session(session_id: session_id, role: 'user', content: request)
2090
2352
 
2091
- # Every request gets a task compass.
2092
- task_summary_plan!(state: ts_state, request: request, on_tool: on_tool) if defined?(TaskSummarizer)
2353
+ trivia = world_knowledge?(request: request)
2354
+ catalog = catalog_lookup?(request: request)
2355
+ browse = request_need(request: request) == :browse
2356
+ skip_compass = trivia || catalog || no_tools || browse
2357
+ # Trivia / catalog / browse do not get an implement-shaped
2358
+ # English compass. Inventing "apply code/host changes" there
2359
+ # keeps the model on a Navigate task after the page already loaded.
2360
+ task_summary_plan!(state: ts_state, request: request, on_tool: on_tool) if defined?(TaskSummarizer) && !skip_compass
2093
2361
  # Re-bind tools from English plan so task list is the sole driver of
2094
2362
  # tool exposure/ranking (Registry keyword router + CORE).
2095
2363
  if ts_state.is_a?(Hash) && defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:relevance_query)
@@ -2103,16 +2371,17 @@ module PWN
2103
2371
  )
2104
2372
  end
2105
2373
  end
2106
- # English-task-as-primary: inject tangible tasks only for autonomous goals.
2107
- inject_task_focus!(messages: messages, state: ts_state, force: true, request: request)
2374
+ # English-task-as-primary: inject tangible tasks only for host work.
2375
+ inject_task_focus!(messages: messages, state: ts_state, force: true, request: request) unless skip_compass
2108
2376
  predicted = nil
2109
2377
  Thread.current[:pwn_plan_predicted] = nil
2110
2378
  cal_state = calibration_state
2111
2379
  force_plan = cal_state[:force_plan]
2112
- # Skip plan_first only for remaining cheap intents.
2113
- skip_plan = %i[howto recall greeting].include?(intent)
2380
+ skip_plan = %i[howto recall greeting].include?(intent) || trivia || catalog || no_tools || browse
2381
+ did_plan = false
2114
2382
  if !skip_plan && (force_plan || agent_flag(key: :plan_first, default: local) || budget_exhaustion_hot?) && !Array(tools).empty?
2115
2383
  predicted = plan_first(messages: messages, request: request, ts_state: ts_state)
2384
+ did_plan = true
2116
2385
  # P22 — prefer explicit return; fall back to thread stash
2117
2386
  predicted = Thread.current[:pwn_plan_predicted] if predicted.nil?
2118
2387
  # unify_plan! may have rewritten English tasks — force refresh focus.
@@ -2129,25 +2398,10 @@ module PWN
2129
2398
  )
2130
2399
  end
2131
2400
  end
2132
- inject_task_focus!(messages: messages, state: ts_state, force: true, request: request)
2401
+ inject_task_focus!(messages: messages, state: ts_state, force: true, request: request) unless skip_compass
2133
2402
  end
2134
- if budget_exhaustion_hot?
2135
- english_open = defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:plan_open?) &&
2136
- TaskSummarizer.plan_open?(state: ts_state, messages: messages)
2137
- hot_hint = if local_engine? && !english_open
2138
- '[pwn-ai/p17] Budget-exhaustion is the top open failure on this host. ' \
2139
- 'Prefer the SHORTEST plan that finishes the ask (≤3 tool calls). ' \
2140
- 'Emit a final answer as soon as you have evidence — do not explore.'
2141
- else
2142
- '[pwn-ai/p17] Budget-exhaustion is the top open failure on this host. ' \
2143
- 'Prefer the shortest plan that FULLY finishes the ask — no polite ' \
2144
- 'handoffs, no exploration side-quests. Emit a final answer as soon ' \
2145
- 'as you have evidence; keep going with tools until the goal is done ' \
2146
- 'or truly blocked.'
2147
- end
2148
- messages << { role: 'user', content: hot_hint }
2149
- end
2150
- if force_plan && cal_state[:cal]
2403
+ debug_progress(msg: "plan_first=#{did_plan} trivia=#{trivia} catalog=#{catalog} browse=#{browse}")
2404
+ if force_plan && cal_state[:cal] && !skip_plan
2151
2405
  messages << {
2152
2406
  role: 'user',
2153
2407
  content: "[pwn-ai/w3] engine=#{active_engine} is overconfident " \
@@ -2161,70 +2415,27 @@ module PWN
2161
2415
  maybe_park_budget_scars!
2162
2416
  maybe_extinguish_parked!
2163
2417
 
2164
- max_iters.times do |i|
2418
+ i = 0
2419
+ loop do
2420
+ i += 1
2165
2421
  # 3.1 — compact history on local so tool dumps don't fill num_ctx
2166
2422
  compact_history!(messages: messages) if local
2167
2423
  # English-task-as-primary: when plan_idx advanced, tell the model
2168
2424
  # which plain-English task is active before the next tool batch.
2169
- inject_task_focus!(messages: messages, state: ts_state, request: request)
2170
-
2171
- # P17 — on the final iteration, strip tools and demand a plain-text
2172
- # answer. Without this the model happily emits one more tool_calls
2173
- # batch, burns the last slot, and lands on budget_exhausted with
2174
- # nothing the user (or ORM) can use.
2175
- # P17 deepen — when budget_hot, force text-only on the LAST TWO
2176
- # iters so a final tool_calls batch cannot burn the terminal slot.
2177
- # P17 deepen³ — under hot, force text-only on last THREE of the
2178
- # 8-iter cap so a late tool binge cannot burn every salvage slot.
2179
- # P17 structural: default hot text-only tail stays 3 (do NOT deepen to 4/6).
2180
- # Plan-faithful headroom — short plan executing cleanly → delay strip to
2181
- # last 1–2 so multi-step goals are not predestined to exhaust under cap 8.
2182
- hot = budget_exhaustion_hot?
2183
- plan_steps = begin
2184
- predicted_plan = predicted || Thread.current[:pwn_plan_predicted]
2185
- if predicted_plan.is_a?(Hash)
2186
- Array(predicted_plan[:steps] || predicted_plan[:tools] || predicted_plan[:plan]).size
2187
- elsif predicted_plan.is_a?(Array)
2188
- predicted_plan.size
2189
- else
2190
- predicted_plan.to_s.scan(/\b(?:shell|pwn_eval|memory_|mistakes_|skill_|extro_|learning_|sessions_)\w*/).size
2191
- end
2192
- rescue StandardError
2193
- 0
2194
- end
2195
- # Plan-faithful: delay the text-only strip when a plan is executing
2196
- # cleanly. Remote hot allows longer plans (runway 25); local hot
2197
- # still favors short plans under the 8-iter cap.
2198
- plan_step_limit = local_engine? ? 3 : 12
2199
- plan_faithful = hot && plan_steps.positive? && plan_steps <= plan_step_limit &&
2200
- turn_fails['empty_final'].to_i.zero? &&
2201
- turn_fails.values.sum < 2
2202
- # Last-iter strips tools only on the true last slot. English
2203
- # leftovers and budget-hot must not steal runway from a live goal.
2204
- text_only_iters = 1
2205
- want_last = (i >= max_iters - text_only_iters)
2206
- still_open = request_unsatisfied?(
2207
- request: request,
2208
- messages: messages,
2209
- last_iter: false
2210
- )
2211
- last_iter = want_last && !still_open
2212
- if last_iter
2213
- tag = i >= max_iters - 1 ? 'FINAL ITERATION' : 'PENULTIMATE — wrap up'
2214
- messages << {
2215
- role: 'user',
2216
- content: "[pwn-ai/p17] #{tag} — do NOT call any more tools. " \
2217
- 'Write the complete answer from evidence already in this ' \
2218
- 'transcript. If unfinished, say BLOCKED with evidence — not a ' \
2219
- 'markdown outline or "# Remaining block". Do NOT ask the user ' \
2220
- 'to confirm the next step.'
2221
- }
2222
- end
2425
+ inject_task_focus!(messages: messages, state: ts_state, request: request) unless skip_compass
2223
2426
 
2224
- msg = call_engine(messages: messages, tools: last_iter ? nil : tools, ts_state: ts_state)
2427
+ t0 = Time.now
2428
+ msg = call_engine(messages: messages, tools: tools, ts_state: ts_state)
2429
+ engine_s += (Time.now - t0)
2430
+ PWN::Plugins::TTYSpinner.halt_all! if defined?(PWN::Plugins::TTYSpinner)
2225
2431
  if msg.nil?
2226
2432
  task_summary_flush!(state: ts_state, on_tool: on_tool)
2227
- return '[pwn-ai] engine returned no message'
2433
+ debug_progress(msg: 'engine returned no message')
2434
+ quiet_debug_tui!(reason: 'engine_empty')
2435
+ txt = '[pwn-ai] engine returned no message'
2436
+ debug_final_text!(text: txt)
2437
+ final_chars = txt.length
2438
+ return txt
2228
2439
  end
2229
2440
 
2230
2441
  calls = Array(msg[:tool_calls])
@@ -2232,7 +2443,7 @@ module PWN
2232
2443
 
2233
2444
  # Belt-and-suspenders: plain-text shell(...) / tool forms from local
2234
2445
  # models under weak TEMPLATE {{ .Prompt }} become real tool_calls.
2235
- if calls.empty? && !text.strip.empty? && !last_iter &&
2446
+ if calls.empty? && !text.strip.empty? &&
2236
2447
  defined?(Dispatch) && Dispatch.respond_to?(:tool_calls_from_text)
2237
2448
  coerced = Dispatch.tool_calls_from_text(text: text)
2238
2449
  if coerced.any?
@@ -2258,6 +2469,7 @@ module PWN
2258
2469
  'Do not reply with an empty message.'
2259
2470
  }
2260
2471
  turn_fails['empty_final'] += 1
2472
+ debug_progress(msg: "bounce empty_final snippet=#{debug_snippet(text: text)}")
2261
2473
  next
2262
2474
  end
2263
2475
 
@@ -2268,6 +2480,7 @@ module PWN
2268
2480
  if incomplete_final?(text: text, last_iter: false)
2269
2481
  turn_fails['incomplete_final'] += 1
2270
2482
  warn "[pwn-ai/loop] incomplete final on iter=#{i}; continuing autonomously"
2483
+ debug_progress(msg: "bounce incomplete_final snippet=#{debug_snippet(text: text)}")
2271
2484
  messages << {
2272
2485
  role: 'user',
2273
2486
  content: '[pwn-ai/p28] That reply was incomplete (handoff or narrated next step). ' \
@@ -2285,6 +2498,7 @@ module PWN
2285
2498
  )
2286
2499
  turn_fails['unsatisfied'] += 1
2287
2500
  warn "[pwn-ai/loop] original request not evidenced on iter=#{i}; continuing"
2501
+ debug_progress(msg: "bounce unsatisfied snippet=#{debug_snippet(text: text)}")
2288
2502
  messages << {
2289
2503
  role: 'user',
2290
2504
  content: '[pwn-ai] The original request is not evidenced yet. ' \
@@ -2293,11 +2507,15 @@ module PWN
2293
2507
  }
2294
2508
  next
2295
2509
  end
2510
+ debug_progress(msg: "final accepted chars=#{text.to_s.length}")
2511
+ quiet_debug_tui!(reason: 'final')
2512
+ debug_final_text!(text: text)
2513
+ final_chars = text.to_s.length
2296
2514
  append_session(session_id: session_id, role: 'assistant', content: text)
2297
- Learning.auto_introspect(session_id: session_id, request: request, final: text, predicted: predicted, plan: ts_state && ts_state[:plan], ts_state: ts_state) if defined?(Learning) && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: i)
2515
+ Learning.auto_introspect(session_id: session_id, request: request, final: text, predicted: predicted, plan: ts_state && ts_state[:plan], ts_state: ts_state) if defined?(Learning) && !nested && !no_tools && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: i)
2298
2516
  maybe_finish_policy(session_id: session_id, proxy_ok: true, ts_state: ts_state)
2299
2517
  task_summary_flush!(state: ts_state, on_tool: on_tool)
2300
- OpenGoal.clear! if defined?(OpenGoal)
2518
+ OpenGoal.clear! if defined?(OpenGoal) && !nested
2301
2519
  return text
2302
2520
  end
2303
2521
 
@@ -2320,6 +2538,8 @@ module PWN
2320
2538
  args = tc.dig(:function, :arguments)
2321
2539
  entry = Registry.lookup(name: name)
2322
2540
  started = Time.now
2541
+ argv_s = args.is_a?(String) ? args.to_s : args.inspect
2542
+ debug_progress(msg: "tool #{name} start:\n#{argv_s}", keep_newlines: true, cap: 0, tee: nil)
2323
2543
  if Thread.current[:pwn_extinguished].is_a?(Hash) && Thread.current[:pwn_extinguished][name]
2324
2544
  raw = JSON.generate(
2325
2545
  success: false,
@@ -2329,6 +2549,7 @@ module PWN
2329
2549
  else
2330
2550
  raw = Dispatch.call(tool_call: tc)
2331
2551
  end
2552
+ tools_called += 1
2332
2553
  tele = record_metrics(name: name, started: started, raw: raw, args: args, session_id: session_id, engine: engine, ts_state: ts_state)
2333
2554
  result = Result.condition(content: raw, entry: entry)
2334
2555
 
@@ -2353,6 +2574,7 @@ module PWN
2353
2574
  end
2354
2575
 
2355
2576
  on_tool&.call(name, args, result)
2577
+ debug_tool_io!(name: name, args: args, result: result)
2356
2578
  task_summary_record!(state: ts_state, name: name, args: args, result: result, on_tool: on_tool)
2357
2579
 
2358
2580
  messages << {
@@ -2380,26 +2602,29 @@ module PWN
2380
2602
  end
2381
2603
  escalated = true
2382
2604
  end
2383
-
2384
- # P17 — exhaust path must still feed Learning so ORM/PRM/HER see the
2385
- # failure (previously we only Mistakes.record'd and returned a bare
2386
- # string — no session row, no judge, no hindsight).
2387
- final_msg = '[pwn-ai] iteration budget exhausted'
2388
- if defined?(Mistakes)
2389
- Mistakes.record(
2390
- tool: 'agent_loop',
2391
- error: 'iteration budget exhausted without a final answer',
2392
- session_id: session_id,
2393
- source: :loop,
2394
- shape: :budget_exhausted
2395
- )
2605
+ rescue Interrupt
2606
+ Thread.current[:pwn_log_progress] = false
2607
+ if defined?(PWN::Plugins::Log) && PWN::Plugins::Log.respond_to?(:note_interrupt!)
2608
+ PWN::Plugins::Log.note_interrupt!(where: 'CTRL+C', which_self: self)
2609
+ else
2610
+ debug_progress(msg: 'Interrupt CTRL+C')
2611
+ end
2612
+ raise
2613
+ rescue StandardError => e
2614
+ if defined?(PWN::Plugins::Log) && PWN::Plugins::Log.respond_to?(:note_exception!)
2615
+ PWN::Plugins::Log.note_exception!(error: e, where: 'Loop.run', which_self: self)
2616
+ else
2617
+ debug_progress(msg: "exception Loop.run #{e.class}: #{e.message}\n#{Array(e.backtrace).join("\n")}", keep_newlines: true, cap: 0)
2396
2618
  end
2397
- append_session(session_id: session_id, role: 'assistant', content: final_msg)
2398
- Learning.auto_introspect(session_id: session_id, request: request, final: final_msg, predicted: predicted, plan: ts_state && ts_state[:plan], ts_state: ts_state) if defined?(Learning) && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: max_iters)
2399
- maybe_finish_policy(session_id: session_id, proxy_ok: false, ts_state: ts_state)
2400
- task_summary_flush!(state: ts_state, on_tool: on_tool)
2401
- final_msg
2619
+ raise
2402
2620
  ensure
2621
+ Thread.current[:pwn_loop_no_tools] = nil
2622
+ finish_debug_request!(
2623
+ iter: i,
2624
+ tools_called: tools_called,
2625
+ engine_s: engine_s,
2626
+ final_chars: final_chars
2627
+ )
2403
2628
  TurnFinalizer.leave_user_path! if defined?(TurnFinalizer)
2404
2629
  end
2405
2630
 
@@ -2433,7 +2658,7 @@ module PWN
2433
2658
  # task_summary_verbose: false
2434
2659
 
2435
2660
  Supported engines: #{ENGINE_MODS.keys.join(', ')}
2436
- Set PWN::Env[:ai][:active] to choose; PWN::Env[:ai][:agent][:max_iters] to bound.
2661
+ Set PWN::Env[:ai][:active] to choose.
2437
2662
 
2438
2663
  Intent routing (all engines; critical for ollama/openwebui):
2439
2664
  how-to / usage questions → text-only explanation (no tools, no plan_first)
@@ -2453,8 +2678,8 @@ module PWN
2453
2678
  :verify_as_reward - E3 ground every final via extro_verify (Boolean)
2454
2679
 
2455
2680
  P28 autonomy: incomplete-final detector refuses mid-goal handoffs.
2456
- max_iters is Env or DEFAULT_MAX_ITERS (#{DEFAULT_MAX_ITERS});
2457
- budget-hot / overconf do not shrink this request's runway.
2681
+ Loop.run keeps CORE_TOOLS until may_finalize? — there is no
2682
+ iteration-budget abort.
2458
2683
 
2459
2684
  #{self}.authors
2460
2685
  USAGE