pwn 0.5.686 → 0.5.688
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.gitignore +2 -0
- data/.ruby-version +1 -1
- data/Gemfile +3 -3
- data/README.md +3 -4
- data/documentation/Agent-Tool-Registry.md +2 -2
- data/documentation/Configuration.md +5 -4
- data/documentation/General-PWN-Usage.md +1 -1
- data/documentation/Home.md +1 -1
- data/documentation/How-PWN-Works.md +1 -1
- data/documentation/Mistakes.md +1 -1
- data/documentation/Reinforcement-Learning.md +4 -6
- data/documentation/Troubleshooting.md +1 -1
- data/documentation/pwn-REPL.md +1 -1
- data/documentation/pwn-ai-Agent.md +2 -3
- data/lib/pwn/ai/agent/curriculum.rb +2 -2
- data/lib/pwn/ai/agent/dispatch.rb +1 -1
- data/lib/pwn/ai/agent/learning.rb +150 -32
- data/lib/pwn/ai/agent/loop.rb +291 -129
- data/lib/pwn/ai/agent/prompt_builder.rb +47 -29
- data/lib/pwn/ai/agent/registry.rb +2 -1
- data/lib/pwn/ai/agent/reward.rb +9 -0
- data/lib/pwn/ai/agent/swarm.rb +3 -1
- data/lib/pwn/ai/agent/task_summarizer.rb +23 -6
- data/lib/pwn/ai/agent/tool_guard.rb +109 -0
- data/lib/pwn/ai/agent/tools/ruby_eval.rb +44 -20
- data/lib/pwn/ai/agent/tools/shell.rb +17 -4
- data/lib/pwn/ai/agent/tools/skills.rb +63 -2
- data/lib/pwn/ai/anthropic.rb +25 -5
- data/lib/pwn/ai/gemini.rb +25 -5
- data/lib/pwn/ai/grok.rb +30 -13
- data/lib/pwn/ai/http_retry.rb +78 -0
- data/lib/pwn/ai/ollama.rb +25 -5
- data/lib/pwn/ai/open_ai.rb +26 -13
- data/lib/pwn/ai/open_web_ui.rb +25 -5
- data/lib/pwn/ai.rb +1 -0
- data/lib/pwn/config.rb +1 -1
- data/lib/pwn/plugins/log.rb +183 -41
- data/lib/pwn/plugins/repl.rb +50 -15
- data/lib/pwn/plugins/transparent_browser.rb +16 -2
- data/lib/pwn/sessions.rb +50 -1
- data/lib/pwn/version.rb +1 -1
- data/spec/integration/reinforced_feedback_loop_spec.rb +9 -11
- data/spec/lib/pwn/ai/agent/loop_spec.rb +180 -20
- data/spec/lib/pwn/ai/agent/prompt_builder_spec.rb +3 -0
- data/spec/lib/pwn/ai/agent/registry_spec.rb +1 -1
- data/spec/lib/pwn/ai/agent/reward_spec.rb +13 -0
- data/spec/lib/pwn/ai/agent/swarm_spec.rb +6 -0
- data/spec/lib/pwn/ai/agent/task_summarizer_spec.rb +36 -0
- data/spec/lib/pwn/ai/agent/tool_guard_spec.rb +46 -0
- data/spec/lib/pwn/ai/agent/tools/ruby_eval_spec.rb +24 -0
- data/spec/lib/pwn/ai/agent/tools/shell_spec.rb +10 -0
- data/spec/lib/pwn/ai/agent/tools/skills_spec.rb +29 -0
- data/spec/lib/pwn/ai/grok_spec.rb +22 -0
- data/spec/lib/pwn/ai/http_retry_spec.rb +55 -0
- data/spec/lib/pwn/plugins/log_spec.rb +170 -19
- data/spec/lib/pwn/plugins/repl_pwn_vault_spec.rb +14 -4
- data/spec/lib/pwn/plugins/transparent_browser_spec.rb +8 -0
- data/spec/lib/pwn/sessions_spec.rb +24 -0
- data/third_party/pwn_rdoc.jsonl +41 -5
- data/tmp_pwn_critic_skills_flaw.txt +9 -0
- metadata +11 -8
data/lib/pwn/ai/agent/loop.rb
CHANGED
|
@@ -77,18 +77,109 @@ module PWN
|
|
|
77
77
|
return false unless debug_on?(opts)
|
|
78
78
|
return false unless defined?(PWN::Plugins::Log)
|
|
79
79
|
|
|
80
|
-
|
|
80
|
+
payload = {
|
|
81
81
|
msg: opts[:msg],
|
|
82
|
-
which_self: opts[:which_self] || self
|
|
82
|
+
which_self: opts[:which_self] || self,
|
|
83
|
+
keep_newlines: opts[:keep_newlines]
|
|
84
|
+
}
|
|
85
|
+
payload[:cap] = opts[:cap] if opts.key?(:cap)
|
|
86
|
+
payload[:tee] = opts[:tee] if opts.key?(:tee)
|
|
87
|
+
PWN::Plugins::Log.progress(payload)
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
private_class_method def self.debug_tools_line(opts = {})
|
|
91
|
+
tools = opts[:tools]
|
|
92
|
+
return 'tools=none' if tools.nil?
|
|
93
|
+
|
|
94
|
+
names = Array(tools).filter_map do |tool|
|
|
95
|
+
next unless tool.is_a?(Hash)
|
|
96
|
+
|
|
97
|
+
tool.dig(:function, :name) ||
|
|
98
|
+
tool.dig('function', 'name') ||
|
|
99
|
+
tool[:name] ||
|
|
100
|
+
tool['name']
|
|
101
|
+
end.map(&:to_s).reject(&:empty?)
|
|
102
|
+
"tools=#{names.length}[#{names.join(',')}]"
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
private_class_method def self.debug_msgs_line(opts = {})
|
|
106
|
+
bits = Array(opts[:messages]).map do |msg|
|
|
107
|
+
role = (msg[:role] || msg['role'] || '?').to_s
|
|
108
|
+
len = (msg[:content] || msg['content']).to_s.length
|
|
109
|
+
"#{role}:#{len}"
|
|
110
|
+
end
|
|
111
|
+
"msgs=#{bits.length}[#{bits.join(',')}]"
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
private_class_method def self.debug_snippet(opts = {})
|
|
115
|
+
text = opts[:text].to_s.tr("\n", ' ').strip
|
|
116
|
+
max = opts[:max].to_i
|
|
117
|
+
max = 400 unless max.positive?
|
|
118
|
+
text.length > max ? "#{text[0, max]}…" : text
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
private_class_method def self.debug_final_text!(opts = {})
|
|
122
|
+
return unless debug_on?(opts)
|
|
123
|
+
|
|
124
|
+
debug_progress(
|
|
125
|
+
msg: "final text:\n#{opts[:text]}",
|
|
126
|
+
keep_newlines: true,
|
|
127
|
+
cap: 65_536,
|
|
128
|
+
debug: opts[:debug]
|
|
129
|
+
)
|
|
130
|
+
end
|
|
131
|
+
|
|
132
|
+
private_class_method def self.debug_tool_io!(opts = {})
|
|
133
|
+
return unless debug_on?(opts)
|
|
134
|
+
|
|
135
|
+
name = opts[:name].to_s
|
|
136
|
+
argv = opts[:args].is_a?(String) ? opts[:args].to_s : opts[:args].inspect
|
|
137
|
+
result = opts[:result].to_s
|
|
138
|
+
debug_progress(
|
|
139
|
+
msg: "tool #{name} request:\n#{argv}\nresult:\n#{result}",
|
|
140
|
+
keep_newlines: true,
|
|
141
|
+
cap: 0,
|
|
142
|
+
tee: nil,
|
|
143
|
+
debug: opts[:debug]
|
|
83
144
|
)
|
|
84
145
|
end
|
|
85
146
|
|
|
86
147
|
private_class_method def self.start_debug_session(opts = {})
|
|
87
148
|
return unless debug_on?(opts)
|
|
88
149
|
return unless defined?(PWN::Plugins::Log)
|
|
89
|
-
return if PWN::Plugins::Log.debug_enabled?
|
|
90
150
|
|
|
91
|
-
|
|
151
|
+
if defined?(TurnFinalizer) && TurnFinalizer.user_path?
|
|
152
|
+
debug_progress(msg: 'nested Loop.run skip_roll', debug: opts[:debug])
|
|
153
|
+
return
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
unless PWN::Plugins::Log.debug_enabled?
|
|
157
|
+
want_trace = opts[:trace] == true
|
|
158
|
+
begin
|
|
159
|
+
want_trace ||= PWN::Env.dig(:ai, :agent, :debug_trace) == true
|
|
160
|
+
rescue StandardError
|
|
161
|
+
nil
|
|
162
|
+
end
|
|
163
|
+
PWN::Plugins::Log.start_debug(
|
|
164
|
+
tee: opts[:debug_tee] || $stdout,
|
|
165
|
+
session_id: opts[:session_id],
|
|
166
|
+
trace: want_trace
|
|
167
|
+
)
|
|
168
|
+
end
|
|
169
|
+
PWN::Plugins::Log.next_request_log!(session_id: opts[:session_id])
|
|
170
|
+
end
|
|
171
|
+
|
|
172
|
+
private_class_method def self.finish_debug_request!(opts = {})
|
|
173
|
+
return unless debug_on?(opts)
|
|
174
|
+
return unless defined?(PWN::Plugins::Log)
|
|
175
|
+
|
|
176
|
+
PWN::Plugins::Log.finish_request_log!(
|
|
177
|
+
iter: opts[:iter],
|
|
178
|
+
tools_called: opts[:tools_called],
|
|
179
|
+
engine_s: opts[:engine_s],
|
|
180
|
+
final_chars: opts[:final_chars],
|
|
181
|
+
nested: opts[:nested]
|
|
182
|
+
)
|
|
92
183
|
end
|
|
93
184
|
|
|
94
185
|
private_class_method def self.quiet_debug_tui!(opts = {})
|
|
@@ -369,6 +460,10 @@ module PWN
|
|
|
369
460
|
LOOKUP_REQUEST_RX = /
|
|
370
461
|
\b(what\s+is\s+my|hostname|uname|cwd|whoami|status|version|how\s+many)\b
|
|
371
462
|
/ix
|
|
463
|
+
SKILLS_CATALOG_RX = /
|
|
464
|
+
\bskills?\b.{0,40}\b(available|installed|loaded|catalog|list)\b |
|
|
465
|
+
\b(what|which|list)\b.{0,40}\bskills?\b
|
|
466
|
+
/ix
|
|
372
467
|
# Real filesystem paths only — not https://host.tld (that was matching //host.tld).
|
|
373
468
|
HOST_PATH_RX = %r{(?:(?<![.:/])/(?!/)|\./)[\w./-]+\.\w+}
|
|
374
469
|
BROWSER_REQUEST_RX = /
|
|
@@ -378,10 +473,23 @@ module PWN
|
|
|
378
473
|
|
|
379
474
|
# True only when the ask needs a live host/file/browser effect. World-knowledge
|
|
380
475
|
# questions ("what color is a cherry") do not.
|
|
476
|
+
public_class_method def self.catalog_lookup?(opts = {})
|
|
477
|
+
request = opts[:request].to_s.strip
|
|
478
|
+
return false if request.empty?
|
|
479
|
+
return false if request.length > 120
|
|
480
|
+
return false if request.match?(ACT_REQUEST_RX)
|
|
481
|
+
return false if request.match?(HOWTO_RX)
|
|
482
|
+
|
|
483
|
+
request.match?(SKILLS_CATALOG_RX)
|
|
484
|
+
rescue StandardError
|
|
485
|
+
false
|
|
486
|
+
end
|
|
487
|
+
|
|
381
488
|
public_class_method def self.world_knowledge?(opts = {})
|
|
382
489
|
request = opts[:request].to_s.strip
|
|
383
490
|
return false if request.empty?
|
|
384
491
|
return false if request.length > 120
|
|
492
|
+
return false if catalog_lookup?(request: request)
|
|
385
493
|
return false if request.match?(ACT_REQUEST_RX)
|
|
386
494
|
return false if request.match?(LOOKUP_REQUEST_RX)
|
|
387
495
|
return false if request.match?(HOST_PATH_RX)
|
|
@@ -398,6 +506,7 @@ module PWN
|
|
|
398
506
|
request = opts[:request].to_s
|
|
399
507
|
return false if request.strip.empty?
|
|
400
508
|
return false if world_knowledge?(request: request)
|
|
509
|
+
return false if catalog_lookup?(request: request)
|
|
401
510
|
|
|
402
511
|
true
|
|
403
512
|
rescue StandardError
|
|
@@ -407,6 +516,7 @@ module PWN
|
|
|
407
516
|
private_class_method def self.request_need(opts = {})
|
|
408
517
|
request = opts[:request].to_s
|
|
409
518
|
return :none if world_knowledge?(request: request)
|
|
519
|
+
return :read if catalog_lookup?(request: request)
|
|
410
520
|
return :read if request.match?(LOOKUP_REQUEST_RX)
|
|
411
521
|
return :browse if request.match?(BROWSER_REQUEST_RX)
|
|
412
522
|
return :write if request.match?(ACT_REQUEST_RX) || request.match?(HOST_PATH_RX)
|
|
@@ -454,9 +564,12 @@ module PWN
|
|
|
454
564
|
request = opts[:request].to_s
|
|
455
565
|
need = request_need(request: request)
|
|
456
566
|
return false if need == :none
|
|
457
|
-
return false
|
|
567
|
+
return false if Thread.current[:pwn_loop_no_tools]
|
|
458
568
|
|
|
459
569
|
effects = tool_effects(messages: opts[:messages])
|
|
570
|
+
return !effects.intersect?(%i[read recall eval]) if need == :read && catalog_lookup?(request: request)
|
|
571
|
+
return false unless needs_host_work?(request: request)
|
|
572
|
+
|
|
460
573
|
live = effects.reject { |fx| %i[recall store].include?(fx) }
|
|
461
574
|
return true if live.empty?
|
|
462
575
|
return true if need == :write && !write_verified?(effects: effects)
|
|
@@ -658,11 +771,19 @@ module PWN
|
|
|
658
771
|
end
|
|
659
772
|
m = nil
|
|
660
773
|
if !sem[:semantic_ok] && defined?(Mistakes) && sem[:shape].to_s != 'invalid_payload' && !raw.include?('extinguished_repeat')
|
|
661
|
-
# E1 — automatic blame attribution: if this tool just tripped a
|
|
662
|
-
# CUSUM changepoint AND extro drift is present, tag the mistake
|
|
663
|
-
# cause: :env_drift so it does NOT count toward [REPEATING].
|
|
664
774
|
cause = attribute_cause(name: name)
|
|
665
|
-
|
|
775
|
+
err = sem[:err] || raw[0, 300]
|
|
776
|
+
shape = sem[:shape]
|
|
777
|
+
if sem[:shape].to_s == 'timeout' && defined?(ToolGuard) && ToolGuard.respond_to?(:timeout_lesson)
|
|
778
|
+
lesson = ToolGuard.timeout_lesson(
|
|
779
|
+
tool: name,
|
|
780
|
+
payload: opts[:args].to_s,
|
|
781
|
+
timeout: err.to_s[/timeout after (\d+)/, 1].to_i
|
|
782
|
+
)
|
|
783
|
+
err = lesson[:error] if lesson[:error].to_s.strip.length.positive?
|
|
784
|
+
shape = :timeout
|
|
785
|
+
end
|
|
786
|
+
m = Mistakes.record(tool: name, error: err, args: opts[:args], session_id: opts[:session_id], source: :tool, cause: cause, shape: shape)
|
|
666
787
|
m = Mistakes.extinguish!(signature: m[:signature], args: opts[:args], shape: sem[:shape]) || m if m && defined?(Mistakes) && Mistakes.respond_to?(:extinguish!)
|
|
667
788
|
end
|
|
668
789
|
{ ok: sem[:semantic_ok], err: sem[:err], mistake: m, benign: sem[:benign] }
|
|
@@ -1140,7 +1261,7 @@ module PWN
|
|
|
1140
1261
|
messages = opts[:messages]
|
|
1141
1262
|
tools = opts[:tools]
|
|
1142
1263
|
debug_progress(
|
|
1143
|
-
msg: "call_engine
|
|
1264
|
+
msg: "call_engine #{debug_tools_line(tools: tools)} #{debug_msgs_line(messages: messages)}"
|
|
1144
1265
|
)
|
|
1145
1266
|
|
|
1146
1267
|
engine = active_engine
|
|
@@ -1162,7 +1283,7 @@ module PWN
|
|
|
1162
1283
|
cwt_opts = {
|
|
1163
1284
|
messages: wire_msgs,
|
|
1164
1285
|
tools: tools,
|
|
1165
|
-
spinner:
|
|
1286
|
+
spinner: true
|
|
1166
1287
|
}
|
|
1167
1288
|
# Ollama + abliterated / weak chat-templates often ignore tools: and
|
|
1168
1289
|
# answer in prose (or print shell(...) as text). Force native
|
|
@@ -1197,6 +1318,40 @@ module PWN
|
|
|
1197
1318
|
# Keep: system, original user, PLAN assistant (if any), last K tool
|
|
1198
1319
|
# pairs (assistant+tool), and the most recent assistant. Stale tool
|
|
1199
1320
|
# bodies are truncated to history_tool_max_chars.
|
|
1321
|
+
private_class_method def self.session_chat_history(opts = {})
|
|
1322
|
+
return [] unless defined?(PWN::Sessions) && PWN::Sessions.respond_to?(:to_llm_messages)
|
|
1323
|
+
|
|
1324
|
+
cap = opts[:max_chars].to_i
|
|
1325
|
+
if cap <= 0
|
|
1326
|
+
cap = local_engine? ? 8_000 : 48_000
|
|
1327
|
+
begin
|
|
1328
|
+
n = PWN::Env.dig(:ai, active_engine, :max_prompt_length).to_i
|
|
1329
|
+
cap = [cap, (n / 4)].min if n.positive?
|
|
1330
|
+
rescue StandardError
|
|
1331
|
+
nil
|
|
1332
|
+
end
|
|
1333
|
+
end
|
|
1334
|
+
PWN::Sessions.to_llm_messages(
|
|
1335
|
+
session_id: opts[:session_id],
|
|
1336
|
+
max_chars: cap,
|
|
1337
|
+
skip_request: opts[:skip_request]
|
|
1338
|
+
)
|
|
1339
|
+
rescue StandardError
|
|
1340
|
+
[]
|
|
1341
|
+
end
|
|
1342
|
+
|
|
1343
|
+
private_class_method def self.chat_response_history(opts = {})
|
|
1344
|
+
hist = session_chat_history(
|
|
1345
|
+
session_id: opts[:session_id],
|
|
1346
|
+
skip_request: opts[:skip_request] || opts[:request]
|
|
1347
|
+
)
|
|
1348
|
+
return if hist.empty?
|
|
1349
|
+
|
|
1350
|
+
{
|
|
1351
|
+
choices: [{ role: 'system', content: opts[:system_role_content].to_s }] + hist
|
|
1352
|
+
}
|
|
1353
|
+
end
|
|
1354
|
+
|
|
1200
1355
|
private_class_method def self.compact_history!(opts = {})
|
|
1201
1356
|
messages = opts[:messages]
|
|
1202
1357
|
return messages unless messages.is_a?(Array) && messages.length > 12
|
|
@@ -1206,17 +1361,21 @@ module PWN
|
|
|
1206
1361
|
|
|
1207
1362
|
head = []
|
|
1208
1363
|
rest = messages.dup
|
|
1209
|
-
#
|
|
1210
|
-
|
|
1364
|
+
# Keep system + this-session user/assistant conversation. Only
|
|
1365
|
+
# compact this-run tool dumps after that.
|
|
1366
|
+
head << rest.shift while rest.any? && rest.first[:role].to_s == 'system'
|
|
1367
|
+
while rest.any?
|
|
1368
|
+
msg = rest.first
|
|
1369
|
+
role = msg[:role].to_s
|
|
1370
|
+
break if role == 'tool'
|
|
1371
|
+
break if role == 'assistant' && Array(msg[:tool_calls]).any?
|
|
1372
|
+
|
|
1211
1373
|
head << rest.shift
|
|
1212
|
-
break if head.any? { |m| m[:role].to_s == 'user' }
|
|
1213
1374
|
end
|
|
1214
1375
|
head << rest.shift if rest.any? && rest.first[:role].to_s == 'assistant' && rest.first[:content].to_s.start_with?('PLAN:')
|
|
1215
1376
|
|
|
1216
|
-
# find indices of tool messages in rest; keep only last keep_pairs tool groups
|
|
1217
1377
|
tool_idxs = rest.each_index.select { |i| rest[i][:role].to_s == 'tool' }
|
|
1218
1378
|
drop_before = tool_idxs.length > keep_pairs ? tool_idxs[-keep_pairs] : 0
|
|
1219
|
-
# include the assistant tool_call message immediately before first kept tool
|
|
1220
1379
|
start = drop_before
|
|
1221
1380
|
start -= 1 if start.positive? && rest[start - 1] && rest[start - 1][:role].to_s == 'assistant'
|
|
1222
1381
|
kept = rest[start..] || []
|
|
@@ -1239,6 +1398,12 @@ module PWN
|
|
|
1239
1398
|
# Cheap answers already returned user-visible text.
|
|
1240
1399
|
return false if %i[greeting howto recall].include?(intent)
|
|
1241
1400
|
|
|
1401
|
+
# Advisor / nested Loop.run (red-team, critic) must not start
|
|
1402
|
+
# another tool-armed review of the same goal.
|
|
1403
|
+
return false if Thread.current[:pwn_loop_no_tools]
|
|
1404
|
+
return false if defined?(TurnFinalizer) && TurnFinalizer.user_path? &&
|
|
1405
|
+
Thread.current[:pwn_turn_finalizer_depth].to_i > 1
|
|
1406
|
+
|
|
1242
1407
|
return true unless opts[:local]
|
|
1243
1408
|
|
|
1244
1409
|
policy = agent_flag(key: :local_introspect, default: :failure_only).to_s.to_sym
|
|
@@ -1534,6 +1699,11 @@ module PWN
|
|
|
1534
1699
|
r = mod.chat(
|
|
1535
1700
|
request: request,
|
|
1536
1701
|
system_role_content: q_sys,
|
|
1702
|
+
response_history: chat_response_history(
|
|
1703
|
+
session_id: session_id,
|
|
1704
|
+
request: request,
|
|
1705
|
+
system_role_content: q_sys
|
|
1706
|
+
),
|
|
1537
1707
|
spinner: false
|
|
1538
1708
|
)
|
|
1539
1709
|
if r.is_a?(Hash)
|
|
@@ -1597,6 +1767,11 @@ module PWN
|
|
|
1597
1767
|
r = mod.chat(
|
|
1598
1768
|
request: request,
|
|
1599
1769
|
system_role_content: q_sys,
|
|
1770
|
+
response_history: chat_response_history(
|
|
1771
|
+
session_id: session_id,
|
|
1772
|
+
request: request,
|
|
1773
|
+
system_role_content: q_sys
|
|
1774
|
+
),
|
|
1600
1775
|
spinner: false
|
|
1601
1776
|
)
|
|
1602
1777
|
if r.is_a?(Hash)
|
|
@@ -1618,6 +1793,7 @@ module PWN
|
|
|
1618
1793
|
|
|
1619
1794
|
debug_progress(msg: "final accepted chars=#{txt.length}")
|
|
1620
1795
|
quiet_debug_tui!(reason: 'question')
|
|
1796
|
+
debug_final_text!(text: txt)
|
|
1621
1797
|
append_session(session_id: session_id, role: 'user', content: request)
|
|
1622
1798
|
append_session(session_id: session_id, role: 'assistant', content: txt)
|
|
1623
1799
|
if defined?(Learning) && should_auto_introspect?(local: local_engine?, turn_fails: {}, iter: 0)
|
|
@@ -1687,6 +1863,11 @@ module PWN
|
|
|
1687
1863
|
r = mod.chat(
|
|
1688
1864
|
request: request,
|
|
1689
1865
|
system_role_content: howto_sys,
|
|
1866
|
+
response_history: chat_response_history(
|
|
1867
|
+
session_id: session_id,
|
|
1868
|
+
request: request,
|
|
1869
|
+
system_role_content: howto_sys
|
|
1870
|
+
),
|
|
1690
1871
|
spinner: false
|
|
1691
1872
|
)
|
|
1692
1873
|
if r.is_a?(Hash)
|
|
@@ -1697,9 +1878,10 @@ module PWN
|
|
|
1697
1878
|
else
|
|
1698
1879
|
# chat_with_tools without tools
|
|
1699
1880
|
messages = [
|
|
1700
|
-
{ role: 'system', content: howto_sys }
|
|
1701
|
-
{ role: 'user', content: request }
|
|
1881
|
+
{ role: 'system', content: howto_sys }
|
|
1702
1882
|
]
|
|
1883
|
+
messages.concat(session_chat_history(session_id: session_id, skip_request: request))
|
|
1884
|
+
messages << { role: 'user', content: request }
|
|
1703
1885
|
msg = call_engine(messages: messages, tools: nil)
|
|
1704
1886
|
msg.is_a?(Hash) ? msg[:content].to_s : msg.to_s
|
|
1705
1887
|
end
|
|
@@ -1983,6 +2165,11 @@ module PWN
|
|
|
1983
2165
|
r = mod.chat(
|
|
1984
2166
|
request: request,
|
|
1985
2167
|
system_role_content: recall_sys,
|
|
2168
|
+
response_history: chat_response_history(
|
|
2169
|
+
session_id: session_id,
|
|
2170
|
+
request: request,
|
|
2171
|
+
system_role_content: recall_sys
|
|
2172
|
+
),
|
|
1986
2173
|
spinner: false
|
|
1987
2174
|
)
|
|
1988
2175
|
if r.is_a?(Hash)
|
|
@@ -2033,9 +2220,14 @@ module PWN
|
|
|
2033
2220
|
request = opts[:request].to_s
|
|
2034
2221
|
session_id = opts[:session_id]
|
|
2035
2222
|
on_tool = opts[:on_tool]
|
|
2223
|
+
i = 0
|
|
2224
|
+
tools_called = 0
|
|
2225
|
+
engine_s = 0.0
|
|
2226
|
+
final_chars = 0
|
|
2036
2227
|
start_debug_session(opts)
|
|
2037
2228
|
loud_debug_tui!(debug: opts[:debug])
|
|
2038
2229
|
debug_progress(msg: "Loop.run start request=#{request[0, 240]}", debug: opts[:debug])
|
|
2230
|
+
nested = defined?(TurnFinalizer) && TurnFinalizer.user_path?
|
|
2039
2231
|
TurnFinalizer.enter_user_path! if defined?(TurnFinalizer)
|
|
2040
2232
|
engine = active_engine
|
|
2041
2233
|
local = local_engine?(engine: engine)
|
|
@@ -2050,7 +2242,7 @@ module PWN
|
|
|
2050
2242
|
opts[:request] = request
|
|
2051
2243
|
intent = request_intent(request: request)
|
|
2052
2244
|
end
|
|
2053
|
-
elsif defined?(OpenGoal) && needs_host_work?(request: request) &&
|
|
2245
|
+
elsif !nested && defined?(OpenGoal) && needs_host_work?(request: request) &&
|
|
2054
2246
|
!%i[greeting howto recall].include?(intent)
|
|
2055
2247
|
OpenGoal.begin!(request: request, session_id: session_id)
|
|
2056
2248
|
end
|
|
@@ -2065,10 +2257,13 @@ module PWN
|
|
|
2065
2257
|
if intent == :greeting && opts[:force_tools] != true
|
|
2066
2258
|
debug_progress(msg: 'path=greeting', debug: opts[:debug])
|
|
2067
2259
|
quiet_debug_tui!(debug: opts[:debug], reason: 'greeting')
|
|
2068
|
-
|
|
2260
|
+
txt = answer_greeting(
|
|
2069
2261
|
request: request,
|
|
2070
2262
|
session_id: session_id
|
|
2071
2263
|
)
|
|
2264
|
+
final_chars = txt.to_s.length
|
|
2265
|
+
debug_final_text!(text: txt, debug: opts[:debug])
|
|
2266
|
+
return txt
|
|
2072
2267
|
end
|
|
2073
2268
|
|
|
2074
2269
|
# Thin system prompt only for remaining cheap paths (howto/recall).
|
|
@@ -2085,20 +2280,26 @@ module PWN
|
|
|
2085
2280
|
if intent == :howto
|
|
2086
2281
|
debug_progress(msg: 'path=howto', debug: opts[:debug])
|
|
2087
2282
|
quiet_debug_tui!(debug: opts[:debug], reason: 'howto')
|
|
2088
|
-
|
|
2283
|
+
txt = answer_howto(
|
|
2089
2284
|
request: request,
|
|
2090
2285
|
session_id: session_id,
|
|
2091
2286
|
system_role_content: system_role_content
|
|
2092
2287
|
)
|
|
2288
|
+
final_chars = txt.to_s.length
|
|
2289
|
+
debug_final_text!(text: txt, debug: opts[:debug])
|
|
2290
|
+
return txt
|
|
2093
2291
|
end
|
|
2094
2292
|
if intent == :recall
|
|
2095
2293
|
debug_progress(msg: 'path=recall', debug: opts[:debug])
|
|
2096
2294
|
quiet_debug_tui!(debug: opts[:debug], reason: 'recall')
|
|
2097
|
-
|
|
2295
|
+
txt = answer_recall(
|
|
2098
2296
|
request: request,
|
|
2099
2297
|
session_id: session_id,
|
|
2100
2298
|
system_role_content: system_role_content
|
|
2101
2299
|
)
|
|
2300
|
+
final_chars = txt.to_s.length
|
|
2301
|
+
debug_final_text!(text: txt, debug: opts[:debug])
|
|
2302
|
+
return txt
|
|
2102
2303
|
end
|
|
2103
2304
|
end
|
|
2104
2305
|
|
|
@@ -2139,15 +2340,24 @@ module PWN
|
|
|
2139
2340
|
core_only: core_only,
|
|
2140
2341
|
intent: intent
|
|
2141
2342
|
)
|
|
2343
|
+
no_tools = Array(tools).empty?
|
|
2344
|
+
Thread.current[:pwn_loop_no_tools] = no_tools
|
|
2142
2345
|
messages = [{ role: 'system', content: system_role_content }]
|
|
2143
2346
|
messages.concat(Learning.exemplars_for(request: request)) if local && defined?(Learning) && Learning.respond_to?(:exemplars_for)
|
|
2347
|
+
messages.concat(
|
|
2348
|
+
session_chat_history(session_id: session_id, skip_request: request)
|
|
2349
|
+
)
|
|
2144
2350
|
messages << { role: 'user', content: request }
|
|
2145
2351
|
append_session(session_id: session_id, role: 'user', content: request)
|
|
2146
2352
|
|
|
2147
2353
|
trivia = world_knowledge?(request: request)
|
|
2148
|
-
|
|
2149
|
-
|
|
2150
|
-
|
|
2354
|
+
catalog = catalog_lookup?(request: request)
|
|
2355
|
+
browse = request_need(request: request) == :browse
|
|
2356
|
+
skip_compass = trivia || catalog || no_tools || browse
|
|
2357
|
+
# Trivia / catalog / browse do not get an implement-shaped
|
|
2358
|
+
# English compass. Inventing "apply code/host changes" there
|
|
2359
|
+
# keeps the model on a Navigate task after the page already loaded.
|
|
2360
|
+
task_summary_plan!(state: ts_state, request: request, on_tool: on_tool) if defined?(TaskSummarizer) && !skip_compass
|
|
2151
2361
|
# Re-bind tools from English plan so task list is the sole driver of
|
|
2152
2362
|
# tool exposure/ranking (Registry keyword router + CORE).
|
|
2153
2363
|
if ts_state.is_a?(Hash) && defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:relevance_query)
|
|
@@ -2162,14 +2372,16 @@ module PWN
|
|
|
2162
2372
|
end
|
|
2163
2373
|
end
|
|
2164
2374
|
# English-task-as-primary: inject tangible tasks only for host work.
|
|
2165
|
-
inject_task_focus!(messages: messages, state: ts_state, force: true, request: request) unless
|
|
2375
|
+
inject_task_focus!(messages: messages, state: ts_state, force: true, request: request) unless skip_compass
|
|
2166
2376
|
predicted = nil
|
|
2167
2377
|
Thread.current[:pwn_plan_predicted] = nil
|
|
2168
2378
|
cal_state = calibration_state
|
|
2169
2379
|
force_plan = cal_state[:force_plan]
|
|
2170
|
-
skip_plan = %i[howto recall greeting].include?(intent) || trivia
|
|
2380
|
+
skip_plan = %i[howto recall greeting].include?(intent) || trivia || catalog || no_tools || browse
|
|
2381
|
+
did_plan = false
|
|
2171
2382
|
if !skip_plan && (force_plan || agent_flag(key: :plan_first, default: local) || budget_exhaustion_hot?) && !Array(tools).empty?
|
|
2172
2383
|
predicted = plan_first(messages: messages, request: request, ts_state: ts_state)
|
|
2384
|
+
did_plan = true
|
|
2173
2385
|
# P22 — prefer explicit return; fall back to thread stash
|
|
2174
2386
|
predicted = Thread.current[:pwn_plan_predicted] if predicted.nil?
|
|
2175
2387
|
# unify_plan! may have rewritten English tasks — force refresh focus.
|
|
@@ -2186,25 +2398,10 @@ module PWN
|
|
|
2186
2398
|
)
|
|
2187
2399
|
end
|
|
2188
2400
|
end
|
|
2189
|
-
inject_task_focus!(messages: messages, state: ts_state, force: true, request: request) unless
|
|
2190
|
-
end
|
|
2191
|
-
if budget_exhaustion_hot?
|
|
2192
|
-
english_open = defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:plan_open?) &&
|
|
2193
|
-
TaskSummarizer.plan_open?(state: ts_state, messages: messages)
|
|
2194
|
-
hot_hint = if local_engine? && !english_open
|
|
2195
|
-
'[pwn-ai/p17] Budget-exhaustion is the top open failure on this host. ' \
|
|
2196
|
-
'Prefer the SHORTEST plan that finishes the ask (≤3 tool calls). ' \
|
|
2197
|
-
'Emit a final answer as soon as you have evidence — do not explore.'
|
|
2198
|
-
else
|
|
2199
|
-
'[pwn-ai/p17] Budget-exhaustion is the top open failure on this host. ' \
|
|
2200
|
-
'Prefer the shortest plan that FULLY finishes the ask — no polite ' \
|
|
2201
|
-
'handoffs, no exploration side-quests. Emit a final answer as soon ' \
|
|
2202
|
-
'as you have evidence; keep going with tools until the goal is done ' \
|
|
2203
|
-
'or truly blocked.'
|
|
2204
|
-
end
|
|
2205
|
-
messages << { role: 'user', content: hot_hint }
|
|
2401
|
+
inject_task_focus!(messages: messages, state: ts_state, force: true, request: request) unless skip_compass
|
|
2206
2402
|
end
|
|
2207
|
-
|
|
2403
|
+
debug_progress(msg: "plan_first=#{did_plan} trivia=#{trivia} catalog=#{catalog} browse=#{browse}")
|
|
2404
|
+
if force_plan && cal_state[:cal] && !skip_plan
|
|
2208
2405
|
messages << {
|
|
2209
2406
|
role: 'user',
|
|
2210
2407
|
content: "[pwn-ai/w3] engine=#{active_engine} is overconfident " \
|
|
@@ -2218,71 +2415,27 @@ module PWN
|
|
|
2218
2415
|
maybe_park_budget_scars!
|
|
2219
2416
|
maybe_extinguish_parked!
|
|
2220
2417
|
|
|
2221
|
-
|
|
2418
|
+
i = 0
|
|
2419
|
+
loop do
|
|
2420
|
+
i += 1
|
|
2222
2421
|
# 3.1 — compact history on local so tool dumps don't fill num_ctx
|
|
2223
2422
|
compact_history!(messages: messages) if local
|
|
2224
2423
|
# English-task-as-primary: when plan_idx advanced, tell the model
|
|
2225
2424
|
# which plain-English task is active before the next tool batch.
|
|
2226
|
-
inject_task_focus!(messages: messages, state: ts_state, request: request) unless
|
|
2227
|
-
|
|
2228
|
-
# P17 — on the final iteration, strip tools and demand a plain-text
|
|
2229
|
-
# answer. Without this the model happily emits one more tool_calls
|
|
2230
|
-
# batch, burns the last slot, and lands on budget_exhausted with
|
|
2231
|
-
# nothing the user (or ORM) can use.
|
|
2232
|
-
# P17 deepen — when budget_hot, force text-only on the LAST TWO
|
|
2233
|
-
# iters so a final tool_calls batch cannot burn the terminal slot.
|
|
2234
|
-
# P17 deepen³ — under hot, force text-only on last THREE of the
|
|
2235
|
-
# 8-iter cap so a late tool binge cannot burn every salvage slot.
|
|
2236
|
-
# P17 structural: default hot text-only tail stays 3 (do NOT deepen to 4/6).
|
|
2237
|
-
# Plan-faithful headroom — short plan executing cleanly → delay strip to
|
|
2238
|
-
# last 1–2 so multi-step goals are not predestined to exhaust under cap 8.
|
|
2239
|
-
hot = budget_exhaustion_hot?
|
|
2240
|
-
plan_steps = begin
|
|
2241
|
-
predicted_plan = predicted || Thread.current[:pwn_plan_predicted]
|
|
2242
|
-
if predicted_plan.is_a?(Hash)
|
|
2243
|
-
Array(predicted_plan[:steps] || predicted_plan[:tools] || predicted_plan[:plan]).size
|
|
2244
|
-
elsif predicted_plan.is_a?(Array)
|
|
2245
|
-
predicted_plan.size
|
|
2246
|
-
else
|
|
2247
|
-
predicted_plan.to_s.scan(/\b(?:shell|pwn_eval|memory_|mistakes_|skill_|extro_|learning_|sessions_)\w*/).size
|
|
2248
|
-
end
|
|
2249
|
-
rescue StandardError
|
|
2250
|
-
0
|
|
2251
|
-
end
|
|
2252
|
-
# Plan-faithful: delay the text-only strip when a plan is executing
|
|
2253
|
-
# cleanly. Remote hot allows longer plans (runway 25); local hot
|
|
2254
|
-
# still favors short plans under the 8-iter cap.
|
|
2255
|
-
plan_step_limit = local_engine? ? 3 : 12
|
|
2256
|
-
plan_faithful = hot && plan_steps.positive? && plan_steps <= plan_step_limit &&
|
|
2257
|
-
turn_fails['empty_final'].to_i.zero? &&
|
|
2258
|
-
turn_fails.values.sum < 2
|
|
2259
|
-
# Last-iter strips tools only on the true last slot. English
|
|
2260
|
-
# leftovers and budget-hot must not steal runway from a live goal.
|
|
2261
|
-
text_only_iters = 1
|
|
2262
|
-
want_last = (i >= max_iters - text_only_iters)
|
|
2263
|
-
still_open = request_unsatisfied?(
|
|
2264
|
-
request: request,
|
|
2265
|
-
messages: messages,
|
|
2266
|
-
last_iter: false
|
|
2267
|
-
)
|
|
2268
|
-
last_iter = want_last && !still_open
|
|
2269
|
-
if last_iter
|
|
2270
|
-
tag = i >= max_iters - 1 ? 'FINAL ITERATION' : 'PENULTIMATE — wrap up'
|
|
2271
|
-
messages << {
|
|
2272
|
-
role: 'user',
|
|
2273
|
-
content: "[pwn-ai/p17] #{tag} — do NOT call any more tools. " \
|
|
2274
|
-
'Write the complete answer from evidence already in this ' \
|
|
2275
|
-
'transcript. If unfinished, say BLOCKED with evidence — not a ' \
|
|
2276
|
-
'markdown outline or "# Remaining block". Do NOT ask the user ' \
|
|
2277
|
-
'to confirm the next step.'
|
|
2278
|
-
}
|
|
2279
|
-
end
|
|
2425
|
+
inject_task_focus!(messages: messages, state: ts_state, request: request) unless skip_compass
|
|
2280
2426
|
|
|
2281
|
-
|
|
2427
|
+
t0 = Time.now
|
|
2428
|
+
msg = call_engine(messages: messages, tools: tools, ts_state: ts_state)
|
|
2429
|
+
engine_s += (Time.now - t0)
|
|
2430
|
+
PWN::Plugins::TTYSpinner.halt_all! if defined?(PWN::Plugins::TTYSpinner)
|
|
2282
2431
|
if msg.nil?
|
|
2283
2432
|
task_summary_flush!(state: ts_state, on_tool: on_tool)
|
|
2433
|
+
debug_progress(msg: 'engine returned no message')
|
|
2284
2434
|
quiet_debug_tui!(reason: 'engine_empty')
|
|
2285
|
-
|
|
2435
|
+
txt = '[pwn-ai] engine returned no message'
|
|
2436
|
+
debug_final_text!(text: txt)
|
|
2437
|
+
final_chars = txt.length
|
|
2438
|
+
return txt
|
|
2286
2439
|
end
|
|
2287
2440
|
|
|
2288
2441
|
calls = Array(msg[:tool_calls])
|
|
@@ -2290,7 +2443,7 @@ module PWN
|
|
|
2290
2443
|
|
|
2291
2444
|
# Belt-and-suspenders: plain-text shell(...) / tool forms from local
|
|
2292
2445
|
# models under weak TEMPLATE {{ .Prompt }} become real tool_calls.
|
|
2293
|
-
if calls.empty? && !text.strip.empty? &&
|
|
2446
|
+
if calls.empty? && !text.strip.empty? &&
|
|
2294
2447
|
defined?(Dispatch) && Dispatch.respond_to?(:tool_calls_from_text)
|
|
2295
2448
|
coerced = Dispatch.tool_calls_from_text(text: text)
|
|
2296
2449
|
if coerced.any?
|
|
@@ -2316,6 +2469,7 @@ module PWN
|
|
|
2316
2469
|
'Do not reply with an empty message.'
|
|
2317
2470
|
}
|
|
2318
2471
|
turn_fails['empty_final'] += 1
|
|
2472
|
+
debug_progress(msg: "bounce empty_final snippet=#{debug_snippet(text: text)}")
|
|
2319
2473
|
next
|
|
2320
2474
|
end
|
|
2321
2475
|
|
|
@@ -2326,6 +2480,7 @@ module PWN
|
|
|
2326
2480
|
if incomplete_final?(text: text, last_iter: false)
|
|
2327
2481
|
turn_fails['incomplete_final'] += 1
|
|
2328
2482
|
warn "[pwn-ai/loop] incomplete final on iter=#{i}; continuing autonomously"
|
|
2483
|
+
debug_progress(msg: "bounce incomplete_final snippet=#{debug_snippet(text: text)}")
|
|
2329
2484
|
messages << {
|
|
2330
2485
|
role: 'user',
|
|
2331
2486
|
content: '[pwn-ai/p28] That reply was incomplete (handoff or narrated next step). ' \
|
|
@@ -2343,6 +2498,7 @@ module PWN
|
|
|
2343
2498
|
)
|
|
2344
2499
|
turn_fails['unsatisfied'] += 1
|
|
2345
2500
|
warn "[pwn-ai/loop] original request not evidenced on iter=#{i}; continuing"
|
|
2501
|
+
debug_progress(msg: "bounce unsatisfied snippet=#{debug_snippet(text: text)}")
|
|
2346
2502
|
messages << {
|
|
2347
2503
|
role: 'user',
|
|
2348
2504
|
content: '[pwn-ai] The original request is not evidenced yet. ' \
|
|
@@ -2353,11 +2509,13 @@ module PWN
|
|
|
2353
2509
|
end
|
|
2354
2510
|
debug_progress(msg: "final accepted chars=#{text.to_s.length}")
|
|
2355
2511
|
quiet_debug_tui!(reason: 'final')
|
|
2512
|
+
debug_final_text!(text: text)
|
|
2513
|
+
final_chars = text.to_s.length
|
|
2356
2514
|
append_session(session_id: session_id, role: 'assistant', content: text)
|
|
2357
|
-
Learning.auto_introspect(session_id: session_id, request: request, final: text, predicted: predicted, plan: ts_state && ts_state[:plan], ts_state: ts_state) if defined?(Learning) && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: i)
|
|
2515
|
+
Learning.auto_introspect(session_id: session_id, request: request, final: text, predicted: predicted, plan: ts_state && ts_state[:plan], ts_state: ts_state) if defined?(Learning) && !nested && !no_tools && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: i)
|
|
2358
2516
|
maybe_finish_policy(session_id: session_id, proxy_ok: true, ts_state: ts_state)
|
|
2359
2517
|
task_summary_flush!(state: ts_state, on_tool: on_tool)
|
|
2360
|
-
OpenGoal.clear! if defined?(OpenGoal)
|
|
2518
|
+
OpenGoal.clear! if defined?(OpenGoal) && !nested
|
|
2361
2519
|
return text
|
|
2362
2520
|
end
|
|
2363
2521
|
|
|
@@ -2380,6 +2538,8 @@ module PWN
|
|
|
2380
2538
|
args = tc.dig(:function, :arguments)
|
|
2381
2539
|
entry = Registry.lookup(name: name)
|
|
2382
2540
|
started = Time.now
|
|
2541
|
+
argv_s = args.is_a?(String) ? args.to_s : args.inspect
|
|
2542
|
+
debug_progress(msg: "tool #{name} start:\n#{argv_s}", keep_newlines: true, cap: 0, tee: nil)
|
|
2383
2543
|
if Thread.current[:pwn_extinguished].is_a?(Hash) && Thread.current[:pwn_extinguished][name]
|
|
2384
2544
|
raw = JSON.generate(
|
|
2385
2545
|
success: false,
|
|
@@ -2389,7 +2549,7 @@ module PWN
|
|
|
2389
2549
|
else
|
|
2390
2550
|
raw = Dispatch.call(tool_call: tc)
|
|
2391
2551
|
end
|
|
2392
|
-
|
|
2552
|
+
tools_called += 1
|
|
2393
2553
|
tele = record_metrics(name: name, started: started, raw: raw, args: args, session_id: session_id, engine: engine, ts_state: ts_state)
|
|
2394
2554
|
result = Result.condition(content: raw, entry: entry)
|
|
2395
2555
|
|
|
@@ -2414,6 +2574,7 @@ module PWN
|
|
|
2414
2574
|
end
|
|
2415
2575
|
|
|
2416
2576
|
on_tool&.call(name, args, result)
|
|
2577
|
+
debug_tool_io!(name: name, args: args, result: result)
|
|
2417
2578
|
task_summary_record!(state: ts_state, name: name, args: args, result: result, on_tool: on_tool)
|
|
2418
2579
|
|
|
2419
2580
|
messages << {
|
|
@@ -2441,28 +2602,29 @@ module PWN
|
|
|
2441
2602
|
end
|
|
2442
2603
|
escalated = true
|
|
2443
2604
|
end
|
|
2444
|
-
|
|
2445
|
-
|
|
2446
|
-
|
|
2447
|
-
|
|
2448
|
-
|
|
2449
|
-
|
|
2450
|
-
|
|
2451
|
-
|
|
2452
|
-
|
|
2453
|
-
|
|
2454
|
-
|
|
2455
|
-
|
|
2456
|
-
|
|
2457
|
-
shape: :budget_exhausted
|
|
2458
|
-
)
|
|
2605
|
+
rescue Interrupt
|
|
2606
|
+
Thread.current[:pwn_log_progress] = false
|
|
2607
|
+
if defined?(PWN::Plugins::Log) && PWN::Plugins::Log.respond_to?(:note_interrupt!)
|
|
2608
|
+
PWN::Plugins::Log.note_interrupt!(where: 'CTRL+C', which_self: self)
|
|
2609
|
+
else
|
|
2610
|
+
debug_progress(msg: 'Interrupt CTRL+C')
|
|
2611
|
+
end
|
|
2612
|
+
raise
|
|
2613
|
+
rescue StandardError => e
|
|
2614
|
+
if defined?(PWN::Plugins::Log) && PWN::Plugins::Log.respond_to?(:note_exception!)
|
|
2615
|
+
PWN::Plugins::Log.note_exception!(error: e, where: 'Loop.run', which_self: self)
|
|
2616
|
+
else
|
|
2617
|
+
debug_progress(msg: "exception Loop.run #{e.class}: #{e.message}\n#{Array(e.backtrace).join("\n")}", keep_newlines: true, cap: 0)
|
|
2459
2618
|
end
|
|
2460
|
-
|
|
2461
|
-
Learning.auto_introspect(session_id: session_id, request: request, final: final_msg, predicted: predicted, plan: ts_state && ts_state[:plan], ts_state: ts_state) if defined?(Learning) && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: max_iters)
|
|
2462
|
-
maybe_finish_policy(session_id: session_id, proxy_ok: false, ts_state: ts_state)
|
|
2463
|
-
task_summary_flush!(state: ts_state, on_tool: on_tool)
|
|
2464
|
-
final_msg
|
|
2619
|
+
raise
|
|
2465
2620
|
ensure
|
|
2621
|
+
Thread.current[:pwn_loop_no_tools] = nil
|
|
2622
|
+
finish_debug_request!(
|
|
2623
|
+
iter: i,
|
|
2624
|
+
tools_called: tools_called,
|
|
2625
|
+
engine_s: engine_s,
|
|
2626
|
+
final_chars: final_chars
|
|
2627
|
+
)
|
|
2466
2628
|
TurnFinalizer.leave_user_path! if defined?(TurnFinalizer)
|
|
2467
2629
|
end
|
|
2468
2630
|
|
|
@@ -2496,7 +2658,7 @@ module PWN
|
|
|
2496
2658
|
# task_summary_verbose: false
|
|
2497
2659
|
|
|
2498
2660
|
Supported engines: #{ENGINE_MODS.keys.join(', ')}
|
|
2499
|
-
Set PWN::Env[:ai][:active] to choose
|
|
2661
|
+
Set PWN::Env[:ai][:active] to choose.
|
|
2500
2662
|
|
|
2501
2663
|
Intent routing (all engines; critical for ollama/openwebui):
|
|
2502
2664
|
how-to / usage questions → text-only explanation (no tools, no plan_first)
|
|
@@ -2516,8 +2678,8 @@ module PWN
|
|
|
2516
2678
|
:verify_as_reward - E3 ground every final via extro_verify (Boolean)
|
|
2517
2679
|
|
|
2518
2680
|
P28 autonomy: incomplete-final detector refuses mid-goal handoffs.
|
|
2519
|
-
|
|
2520
|
-
budget
|
|
2681
|
+
Loop.run keeps CORE_TOOLS until may_finalize? — there is no
|
|
2682
|
+
iteration-budget abort.
|
|
2521
2683
|
|
|
2522
2684
|
#{self}.authors
|
|
2523
2685
|
USAGE
|