pwn 0.5.680 → 0.5.683
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.gitignore +1 -0
- data/documentation/AI-Integration.md +1 -1
- data/documentation/Agent-Tool-Registry.md +11 -7
- data/documentation/Configuration.md +6 -7
- data/documentation/How-PWN-Works.md +2 -2
- data/documentation/Reinforcement-Learning.md +1 -1
- data/documentation/diagrams/agent-tool-registry.svg +1 -1
- data/documentation/diagrams/dot/agent-tool-registry.dot +1 -1
- data/documentation/diagrams/dot/task-summarizer.dot +3 -3
- data/documentation/pwn-ai-Agent.md +16 -35
- data/lib/pwn/ai/agent/loop.rb +265 -192
- data/lib/pwn/ai/agent/mistakes.rb +17 -0
- data/lib/pwn/ai/agent/policy.rb +53 -4
- data/lib/pwn/ai/agent/prompt_builder.rb +46 -19
- data/lib/pwn/ai/agent/registry.rb +18 -13
- data/lib/pwn/ai/agent/task_summarizer.rb +374 -503
- data/lib/pwn/ai/agent/tools/sessions.rb +32 -0
- data/lib/pwn/ai/agent/tools/skills.rb +56 -0
- data/lib/pwn/ai/anthropic.rb +0 -1
- data/lib/pwn/ai/gemini.rb +0 -1
- data/lib/pwn/ai/grok.rb +0 -1
- data/lib/pwn/ai/ollama.rb +0 -1
- data/lib/pwn/ai/open_ai.rb +0 -1
- data/lib/pwn/ai/open_web_ui.rb +0 -1
- data/lib/pwn/config.rb +1 -1
- data/lib/pwn/plugins/repl.rb +37 -0
- data/lib/pwn/plugins/tty_spinner.rb +55 -6
- data/lib/pwn/sessions.rb +82 -0
- data/lib/pwn/version.rb +1 -1
- data/spec/integration/prompt_builder_spec.rb +6 -4
- data/spec/lib/pwn/ai/agent/loop_spec.rb +314 -34
- data/spec/lib/pwn/ai/agent/mistakes_spec.rb +14 -0
- data/spec/lib/pwn/ai/agent/policy_spec.rb +52 -1
- data/spec/lib/pwn/ai/agent/prompt_builder_spec.rb +8 -9
- data/spec/lib/pwn/ai/agent/registry_spec.rb +30 -3
- data/spec/lib/pwn/ai/agent/signal_hygiene_spec.rb +4 -5
- data/spec/lib/pwn/ai/agent/task_summarizer_spec.rb +321 -90
- data/spec/lib/pwn/ai/agent/tools/sessions_spec.rb +5 -0
- data/spec/lib/pwn/ai/agent/tools/skills_spec.rb +16 -0
- data/spec/lib/pwn/ai/red_team/test_case_engine_spec.rb +20 -0
- data/spec/lib/pwn/plugins/repl_spec.rb +9 -0
- data/spec/lib/pwn/plugins/tty_spinner_spec.rb +30 -0
- data/spec/lib/pwn/sessions_spec.rb +29 -0
- data/spec/spec_helper.rb +6 -0
- data/third_party/pwn_rdoc.jsonl +30 -9
- metadata +1 -1
data/lib/pwn/ai/agent/loop.rb
CHANGED
|
@@ -34,6 +34,12 @@ module PWN
|
|
|
34
34
|
# PromptBuilder.mistakes_block re-injects the top open mistakes and
|
|
35
35
|
# top known fixes into the system prompt of every future turn.
|
|
36
36
|
#
|
|
37
|
+
# COMPLETION
|
|
38
|
+
# ----------
|
|
39
|
+
# The original request is the completion signal. TaskSummarizer and
|
|
40
|
+
# Policy are advisory (compass / rank). Loop keeps calling CORE_TOOLS
|
|
41
|
+
# until that request is done or truly blocked, then stops.
|
|
42
|
+
#
|
|
37
43
|
# LOCAL-MODEL SCAFFOLDING
|
|
38
44
|
# -----------------------
|
|
39
45
|
# When the active engine is :ollama (or the corresponding :agent flags
|
|
@@ -160,22 +166,31 @@ module PWN
|
|
|
160
166
|
|
|
161
167
|
private_class_method def self.maybe_park_budget_scars!
|
|
162
168
|
return unless defined?(Mistakes)
|
|
163
|
-
return if budget_exhaustion_hot?
|
|
164
169
|
return unless Mistakes.respond_to?(:park)
|
|
165
170
|
|
|
166
|
-
top = Mistakes.top(limit:
|
|
171
|
+
top = Mistakes.top(limit: 24, unresolved_only: true)
|
|
167
172
|
now = Time.now
|
|
168
|
-
top.
|
|
169
|
-
|
|
170
|
-
next if mistake[:parked]
|
|
171
|
-
|
|
173
|
+
budget = top.select { |mistake| budget_hit?(mistake: mistake) && !mistake[:parked] }
|
|
174
|
+
budget.each do |mistake|
|
|
172
175
|
stamp = mistake_ts(mistake: mistake)
|
|
173
|
-
# cool detector + scar older than PARK_COOL_SECS → park
|
|
174
176
|
next if stamp && (now - stamp) <= PARK_COOL_SECS
|
|
175
177
|
|
|
176
178
|
Mistakes.park(
|
|
177
179
|
signature: mistake[:signature].to_s,
|
|
178
|
-
reason: 'p17 rate-cool: outside PARK_COOL_SECS
|
|
180
|
+
reason: 'p17 rate-cool: outside PARK_COOL_SECS'
|
|
181
|
+
)
|
|
182
|
+
end
|
|
183
|
+
live = budget.reject do |mistake|
|
|
184
|
+
row = Mistakes.find(signature: mistake[:signature].to_s)
|
|
185
|
+
row.nil? || row[:parked]
|
|
186
|
+
end
|
|
187
|
+
return if live.length <= 1
|
|
188
|
+
|
|
189
|
+
# Never let 2+ budget scars latch hot forever. Keep only the newest.
|
|
190
|
+
live.sort_by { |mistake| mistake_ts(mistake: mistake) || Time.at(0) }[0...-1].each do |mistake|
|
|
191
|
+
Mistakes.park(
|
|
192
|
+
signature: mistake[:signature].to_s,
|
|
193
|
+
reason: 'p17 keep-newest budget scar; extras parked so tomorrow is not hot'
|
|
179
194
|
)
|
|
180
195
|
end
|
|
181
196
|
rescue StandardError
|
|
@@ -215,10 +230,9 @@ module PWN
|
|
|
215
230
|
end
|
|
216
231
|
|
|
217
232
|
# P17 — evidence-enough early final: latest tool rounds already answer
|
|
218
|
-
# the
|
|
219
|
-
#
|
|
220
|
-
#
|
|
221
|
-
# (mid-fix "write the complete final answer now" thrash).
|
|
233
|
+
# the original request → force synthesis. English tasks are an advisory
|
|
234
|
+
# compass only — an open verify tail must not block a finished ask.
|
|
235
|
+
# The original request is the completion signal.
|
|
222
236
|
private_class_method def self.evidence_enough_to_finalize?(opts = {})
|
|
223
237
|
messages = Array(opts[:messages])
|
|
224
238
|
turn_fails = opts[:turn_fails] || {}
|
|
@@ -226,30 +240,15 @@ module PWN
|
|
|
226
240
|
max_i = opts[:max_iters].to_i
|
|
227
241
|
request = opts[:request].to_s
|
|
228
242
|
return false if max_i <= 0 || iter < 2
|
|
229
|
-
# Need runway before the text-only strip, and no thrash.
|
|
230
243
|
return false if turn_fails['empty_final'].to_i.positive?
|
|
231
244
|
return false if turn_fails['incomplete_final'].to_i > 1
|
|
232
245
|
|
|
233
246
|
fail_n = turn_fails.values.sum
|
|
234
247
|
return false if fail_n >= 3
|
|
235
248
|
|
|
236
|
-
# English-task gate: multi-step plans only early-final on/after the
|
|
237
|
-
# last tangible task. plan_idx is 0-based; open work => not enough.
|
|
238
|
-
ts_state = opts[:ts_state]
|
|
239
|
-
if ts_state.is_a?(Hash)
|
|
240
|
-
plan = Array(ts_state[:plan])
|
|
241
|
-
if plan.length >= 2
|
|
242
|
-
idx = ts_state[:plan_idx].to_i
|
|
243
|
-
return false if idx < (plan.length - 1)
|
|
244
|
-
end
|
|
245
|
-
end
|
|
246
|
-
|
|
247
249
|
tools_ok = messages.select { |msg| msg[:role].to_s == 'tool' }
|
|
248
250
|
return false if tools_ok.size < 2
|
|
249
251
|
|
|
250
|
-
# Last two tool payloads should look like successful evidence, not errors.
|
|
251
|
-
# Agent tool wrappers always emit "success":true on ok — that alone is
|
|
252
|
-
# NOT proof the user goal is done (do not match bare success JSON).
|
|
253
252
|
last2 = tools_ok.last(2)
|
|
254
253
|
return false if last2.any? do |msg|
|
|
255
254
|
content = msg[:content].to_s
|
|
@@ -257,20 +256,19 @@ module PWN
|
|
|
257
256
|
!content.match?(/"success"\s*:\s*true/i)
|
|
258
257
|
end
|
|
259
258
|
|
|
260
|
-
# Prefer when plan was short / we already spent half the budget usefully.
|
|
261
259
|
plan_steps = opts[:plan_steps].to_i
|
|
262
260
|
short_plan = plan_steps.positive? && plan_steps <= 3
|
|
263
261
|
deep_enough = tools_ok.size >= 3 || (short_plan && tools_ok.size >= plan_steps)
|
|
264
262
|
return false unless deep_enough
|
|
265
263
|
|
|
266
264
|
recent_txt = last2.map { |msg| msg[:content].to_s[0, 500] }.join(' ')
|
|
267
|
-
# Goal-shaped completion only — write/patch/verify, not shell success wrappers.
|
|
268
265
|
mutation_done = recent_txt.match?(
|
|
269
|
-
/syntax ok|wrote |patched|
|
|
266
|
+
/syntax ok|wrote |patched|File\.write|ruby -c|0 offenses|examples?,\s*0 failures/i
|
|
270
267
|
)
|
|
271
268
|
return true if mutation_done
|
|
269
|
+
return true if request_path_evidenced?(request: request, blob: recent_txt)
|
|
272
270
|
return true if short_plan && tools_ok.size >= plan_steps && fail_n.zero? &&
|
|
273
|
-
request.match?(/\b(what|who|when|where|which|how many|status|list|show|print|uname|cwd|version)\b/i)
|
|
271
|
+
request.match?(/\b(what|who|when|where|which|how many|status|list|show|print|uname|cwd|version|hostname)\b/i)
|
|
274
272
|
|
|
275
273
|
false
|
|
276
274
|
rescue StandardError
|
|
@@ -312,6 +310,107 @@ module PWN
|
|
|
312
310
|
)
|
|
313
311
|
/ix
|
|
314
312
|
|
|
313
|
+
ACT_REQUEST_RX = /
|
|
314
|
+
\b(write|create|implement|fix|patch|replace|refactor|overwrite|
|
|
315
|
+
add (?:a |the )?|update|install|delete|remove|rename)\b
|
|
316
|
+
/ix
|
|
317
|
+
MUTATION_EVIDENCE_RX = /
|
|
318
|
+
printf\s|tee\s|sed\s+-i|ruby\s+-i|>\s|>>\s|file\.write|binwrite|
|
|
319
|
+
patched|wrote\s|syntax\sok|0\s+offenses|examples?,\s*0\s+failures
|
|
320
|
+
/ix
|
|
321
|
+
LOOKUP_REQUEST_RX = /
|
|
322
|
+
\b(what\s+is\s+my|hostname|uname|cwd|whoami|status|version|how\s+many)\b
|
|
323
|
+
/ix
|
|
324
|
+
# Real filesystem paths only — not https://host.tld (that was matching //host.tld).
|
|
325
|
+
HOST_PATH_RX = %r{(?:(?<![.:/])/(?!/)|\./)[\w./-]+\.\w+}
|
|
326
|
+
BROWSER_REQUEST_RX = /
|
|
327
|
+
TransparentBrowser|browser_obj|\bdevtools\b|
|
|
328
|
+
\b(navigate|dump_links|headless_?chrome|watir)\b
|
|
329
|
+
/ix
|
|
330
|
+
BROWSER_EVIDENCE_RX = %r{https?://|dump_links|TransparentBrowser|\.close\b|browser_obj}i
|
|
331
|
+
|
|
332
|
+
# True only when the ask needs a live host/file/browser effect. World-knowledge
|
|
333
|
+
# questions ("what color is a cherry") do not.
|
|
334
|
+
public_class_method def self.needs_host_work?(opts = {})
|
|
335
|
+
request = opts[:request].to_s
|
|
336
|
+
return false if request.strip.empty?
|
|
337
|
+
return true if request.match?(ACT_REQUEST_RX)
|
|
338
|
+
return true if request.match?(LOOKUP_REQUEST_RX)
|
|
339
|
+
return true if request.match?(HOST_PATH_RX)
|
|
340
|
+
return true if request.match?(BROWSER_REQUEST_RX)
|
|
341
|
+
|
|
342
|
+
false
|
|
343
|
+
rescue StandardError
|
|
344
|
+
false
|
|
345
|
+
end
|
|
346
|
+
|
|
347
|
+
# Short world-knowledge asks (no host/file work). Skip the planner LLM
|
|
348
|
+
# and do not bounce a text-only answer.
|
|
349
|
+
public_class_method def self.world_knowledge?(opts = {})
|
|
350
|
+
request = opts[:request].to_s.strip
|
|
351
|
+
return false if request.empty?
|
|
352
|
+
return false if needs_host_work?(request: request)
|
|
353
|
+
return false if request.length > 120
|
|
354
|
+
return false if request.match?(%r{\b(this\s+(?:host|machine|box|system|subnet|file|repo)|/opt/|implement|scan|hosts?)\b}i)
|
|
355
|
+
|
|
356
|
+
request.match?(/\A(?:what|why|who|when|where|which|how)\b/i)
|
|
357
|
+
rescue StandardError
|
|
358
|
+
false
|
|
359
|
+
end
|
|
360
|
+
|
|
361
|
+
# True when a text-only reply cannot yet be the original request.
|
|
362
|
+
# Distinct from English-task leftovers (advisory) and polite handoffs.
|
|
363
|
+
# Never bounce world-knowledge / no-host-work asks.
|
|
364
|
+
private_class_method def self.request_unsatisfied?(opts = {})
|
|
365
|
+
return false if opts[:last_iter]
|
|
366
|
+
return false if world_knowledge?(request: opts[:request])
|
|
367
|
+
return false unless needs_host_work?(request: opts[:request])
|
|
368
|
+
|
|
369
|
+
request = opts[:request].to_s
|
|
370
|
+
messages = Array(opts[:messages])
|
|
371
|
+
tools = messages.select { |msg| msg.is_a?(Hash) && msg[:role].to_s == 'tool' }
|
|
372
|
+
blob = +''
|
|
373
|
+
tools.each do |msg|
|
|
374
|
+
blob << msg[:name].to_s << ' ' << msg[:content].to_s << "\n"
|
|
375
|
+
end
|
|
376
|
+
messages.each do |msg|
|
|
377
|
+
next unless msg.is_a?(Hash) && msg[:role].to_s == 'assistant'
|
|
378
|
+
|
|
379
|
+
Array(msg[:tool_calls]).each do |tc|
|
|
380
|
+
blob << tc.dig(:function, :name).to_s << ' '
|
|
381
|
+
blob << tc.dig(:function, :arguments).to_s << "\n"
|
|
382
|
+
end
|
|
383
|
+
end
|
|
384
|
+
|
|
385
|
+
return false if request.match?(LOOKUP_REQUEST_RX) && tools.any? && blob.length >= 20
|
|
386
|
+
return true if tools.empty?
|
|
387
|
+
|
|
388
|
+
if request.match?(ACT_REQUEST_RX) || request.match?(HOST_PATH_RX)
|
|
389
|
+
return false if blob.match?(MUTATION_EVIDENCE_RX)
|
|
390
|
+
return false if request_path_evidenced?(request: request, blob: blob)
|
|
391
|
+
|
|
392
|
+
return true
|
|
393
|
+
end
|
|
394
|
+
return false if request.match?(BROWSER_REQUEST_RX) && blob.match?(BROWSER_EVIDENCE_RX)
|
|
395
|
+
|
|
396
|
+
blob.length < 20
|
|
397
|
+
rescue StandardError
|
|
398
|
+
false
|
|
399
|
+
end
|
|
400
|
+
|
|
401
|
+
private_class_method def self.request_path_evidenced?(opts = {})
|
|
402
|
+
request = opts[:request].to_s
|
|
403
|
+
blob = opts[:blob].to_s
|
|
404
|
+
paths = request.scan(%r{(?:/|\./)[\w./-]+\.\w+})
|
|
405
|
+
return false if paths.empty?
|
|
406
|
+
|
|
407
|
+
paths.any? do |path|
|
|
408
|
+
blob.include?(path) && blob.match?(
|
|
409
|
+
/open\(|File\.(?:write|open|binwrite)|write\(|puts\s|print\s|>\s|>>\s|tee\s|sed\s+-i|ruby\s+-i|patched|wrote/i
|
|
410
|
+
)
|
|
411
|
+
end
|
|
412
|
+
end
|
|
413
|
+
|
|
315
414
|
private_class_method def self.incomplete_final?(opts = {})
|
|
316
415
|
text = opts[:text].to_s
|
|
317
416
|
return false if text.strip.empty?
|
|
@@ -999,18 +1098,11 @@ module PWN
|
|
|
999
1098
|
if env_tc && !env_tc.to_s.empty?
|
|
1000
1099
|
cwt_opts[:tool_choice] = env_tc
|
|
1001
1100
|
else
|
|
1002
|
-
#
|
|
1003
|
-
#
|
|
1004
|
-
# monologue as content. Stay on required until a tool result
|
|
1005
|
-
# exists AND the last assistant turn already looks like a
|
|
1006
|
-
# genuine final (no monologue / handoff markers). That keeps
|
|
1007
|
-
# pressure on native tool_calls through the mid-loop thrash
|
|
1008
|
-
# that previously returned "Wait, let's try hping3…" as FINAL.
|
|
1101
|
+
# After the first tool result, auto so the model can emit a
|
|
1102
|
+
# real final. Leftover English tasks do not keep required.
|
|
1009
1103
|
has_tool_result = Array(messages).any? { |m| m[:role].to_s == 'tool' }
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
still_acting = last_txt.strip.empty? || incomplete_final?(text: last_txt, last_iter: false)
|
|
1013
|
-
cwt_opts[:tool_choice] = has_tool_result && !still_acting ? 'auto' : 'required'
|
|
1104
|
+
need_tools = needs_host_work?(request: Array(messages).find { |m| m[:role].to_s == 'user' }&.[](:content))
|
|
1105
|
+
cwt_opts[:tool_choice] = has_tool_result || !need_tools ? 'auto' : 'required'
|
|
1014
1106
|
end
|
|
1015
1107
|
end
|
|
1016
1108
|
response = mod.chat_with_tools(cwt_opts)
|
|
@@ -1063,15 +1155,9 @@ module PWN
|
|
|
1063
1155
|
# 3.2 — local models cannot afford auto_introspect (judge+prm+critic+
|
|
1064
1156
|
# sentinel+extro) on every success. Default :failure_only when local.
|
|
1065
1157
|
private_class_method def self.should_auto_introspect?(opts = {})
|
|
1066
|
-
kind = (opts[:kind] || Thread.current[:pwn_request_kind]).to_s.to_sym
|
|
1067
1158
|
intent = (opts[:intent] || Thread.current[:pwn_request_intent]).to_s.to_sym
|
|
1068
|
-
|
|
1069
|
-
# Cheap answers already returned user-visible text. The post-answer
|
|
1070
|
-
# critic + 12s ORM printed ERROR: Timed out reading data from server
|
|
1071
|
-
# after greetings / takes / questions.
|
|
1159
|
+
# Cheap answers already returned user-visible text.
|
|
1072
1160
|
return false if %i[greeting howto recall].include?(intent)
|
|
1073
|
-
return false if kind == :statement
|
|
1074
|
-
return false if kind == :question && fails.zero?
|
|
1075
1161
|
|
|
1076
1162
|
return true unless opts[:local]
|
|
1077
1163
|
|
|
@@ -1116,9 +1202,13 @@ module PWN
|
|
|
1116
1202
|
messages = opts[:messages]
|
|
1117
1203
|
return nil unless state.is_a?(Hash) && messages.is_a?(Array)
|
|
1118
1204
|
return nil unless defined?(TaskSummarizer) && TaskSummarizer.enabled?
|
|
1205
|
+
return nil if respond_to?(:needs_host_work?) && !needs_host_work?(request: opts[:request] || state[:original_request] || state[:request])
|
|
1206
|
+
return nil unless TaskSummarizer.plan_open?(state: state, messages: messages)
|
|
1119
1207
|
|
|
1208
|
+
req = opts[:request]
|
|
1209
|
+
req = state[:original_request] || state[:request] if req.to_s.strip.empty? && state.is_a?(Hash)
|
|
1120
1210
|
text =
|
|
1121
|
-
(TaskSummarizer.active_task_prompt(state: state, force: opts[:force]) if TaskSummarizer.respond_to?(:active_task_prompt))
|
|
1211
|
+
(TaskSummarizer.active_task_prompt(state: state, force: opts[:force], request: req) if TaskSummarizer.respond_to?(:active_task_prompt))
|
|
1122
1212
|
return nil if text.to_s.strip.empty?
|
|
1123
1213
|
|
|
1124
1214
|
messages << { role: 'user', content: text }
|
|
@@ -1249,6 +1339,7 @@ module PWN
|
|
|
1249
1339
|
last\s+thing\s+(?:i|you)\s+said
|
|
1250
1340
|
)\b
|
|
1251
1341
|
/ix
|
|
1342
|
+
LAST_SESSION_RX = /\b(?:in|from|of)\s+(?:the\s+)?(?:last|previous|prior)\s+session\b|\blast\s+session\b/i
|
|
1252
1343
|
|
|
1253
1344
|
# Pure greeting / light smalltalk — never full :act tool loop.
|
|
1254
1345
|
# Anchored short forms only so "hi, please scan X" stays :act/:recon_act.
|
|
@@ -1316,6 +1407,14 @@ module PWN
|
|
|
1316
1407
|
return :recall unless doing
|
|
1317
1408
|
end
|
|
1318
1409
|
|
|
1410
|
+
if req.match?(LAST_SESSION_RX) && !req.match?(HOWTO_RX) && !req.match?(LIVE_RECON_RX)
|
|
1411
|
+
doing = req.match?(
|
|
1412
|
+
/\b(implement|fix|patch|refactor|run|execute|scan|write|edit|
|
|
1413
|
+
change|deploy|install|build|compile|commit|push)\b/ix
|
|
1414
|
+
)
|
|
1415
|
+
return :recall unless doing
|
|
1416
|
+
end
|
|
1417
|
+
|
|
1319
1418
|
# Live-action recon takes precedence over bare "how to" when both appear
|
|
1320
1419
|
# only if the user clearly asks the agent to do the sweep here.
|
|
1321
1420
|
live = req.match?(LIVE_RECON_RX) && req.match?(
|
|
@@ -1335,55 +1434,6 @@ module PWN
|
|
|
1335
1434
|
:act
|
|
1336
1435
|
end
|
|
1337
1436
|
|
|
1338
|
-
# Top-level request kind for task planning (statement | question | autonomous_goal).
|
|
1339
|
-
# Single source of truth: TaskSummarizer.request_kind (LLM + heuristics).
|
|
1340
|
-
# Mirrors intent/heuristics only when TaskSummarizer is unavailable.
|
|
1341
|
-
#
|
|
1342
|
-
# Supported Method Parameters::
|
|
1343
|
-
# kind = PWN::AI::Agent::Loop.request_kind(
|
|
1344
|
-
# request: 'required - user text',
|
|
1345
|
-
# kind: 'optional - precomputed',
|
|
1346
|
-
# llm_kind: 'optional - injected LLM label',
|
|
1347
|
-
# heuristic_only: 'optional - skip LLM'
|
|
1348
|
-
# )
|
|
1349
|
-
public_class_method def self.request_kind(opts = {})
|
|
1350
|
-
req = opts[:request].to_s
|
|
1351
|
-
if defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:request_kind)
|
|
1352
|
-
return TaskSummarizer.request_kind(
|
|
1353
|
-
request: req,
|
|
1354
|
-
kind: opts[:kind],
|
|
1355
|
-
llm_kind: opts[:llm_kind],
|
|
1356
|
-
heuristic_only: opts[:heuristic_only]
|
|
1357
|
-
)
|
|
1358
|
-
end
|
|
1359
|
-
|
|
1360
|
-
case request_intent(request: req)
|
|
1361
|
-
when :greeting, :empty
|
|
1362
|
-
:statement
|
|
1363
|
-
when :howto, :recall
|
|
1364
|
-
:question
|
|
1365
|
-
when :recon_act
|
|
1366
|
-
:autonomous_goal
|
|
1367
|
-
else
|
|
1368
|
-
# :act — distinguish bare questions from work the agent must do.
|
|
1369
|
-
# Host-local facts need tools → autonomous_goal.
|
|
1370
|
-
if defined?(TaskSummarizer) && TaskSummarizer.const_defined?(:NEEDS_LOCAL_EVIDENCE_RX)
|
|
1371
|
-
return :autonomous_goal if req.match?(TaskSummarizer::NEEDS_LOCAL_EVIDENCE_RX)
|
|
1372
|
-
elsif req.match?(/\b(?:hostname|whoami|\bcwd\b|\bpwd\b|my\s+ip)\b/i)
|
|
1373
|
-
return :autonomous_goal
|
|
1374
|
-
end
|
|
1375
|
-
return :question if req.match?(/\?\s*\z/) && !req.match?(
|
|
1376
|
-
/\b(please|implement|fix|patch|refactor|run|scan|find|write|change)\b/i
|
|
1377
|
-
)
|
|
1378
|
-
return :question if req.match?(/\A\s*(?:what|why|when|where|who|which|how)\b/i) &&
|
|
1379
|
-
!req.match?(/\b(please|implement|fix|patch|run|scan)\b/i)
|
|
1380
|
-
|
|
1381
|
-
:autonomous_goal
|
|
1382
|
-
end
|
|
1383
|
-
rescue StandardError
|
|
1384
|
-
:autonomous_goal
|
|
1385
|
-
end
|
|
1386
|
-
|
|
1387
1437
|
public_class_method def self.recon_authorized?(opts = {})
|
|
1388
1438
|
req = opts[:request].to_s
|
|
1389
1439
|
return true if req.match?(AUTH_SCOPE_RX)
|
|
@@ -1687,6 +1737,18 @@ module PWN
|
|
|
1687
1737
|
session_id = opts[:session_id]
|
|
1688
1738
|
system_role_content = opts[:system_role_content].to_s
|
|
1689
1739
|
target = recall_target(request: request)
|
|
1740
|
+
last_session = request.match?(LAST_SESSION_RX)
|
|
1741
|
+
if last_session && defined?(PWN::Sessions) && PWN::Sessions.respond_to?(:previous_id)
|
|
1742
|
+
prev = PWN::Sessions.previous_id(exclude_session_id: session_id)
|
|
1743
|
+
if prev.to_s.empty?
|
|
1744
|
+
txt = 'I do not have a previous session transcript yet.'
|
|
1745
|
+
append_session(session_id: opts[:session_id], role: 'user', content: request)
|
|
1746
|
+
append_session(session_id: opts[:session_id], role: 'assistant', content: txt)
|
|
1747
|
+
return txt
|
|
1748
|
+
end
|
|
1749
|
+
|
|
1750
|
+
session_id = prev
|
|
1751
|
+
end
|
|
1690
1752
|
|
|
1691
1753
|
prior_user = nil
|
|
1692
1754
|
prior_asst = nil
|
|
@@ -1729,7 +1791,7 @@ module PWN
|
|
|
1729
1791
|
# User-target / fallback: skip meta intermediate recall asks so
|
|
1730
1792
|
# "what did I just say?" after a nested chain still surfaces the
|
|
1731
1793
|
# original utterance when appropriate; default stays newest.
|
|
1732
|
-
skip_meta = target == :assistant
|
|
1794
|
+
skip_meta = target == :assistant || last_session
|
|
1733
1795
|
prior_user = PWN::Memory.prior_user_message(
|
|
1734
1796
|
session_id: session_id,
|
|
1735
1797
|
max_chars: 4_000,
|
|
@@ -1913,73 +1975,76 @@ module PWN
|
|
|
1913
1975
|
session_id = opts[:session_id]
|
|
1914
1976
|
on_tool = opts[:on_tool]
|
|
1915
1977
|
TurnFinalizer.enter_user_path! if defined?(TurnFinalizer)
|
|
1916
|
-
# Live coalesced "what am I doing" lines for the TUI (not a model tool).
|
|
1917
|
-
ts_state = (TaskSummarizer.fresh(request: request) if defined?(TaskSummarizer) && TaskSummarizer.enabled? && Thread.current[:pwn_reflect_depth].to_i.zero?)
|
|
1918
1978
|
engine = active_engine
|
|
1919
1979
|
local = local_engine?(engine: engine)
|
|
1920
|
-
system_role_content = opts[:system_role_content] ||= PWN::AI::Agent::PromptBuilder.build(session_id: session_id, request: request)
|
|
1921
|
-
|
|
1922
|
-
Registry.discover
|
|
1923
|
-
maybe_refresh_extro_snapshot!
|
|
1924
|
-
opts[:enabled_toolsets] = default_interactive_toolsets(request: request) unless opts.key?(:enabled_toolsets)
|
|
1925
|
-
expose_current_session(session_id: session_id)
|
|
1926
|
-
Mistakes.check_user_correction(request: request, session_id: session_id) if defined?(Mistakes)
|
|
1927
1980
|
|
|
1981
|
+
# Cheap intent/kind FIRST - before PromptBuilder / Registry / TaskSummarizer
|
|
1982
|
+
# so greetings, FYIs, how-tos, recall, and simple Qs never pay the fat path.
|
|
1928
1983
|
intent = request_intent(request: request)
|
|
1929
|
-
kind = request_kind(request: request)
|
|
1930
1984
|
Thread.current[:pwn_request_intent] = intent
|
|
1931
|
-
Thread.current[:pwn_request_kind] = kind
|
|
1932
1985
|
Thread.current[:pwn_recon_authorized] = recon_authorized?(request: request)
|
|
1933
1986
|
Thread.current[:pwn_extinguished] = {}
|
|
1934
|
-
|
|
1987
|
+
expose_current_session(session_id: session_id)
|
|
1988
|
+
Mistakes.check_user_correction(request: request, session_id: session_id) if defined?(Mistakes)
|
|
1989
|
+
|
|
1990
|
+
cheap = opts[:force_tools] != true && %i[greeting howto recall].include?(intent)
|
|
1991
|
+
|
|
1992
|
+
# Greeting / light smalltalk: deterministic ack - no weather echo, no tools,
|
|
1993
|
+
# no PromptBuilder, no Registry.
|
|
1935
1994
|
if intent == :greeting && opts[:force_tools] != true
|
|
1936
1995
|
return answer_greeting(
|
|
1937
1996
|
request: request,
|
|
1938
1997
|
session_id: session_id
|
|
1939
1998
|
)
|
|
1940
1999
|
end
|
|
1941
|
-
|
|
1942
|
-
|
|
1943
|
-
|
|
1944
|
-
|
|
1945
|
-
|
|
1946
|
-
system_role_content
|
|
1947
|
-
|
|
1948
|
-
|
|
1949
|
-
|
|
1950
|
-
|
|
1951
|
-
|
|
1952
|
-
|
|
1953
|
-
|
|
1954
|
-
|
|
1955
|
-
|
|
1956
|
-
|
|
1957
|
-
|
|
1958
|
-
|
|
1959
|
-
|
|
1960
|
-
|
|
1961
|
-
|
|
1962
|
-
|
|
1963
|
-
|
|
1964
|
-
|
|
1965
|
-
|
|
1966
|
-
|
|
1967
|
-
|
|
1968
|
-
|
|
1969
|
-
return answer_question(
|
|
1970
|
-
request: request,
|
|
1971
|
-
session_id: session_id,
|
|
1972
|
-
system_role_content: system_role_content
|
|
1973
|
-
)
|
|
2000
|
+
|
|
2001
|
+
# Thin system prompt only for remaining cheap paths (howto/recall/statement/question).
|
|
2002
|
+
if cheap
|
|
2003
|
+
system_role_content = opts[:system_role_content]
|
|
2004
|
+
if system_role_content.nil? || system_role_content.to_s.empty?
|
|
2005
|
+
system_role_content = PWN::AI::Agent::PromptBuilder.build(
|
|
2006
|
+
session_id: session_id,
|
|
2007
|
+
request: request,
|
|
2008
|
+
thin: true
|
|
2009
|
+
)
|
|
2010
|
+
opts[:system_role_content] = system_role_content
|
|
2011
|
+
end
|
|
2012
|
+
# How-to: never enter plan_first / task recon / tool thrash.
|
|
2013
|
+
if intent == :howto
|
|
2014
|
+
return answer_howto(
|
|
2015
|
+
request: request,
|
|
2016
|
+
session_id: session_id,
|
|
2017
|
+
system_role_content: system_role_content
|
|
2018
|
+
)
|
|
2019
|
+
end
|
|
2020
|
+
# Pure prior-turn / vague memory recall: one cheap path, no plan_first.
|
|
2021
|
+
if intent == :recall
|
|
2022
|
+
return answer_recall(
|
|
2023
|
+
request: request,
|
|
2024
|
+
session_id: session_id,
|
|
2025
|
+
system_role_content: system_role_content
|
|
2026
|
+
)
|
|
2027
|
+
end
|
|
1974
2028
|
end
|
|
1975
2029
|
|
|
2030
|
+
# --- act / recon / autonomous_goal: full context + tools ---
|
|
2031
|
+
# Reuse precomputed kind so TaskSummarizer.fresh does not classify twice.
|
|
2032
|
+
ts_state = (TaskSummarizer.fresh(request: request) if defined?(TaskSummarizer) && TaskSummarizer.enabled? && Thread.current[:pwn_reflect_depth].to_i.zero?)
|
|
2033
|
+
system_role_content = opts[:system_role_content] ||= PWN::AI::Agent::PromptBuilder.build(
|
|
2034
|
+
session_id: session_id,
|
|
2035
|
+
request: request
|
|
2036
|
+
)
|
|
2037
|
+
|
|
2038
|
+
Registry.discover
|
|
2039
|
+
maybe_refresh_extro_snapshot!
|
|
2040
|
+
opts[:enabled_toolsets] = default_interactive_toolsets(request: request) unless opts.key?(:enabled_toolsets)
|
|
2041
|
+
|
|
1976
2042
|
# R5 — open the live MDP episode BEFORE the first Registry.rank so
|
|
1977
2043
|
# Q(s,a) can advise this turn. Planning still owns the task list.
|
|
1978
2044
|
if defined?(PWN::AI::Agent::Policy) && Policy.respond_to?(:begin_episode)
|
|
1979
2045
|
Policy.begin_episode(
|
|
1980
2046
|
session_id: session_id,
|
|
1981
2047
|
request: request,
|
|
1982
|
-
kind: kind,
|
|
1983
2048
|
intent: intent,
|
|
1984
2049
|
engine: engine,
|
|
1985
2050
|
ts_state: ts_state
|
|
@@ -1990,53 +2055,43 @@ module PWN
|
|
|
1990
2055
|
# TaskSummarizer.emit_plan! we re-rank using English tangible tasks
|
|
1991
2056
|
# so generated tasks — not the bare request — drive which tools
|
|
1992
2057
|
# the model may call.
|
|
1993
|
-
|
|
2058
|
+
# CORE_TOOLS is the default action space. Extra schemas are
|
|
2059
|
+
# opt-in via enabled_toolsets + core_only: false.
|
|
2060
|
+
core_only = opts.fetch(:core_only, true)
|
|
2061
|
+
tools = Registry.definitions(
|
|
2062
|
+
enabled: opts[:enabled_toolsets],
|
|
2063
|
+
relevance: request,
|
|
2064
|
+
core_only: core_only,
|
|
2065
|
+
intent: intent
|
|
2066
|
+
)
|
|
1994
2067
|
messages = [{ role: 'system', content: system_role_content }]
|
|
1995
2068
|
messages.concat(Learning.exemplars_for(request: request)) if local && defined?(Learning) && Learning.respond_to?(:exemplars_for)
|
|
1996
2069
|
messages << { role: 'user', content: request }
|
|
1997
2070
|
append_session(session_id: session_id, role: 'user', content: request)
|
|
1998
2071
|
|
|
1999
|
-
#
|
|
2000
|
-
|
|
2001
|
-
needs_breakdown =
|
|
2002
|
-
if defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:needs_task_breakdown?)
|
|
2003
|
-
TaskSummarizer.needs_task_breakdown?(kind: kind, request: request)
|
|
2004
|
-
else
|
|
2005
|
-
kind.to_sym == :autonomous_goal
|
|
2006
|
-
end
|
|
2007
|
-
ts_state[:request_kind] = kind if ts_state.is_a?(Hash)
|
|
2008
|
-
if needs_breakdown
|
|
2009
|
-
task_summary_plan!(state: ts_state, request: request, on_tool: on_tool)
|
|
2010
|
-
elsif ts_state.is_a?(Hash) && defined?(TaskSummarizer)
|
|
2011
|
-
# Record kind on state; optional one-line kind banner (no task list).
|
|
2012
|
-
ts_state[:plan] = []
|
|
2013
|
-
ts_state[:request_kind] = kind
|
|
2014
|
-
if TaskSummarizer.respond_to?(:format_plan)
|
|
2015
|
-
banner = TaskSummarizer.format_plan(tasks: [], request: request, request_kind: kind)
|
|
2016
|
-
if banner && !banner.to_s.empty?
|
|
2017
|
-
ts_state[:plan_text] = banner
|
|
2018
|
-
ts_state[:plan_emitted] = true
|
|
2019
|
-
emit_task_summary(line: banner, on_tool: on_tool)
|
|
2020
|
-
end
|
|
2021
|
-
end
|
|
2022
|
-
end
|
|
2072
|
+
# Every request gets a task compass.
|
|
2073
|
+
task_summary_plan!(state: ts_state, request: request, on_tool: on_tool) if defined?(TaskSummarizer)
|
|
2023
2074
|
# Re-bind tools from English plan so task list is the sole driver of
|
|
2024
2075
|
# tool exposure/ranking (Registry keyword router + CORE).
|
|
2025
2076
|
if ts_state.is_a?(Hash) && defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:relevance_query)
|
|
2026
2077
|
rq = TaskSummarizer.relevance_query(state: ts_state, request: request)
|
|
2027
|
-
|
|
2078
|
+
unless rq.to_s.strip.empty?
|
|
2079
|
+
tools = Registry.definitions(
|
|
2080
|
+
enabled: opts[:enabled_toolsets],
|
|
2081
|
+
relevance: rq,
|
|
2082
|
+
core_only: core_only,
|
|
2083
|
+
intent: intent
|
|
2084
|
+
)
|
|
2085
|
+
end
|
|
2028
2086
|
end
|
|
2029
2087
|
# English-task-as-primary: inject tangible tasks only for autonomous goals.
|
|
2030
|
-
inject_task_focus!(messages: messages, state: ts_state, force: true
|
|
2031
|
-
|
|
2088
|
+
inject_task_focus!(messages: messages, state: ts_state, force: true, request: request)
|
|
2032
2089
|
predicted = nil
|
|
2033
2090
|
Thread.current[:pwn_plan_predicted] = nil
|
|
2034
2091
|
cal_state = calibration_state
|
|
2035
2092
|
force_plan = cal_state[:force_plan]
|
|
2036
|
-
# Skip plan_first for
|
|
2037
|
-
skip_plan = %i[howto recall greeting].include?(intent)
|
|
2038
|
-
%i[statement question].include?(kind.to_sym) ||
|
|
2039
|
-
!needs_breakdown
|
|
2093
|
+
# Skip plan_first only for remaining cheap intents.
|
|
2094
|
+
skip_plan = %i[howto recall greeting].include?(intent)
|
|
2040
2095
|
if !skip_plan && (force_plan || agent_flag(key: :plan_first, default: local) || budget_exhaustion_hot?) && !Array(tools).empty?
|
|
2041
2096
|
predicted = plan_first(messages: messages, request: request, ts_state: ts_state)
|
|
2042
2097
|
# P22 — prefer explicit return; fall back to thread stash
|
|
@@ -2046,12 +2101,21 @@ module PWN
|
|
|
2046
2101
|
# PLAN: tool-call scaffold jargon (unify_plan! refuses that).
|
|
2047
2102
|
if ts_state.is_a?(Hash) && defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:relevance_query)
|
|
2048
2103
|
rq = TaskSummarizer.relevance_query(state: ts_state, request: request)
|
|
2049
|
-
|
|
2104
|
+
unless rq.to_s.strip.empty?
|
|
2105
|
+
tools = Registry.definitions(
|
|
2106
|
+
enabled: opts[:enabled_toolsets],
|
|
2107
|
+
relevance: rq,
|
|
2108
|
+
core_only: core_only,
|
|
2109
|
+
intent: intent
|
|
2110
|
+
)
|
|
2111
|
+
end
|
|
2050
2112
|
end
|
|
2051
|
-
inject_task_focus!(messages: messages, state: ts_state, force: true)
|
|
2113
|
+
inject_task_focus!(messages: messages, state: ts_state, force: true, request: request)
|
|
2052
2114
|
end
|
|
2053
2115
|
if budget_exhaustion_hot?
|
|
2054
|
-
|
|
2116
|
+
english_open = defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:plan_open?) &&
|
|
2117
|
+
TaskSummarizer.plan_open?(state: ts_state, messages: messages)
|
|
2118
|
+
hot_hint = if local_engine? && !english_open
|
|
2055
2119
|
'[pwn-ai/p17] Budget-exhaustion is the top open failure on this host. ' \
|
|
2056
2120
|
'Prefer the SHORTEST plan that finishes the ask (≤3 tool calls). ' \
|
|
2057
2121
|
'Emit a final answer as soon as you have evidence — do not explore.'
|
|
@@ -2083,7 +2147,7 @@ module PWN
|
|
|
2083
2147
|
compact_history!(messages: messages) if local
|
|
2084
2148
|
# English-task-as-primary: when plan_idx advanced, tell the model
|
|
2085
2149
|
# which plain-English task is active before the next tool batch.
|
|
2086
|
-
inject_task_focus!(messages: messages, state: ts_state)
|
|
2150
|
+
inject_task_focus!(messages: messages, state: ts_state, request: request)
|
|
2087
2151
|
|
|
2088
2152
|
# P17 — on the final iteration, strip tools and demand a plain-text
|
|
2089
2153
|
# answer. Without this the model happily emits one more tool_calls
|
|
@@ -2116,15 +2180,9 @@ module PWN
|
|
|
2116
2180
|
plan_faithful = hot && plan_steps.positive? && plan_steps <= plan_step_limit &&
|
|
2117
2181
|
turn_fails['empty_final'].to_i.zero? &&
|
|
2118
2182
|
turn_fails.values.sum < 2
|
|
2119
|
-
|
|
2120
|
-
|
|
2121
|
-
|
|
2122
|
-
else
|
|
2123
|
-
(local_engine? ? 3 : 2)
|
|
2124
|
-
end
|
|
2125
|
-
else
|
|
2126
|
-
1
|
|
2127
|
-
end
|
|
2183
|
+
# Last-iter strips tools only on the true last slot. English
|
|
2184
|
+
# leftovers and budget-hot must not steal runway from a live goal.
|
|
2185
|
+
text_only_iters = 1
|
|
2128
2186
|
last_iter = (i >= max_iters - text_only_iters)
|
|
2129
2187
|
if last_iter
|
|
2130
2188
|
tag = i >= max_iters - 1 ? 'FINAL ITERATION' : 'PENULTIMATE — wrap up'
|
|
@@ -2138,7 +2196,7 @@ module PWN
|
|
|
2138
2196
|
}
|
|
2139
2197
|
end
|
|
2140
2198
|
|
|
2141
|
-
msg = call_engine(messages: messages, tools: last_iter ? nil : tools)
|
|
2199
|
+
msg = call_engine(messages: messages, tools: last_iter ? nil : tools, ts_state: ts_state)
|
|
2142
2200
|
if msg.nil?
|
|
2143
2201
|
task_summary_flush!(state: ts_state, on_tool: on_tool)
|
|
2144
2202
|
return '[pwn-ai] engine returned no message'
|
|
@@ -2195,6 +2253,21 @@ module PWN
|
|
|
2195
2253
|
}
|
|
2196
2254
|
next
|
|
2197
2255
|
end
|
|
2256
|
+
if request_unsatisfied?(
|
|
2257
|
+
request: request,
|
|
2258
|
+
messages: messages,
|
|
2259
|
+
last_iter: last_iter
|
|
2260
|
+
) && turn_fails['unsatisfied'].to_i < 4
|
|
2261
|
+
turn_fails['unsatisfied'] += 1
|
|
2262
|
+
warn "[pwn-ai/loop] original request not evidenced on iter=#{i}; continuing"
|
|
2263
|
+
messages << {
|
|
2264
|
+
role: 'user',
|
|
2265
|
+
content: '[pwn-ai] The original request is not evidenced yet. ' \
|
|
2266
|
+
'Keep calling CORE_TOOLS (shell, pwn_eval) until that request is ' \
|
|
2267
|
+
'done or truly blocked. Do not declare completion from a listing alone.'
|
|
2268
|
+
}
|
|
2269
|
+
next
|
|
2270
|
+
end
|
|
2198
2271
|
append_session(session_id: session_id, role: 'assistant', content: text)
|
|
2199
2272
|
Learning.auto_introspect(session_id: session_id, request: request, final: text, predicted: predicted, plan: ts_state && ts_state[:plan], ts_state: ts_state) if defined?(Learning) && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: i)
|
|
2200
2273
|
maybe_finish_policy(session_id: session_id, proxy_ok: true, ts_state: ts_state)
|