pwn 0.5.666 → 0.5.668
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.rubocop_todo.yml +1 -0
- data/Gemfile +4 -4
- data/README.md +12 -3
- data/lib/pwn/ai/agent/dispatch.rb +161 -0
- data/lib/pwn/ai/agent/extrospection.rb +9 -5
- data/lib/pwn/ai/agent/learning.rb +2 -1
- data/lib/pwn/ai/agent/loop.rb +311 -18
- data/lib/pwn/ai/agent/prompt_builder.rb +12 -5
- data/lib/pwn/ai/agent/reflect.rb +1 -0
- data/lib/pwn/ai/agent/registry.rb +1 -1
- data/lib/pwn/ai/agent/result.rb +3 -3
- data/lib/pwn/ai/agent/swarm.rb +1 -1
- data/lib/pwn/ai/agent/task_summarizer.rb +1 -0
- data/lib/pwn/ai/agent/tools/memory.rb +18 -4
- data/lib/pwn/ai/grok.rb +3 -0
- data/lib/pwn/ai/ollama.rb +131 -78
- data/lib/pwn/ai/open_ai.rb +3 -0
- data/lib/pwn/ai/open_web_ui.rb +715 -0
- data/lib/pwn/ai.rb +1 -0
- data/lib/pwn/config.rb +70 -36
- data/lib/pwn/memory.rb +140 -11
- data/lib/pwn/memory_index.rb +65 -19
- data/lib/pwn/plugins/repl.rb +6 -4
- data/lib/pwn/version.rb +1 -1
- data/spec/integration/reinforced_feedback_loop_spec.rb +8 -6
- data/spec/lib/pwn/ai/agent/dispatch_spec.rb +78 -0
- data/spec/lib/pwn/ai/agent/extrospection_spec.rb +36 -0
- data/spec/lib/pwn/ai/agent/loop_spec.rb +117 -0
- data/spec/lib/pwn/ai/grok_spec.rb +5 -0
- data/spec/lib/pwn/ai/ollama_spec.rb +36 -0
- data/spec/lib/pwn/ai/open_ai_spec.rb +5 -0
- data/spec/lib/pwn/ai/open_web_ui_spec.rb +302 -0
- data/spec/lib/pwn/config_spec.rb +70 -0
- data/spec/lib/pwn/memory_spec.rb +63 -0
- data/third_party/pwn_rdoc.jsonl +32 -5
- metadata +11 -9
data/lib/pwn/ai/agent/loop.rb
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require 'json'
|
|
4
|
+
require 'securerandom'
|
|
4
5
|
require 'digest'
|
|
5
6
|
require 'pwn/ai/agent/mistakes'
|
|
6
7
|
|
|
@@ -60,6 +61,7 @@ module PWN
|
|
|
60
61
|
openai: 'PWN::AI::OpenAI',
|
|
61
62
|
grok: 'PWN::AI::Grok',
|
|
62
63
|
ollama: 'PWN::AI::Ollama',
|
|
64
|
+
openwebui: 'PWN::AI::OpenWebUI',
|
|
63
65
|
anthropic: 'PWN::AI::Anthropic',
|
|
64
66
|
gemini: 'PWN::AI::Gemini'
|
|
65
67
|
}.freeze
|
|
@@ -237,6 +239,11 @@ module PWN
|
|
|
237
239
|
# goal was done ("shall I proceed?", "next step:", "want me to…").
|
|
238
240
|
# Loop.run treats no-tool_calls as FINAL; this detector lets us refuse
|
|
239
241
|
# that handoff and keep the tool loop alive for multi-step autonomy.
|
|
242
|
+
#
|
|
243
|
+
# Local/thinking models (gemma/Qwen abliterated etc.) often emit a
|
|
244
|
+
# monologue that NARRATES the next tool ("Wait, let's try hping3…")
|
|
245
|
+
# without producing native tool_calls or shell(...). Treat that as
|
|
246
|
+
# incomplete too so the loop re-pressures tools instead of FINAL.
|
|
240
247
|
INCOMPLETE_FINAL_RX = /
|
|
241
248
|
\b(shall\s+i|should\s+i|may\s+i|can\s+i|want\s+me\s+to|do\s+you\s+want\s+me|
|
|
242
249
|
next\s+single\s+step|next\s+step\s*:|awaiting\s+your\s+(ok|approval|go-ahead|confirmation)|
|
|
@@ -248,6 +255,21 @@ module PWN
|
|
|
248
255
|
)\b
|
|
249
256
|
/ix
|
|
250
257
|
|
|
258
|
+
# Narrated-intent monologue without a structured tool call.
|
|
259
|
+
# Distinct from INCOMPLETE_FINAL_RX (polite handoff to the human).
|
|
260
|
+
MONOLOGUE_TOOL_INTENT_RX = /
|
|
261
|
+
\b(
|
|
262
|
+
wait[,\s]+let'?s\s+try|
|
|
263
|
+
let'?s\s+try\s+(one|to|again|hping|nmap|ping|sudo|shell|running|checking)|
|
|
264
|
+
i\s+(?:will|'ll)\s+(?:just\s+)?(?:try|run|check|probe|scan|use)\b|
|
|
265
|
+
actually,?\s+i\s+will\b|
|
|
266
|
+
one\s+more\s+thing\b|
|
|
267
|
+
if\s+it\s+fails\b.{0,80}\bthen\s+we\s+can\b|
|
|
268
|
+
verification\s+complete\b|
|
|
269
|
+
report\s+that\s+(?:the\s+)?verification\s+failed\b
|
|
270
|
+
)
|
|
271
|
+
/ix
|
|
272
|
+
|
|
251
273
|
private_class_method def self.incomplete_final?(opts = {})
|
|
252
274
|
text = opts[:text].to_s
|
|
253
275
|
return false if text.strip.empty?
|
|
@@ -258,6 +280,22 @@ module PWN
|
|
|
258
280
|
return true if text.include?('?') && text.length < 900 &&
|
|
259
281
|
text.match?(/\b(proceed|continue|confirm|apply|next)\b/i)
|
|
260
282
|
|
|
283
|
+
# Thinking/monologue that plans tools in prose (esp. Ollama gemma).
|
|
284
|
+
# Cap length so a real long final answer is not bounced forever.
|
|
285
|
+
if text.length < 12_000 && text.match?(MONOLOGUE_TOOL_INTENT_RX)
|
|
286
|
+
# Prefer bounce when no concrete shell(...) / tool call is embedded.
|
|
287
|
+
return true unless defined?(Dispatch) && Dispatch.respond_to?(:tool_calls_from_text)
|
|
288
|
+
return true if Dispatch.tool_calls_from_text(text: text).empty?
|
|
289
|
+
|
|
290
|
+
end
|
|
291
|
+
|
|
292
|
+
# Degenerate repetition ("Wait, let's try…" loop) always incomplete.
|
|
293
|
+
lines = text.lines.map { |l| l.strip.downcase }.reject(&:empty?)
|
|
294
|
+
if lines.length >= 6
|
|
295
|
+
uniq_ratio = lines.uniq.length.to_f / lines.length
|
|
296
|
+
return true if uniq_ratio < 0.35
|
|
297
|
+
end
|
|
298
|
+
|
|
261
299
|
false
|
|
262
300
|
rescue StandardError
|
|
263
301
|
false
|
|
@@ -268,7 +306,7 @@ module PWN
|
|
|
268
306
|
n = v.to_i.positive? ? v.to_i : DEFAULT_MAX_ITERS
|
|
269
307
|
# 0.3 — frontier leakage: live max_iters=80 burns local models.
|
|
270
308
|
# Cap ollama at 25 unless the operator set an explicit lower value.
|
|
271
|
-
n = 75 if
|
|
309
|
+
n = 75 if local_engine? && n > 75
|
|
272
310
|
# P7 — W3 controller: when this engine is badly overconfident,
|
|
273
311
|
# shrink the tool budget so thrash can't compound on bad plans.
|
|
274
312
|
cal = calibration_state
|
|
@@ -282,7 +320,7 @@ module PWN
|
|
|
282
320
|
# refuse polite handoffs and finish early when evidence is enough.
|
|
283
321
|
# CF / red_team forks stay suppressed while hot.
|
|
284
322
|
if budget_exhaustion_hot?
|
|
285
|
-
hot_cap =
|
|
323
|
+
hot_cap = local_engine? ? 24 : 75
|
|
286
324
|
n = [n, hot_cap].min
|
|
287
325
|
end
|
|
288
326
|
n
|
|
@@ -311,7 +349,7 @@ module PWN
|
|
|
311
349
|
remote_cap = 120
|
|
312
350
|
local_cap = 24
|
|
313
351
|
cap = if bad
|
|
314
|
-
(eng
|
|
352
|
+
(local_engine?(engine: eng) ? local_cap : remote_cap)
|
|
315
353
|
else
|
|
316
354
|
75
|
|
317
355
|
end
|
|
@@ -326,6 +364,16 @@ module PWN
|
|
|
326
364
|
{ overconfident: false, force_plan: false, force_critic: false, max_iters_cap: 75 }
|
|
327
365
|
end
|
|
328
366
|
|
|
367
|
+
# Local/on-box engines share tight-context scaffolding (plan_first,
|
|
368
|
+
# tool_router, result caps, tool_choice pressure). Both :ollama
|
|
369
|
+
# (native server) and :openwebui (gateway in front of models) count.
|
|
370
|
+
private_class_method def self.local_engine?(opts = {})
|
|
371
|
+
eng = opts.key?(:engine) ? opts[:engine].to_s.downcase.to_sym : active_engine
|
|
372
|
+
%i[ollama openwebui].include?(eng)
|
|
373
|
+
rescue StandardError
|
|
374
|
+
false
|
|
375
|
+
end
|
|
376
|
+
|
|
329
377
|
private_class_method def self.active_engine
|
|
330
378
|
e = (PWN::Env.dig(:ai, :active) if defined?(PWN::Env)).to_s.downcase.to_sym
|
|
331
379
|
e == :'' ? :openai : e
|
|
@@ -452,7 +500,7 @@ module PWN
|
|
|
452
500
|
rescue StandardError
|
|
453
501
|
false
|
|
454
502
|
end
|
|
455
|
-
plan_prompt = if hot &&
|
|
503
|
+
plan_prompt = if hot && local_engine?
|
|
456
504
|
# Local hot: ultra-short plan — finish-under-8 is the skill gap.
|
|
457
505
|
'Before acting: write AT MOST 3 numbered tool calls (name + key args) that finish the ask. Prefer fewer. LAST line: "p(success)=<0.0-1.0>". Reply ONLY with the plan + that line — no tools, no prose.'
|
|
458
506
|
elsif hot
|
|
@@ -587,6 +635,184 @@ module PWN
|
|
|
587
635
|
warn "[pwn-ai/loop] publish_usage swallowed: #{e.class}: #{e.message}"
|
|
588
636
|
end
|
|
589
637
|
|
|
638
|
+
# Ollama / Open WebUI proxied /ollama/api/chat expect function.arguments
|
|
639
|
+
# as a JSON *object* (map), not a JSON-encoded string. Loop.normalize_llm
|
|
640
|
+
# and openai_wire_tool_call stringify for OpenAI/xAI; replaying that
|
|
641
|
+
# history into local engines produces HTTP 400:
|
|
642
|
+
# {"detail":"Value looks like object, but can't find closing '}' symbol"}
|
|
643
|
+
# because the gateway parses the string as if it were raw object text.
|
|
644
|
+
private_class_method def self.parse_tool_arguments(opts = {})
|
|
645
|
+
raw = opts[:arguments]
|
|
646
|
+
case raw
|
|
647
|
+
when Hash, Array
|
|
648
|
+
# Object form (Hash) is Ollama-native; Array is rare but keep as-is.
|
|
649
|
+
raw
|
|
650
|
+
when nil
|
|
651
|
+
{}
|
|
652
|
+
when String
|
|
653
|
+
s = raw.strip
|
|
654
|
+
return {} if s.empty?
|
|
655
|
+
|
|
656
|
+
begin
|
|
657
|
+
parsed = JSON.parse(s, symbolize_names: true)
|
|
658
|
+
return parsed if parsed.is_a?(Hash) || parsed.is_a?(Array)
|
|
659
|
+
rescue JSON::ParserError
|
|
660
|
+
# fall through
|
|
661
|
+
end
|
|
662
|
+
# Non-JSON bare string — wrap so the schema still gets an object.
|
|
663
|
+
{ value: s }
|
|
664
|
+
else
|
|
665
|
+
{ value: raw.to_s }
|
|
666
|
+
end
|
|
667
|
+
end
|
|
668
|
+
|
|
669
|
+
private_class_method def self.ollama_wire_tool_call(opts = {})
|
|
670
|
+
tc = opts[:tool_call]
|
|
671
|
+
return nil unless tc.is_a?(Hash)
|
|
672
|
+
|
|
673
|
+
fn = tc[:function] || tc['function'] || {}
|
|
674
|
+
name = fn[:name] || fn['name'] || tc[:name] || tc['name']
|
|
675
|
+
args = fn[:arguments] || fn['arguments'] || tc[:arguments] || tc['arguments']
|
|
676
|
+
{
|
|
677
|
+
id: (tc[:id] || tc['id'] || "call_#{SecureRandom.hex(4)}").to_s,
|
|
678
|
+
type: (tc[:type] || tc['type'] || 'function').to_s,
|
|
679
|
+
function: {
|
|
680
|
+
name: name.to_s,
|
|
681
|
+
arguments: parse_tool_arguments(arguments: args)
|
|
682
|
+
}
|
|
683
|
+
}
|
|
684
|
+
end
|
|
685
|
+
|
|
686
|
+
# Supported Method Parameters::
|
|
687
|
+
# wire = PWN::AI::Agent::Loop.ollama_wire_messages(
|
|
688
|
+
# messages: 'required - in-memory OpenAI-ish messages (may have String args)'
|
|
689
|
+
# )
|
|
690
|
+
#
|
|
691
|
+
# Returns a deep-copied array safe for Ollama / Open WebUI ollama/api/chat:
|
|
692
|
+
# - parses JSON-string function.arguments into Hash/Array objects
|
|
693
|
+
# - coerces nil assistant content to '' when tool_calls present
|
|
694
|
+
# (Open WebUI GenerateChatCompletionForm rejects content:null alone)
|
|
695
|
+
# - drops _native_content / _text_tool_coerced / thinking private keys
|
|
696
|
+
# - stringifies Hash/Array message content (tool results) to JSON text
|
|
697
|
+
|
|
698
|
+
public_class_method def self.ollama_wire_messages(opts = {})
|
|
699
|
+
messages = opts[:messages]
|
|
700
|
+
Array(messages).filter_map do |m|
|
|
701
|
+
next unless m.is_a?(Hash)
|
|
702
|
+
|
|
703
|
+
role = (m[:role] || m['role']).to_s
|
|
704
|
+
out = { role: role }
|
|
705
|
+
|
|
706
|
+
tcs = m[:tool_calls] || m['tool_calls']
|
|
707
|
+
wired_tcs = nil
|
|
708
|
+
if tcs
|
|
709
|
+
wired_tcs = Array(tcs).filter_map { |tc| ollama_wire_tool_call(tool_call: tc) }
|
|
710
|
+
out[:tool_calls] = wired_tcs unless wired_tcs.empty?
|
|
711
|
+
end
|
|
712
|
+
|
|
713
|
+
if m.key?(:content) || m.key?('content')
|
|
714
|
+
content = m.key?(:content) ? m[:content] : m['content']
|
|
715
|
+
out[:content] = case content
|
|
716
|
+
when nil
|
|
717
|
+
# Open WebUI: null content without tool_calls 400s;
|
|
718
|
+
# with tool_calls prefer "" over null.
|
|
719
|
+
wired_tcs && !wired_tcs.empty? ? '' : nil
|
|
720
|
+
when String then content
|
|
721
|
+
when Hash, Array then JSON.generate(content)
|
|
722
|
+
else content.to_s
|
|
723
|
+
end
|
|
724
|
+
elsif wired_tcs && !wired_tcs.empty?
|
|
725
|
+
out[:content] = ''
|
|
726
|
+
end
|
|
727
|
+
|
|
728
|
+
name = m[:name] || m['name']
|
|
729
|
+
out[:name] = name.to_s if name && !name.to_s.empty?
|
|
730
|
+
|
|
731
|
+
tcid = m[:tool_call_id] || m['tool_call_id']
|
|
732
|
+
out[:tool_call_id] = tcid.to_s if tcid && !tcid.to_s.empty?
|
|
733
|
+
|
|
734
|
+
out
|
|
735
|
+
end
|
|
736
|
+
end
|
|
737
|
+
|
|
738
|
+
# OpenAI-compatible wire format (xAI Grok / OpenAI chat.completions):
|
|
739
|
+
# function.arguments MUST be a JSON string, not a map; message.content
|
|
740
|
+
# MUST be a string (or null). Internal Hash arguments from Ollama or
|
|
741
|
+
# Dispatch.tool_calls_from_text caused:
|
|
742
|
+
# 422 Unprocessable Entity: messages[N]: invalid type: map, expected a string
|
|
743
|
+
private_class_method def self.stringify_tool_arguments(opts = {})
|
|
744
|
+
raw = opts[:arguments]
|
|
745
|
+
case raw
|
|
746
|
+
when String then raw.empty? ? '{}' : raw
|
|
747
|
+
when Hash, Array then JSON.generate(raw)
|
|
748
|
+
when nil then '{}'
|
|
749
|
+
else
|
|
750
|
+
s = raw.to_s
|
|
751
|
+
s.empty? ? '{}' : s
|
|
752
|
+
end
|
|
753
|
+
end
|
|
754
|
+
|
|
755
|
+
private_class_method def self.openai_wire_tool_call(opts = {})
|
|
756
|
+
tc = opts[:tool_call]
|
|
757
|
+
return nil unless tc.is_a?(Hash)
|
|
758
|
+
|
|
759
|
+
fn = tc[:function] || tc['function'] || {}
|
|
760
|
+
name = fn[:name] || fn['name'] || tc[:name] || tc['name']
|
|
761
|
+
args = fn[:arguments] || fn['arguments'] || tc[:arguments] || tc['arguments']
|
|
762
|
+
{
|
|
763
|
+
id: (tc[:id] || tc['id'] || "call_#{SecureRandom.hex(4)}").to_s,
|
|
764
|
+
type: (tc[:type] || tc['type'] || 'function').to_s,
|
|
765
|
+
function: {
|
|
766
|
+
name: name.to_s,
|
|
767
|
+
arguments: stringify_tool_arguments(arguments: args)
|
|
768
|
+
}
|
|
769
|
+
}
|
|
770
|
+
end
|
|
771
|
+
|
|
772
|
+
# Supported Method Parameters::
|
|
773
|
+
# wire = PWN::AI::Agent::Loop.openai_wire_messages(
|
|
774
|
+
# messages: 'required - in-memory OpenAI-ish messages (may have Hash args / internal keys)'
|
|
775
|
+
# )
|
|
776
|
+
#
|
|
777
|
+
# Returns a deep-copied array safe for OpenAI / xAI chat.completions:
|
|
778
|
+
# - drops _native_content / _text_tool_coerced / thinking private keys
|
|
779
|
+
# - stringifies function.arguments maps
|
|
780
|
+
# - coerces Hash/non-string content to JSON/string (nil kept for assistant tool turns)
|
|
781
|
+
|
|
782
|
+
public_class_method def self.openai_wire_messages(opts = {})
|
|
783
|
+
messages = opts[:messages]
|
|
784
|
+
Array(messages).filter_map do |m|
|
|
785
|
+
next unless m.is_a?(Hash)
|
|
786
|
+
|
|
787
|
+
role = (m[:role] || m['role']).to_s
|
|
788
|
+
out = { role: role }
|
|
789
|
+
|
|
790
|
+
if m.key?(:content) || m.key?('content')
|
|
791
|
+
content = m.key?(:content) ? m[:content] : m['content']
|
|
792
|
+
out[:content] = case content
|
|
793
|
+
when nil then nil
|
|
794
|
+
when String then content
|
|
795
|
+
when Hash, Array then JSON.generate(content)
|
|
796
|
+
else content.to_s
|
|
797
|
+
end
|
|
798
|
+
end
|
|
799
|
+
|
|
800
|
+
name = m[:name] || m['name']
|
|
801
|
+
out[:name] = name.to_s if name && !name.to_s.empty?
|
|
802
|
+
|
|
803
|
+
tcid = m[:tool_call_id] || m['tool_call_id']
|
|
804
|
+
out[:tool_call_id] = tcid.to_s if tcid && !tcid.to_s.empty?
|
|
805
|
+
|
|
806
|
+
tcs = m[:tool_calls] || m['tool_calls']
|
|
807
|
+
if tcs
|
|
808
|
+
wired = Array(tcs).filter_map { |tc| openai_wire_tool_call(tool_call: tc) }
|
|
809
|
+
out[:tool_calls] = wired unless wired.empty?
|
|
810
|
+
end
|
|
811
|
+
|
|
812
|
+
out
|
|
813
|
+
end
|
|
814
|
+
end
|
|
815
|
+
|
|
590
816
|
# Supported Method Parameters::
|
|
591
817
|
# msg = PWN::AI::Agent::Loop.normalize_llm(
|
|
592
818
|
# response: 'required - chat_with_tools response Hash from any provider'
|
|
@@ -618,7 +844,9 @@ module PWN
|
|
|
618
844
|
type: 'function',
|
|
619
845
|
function: {
|
|
620
846
|
name: tc.dig(:function, :name) || tc[:name],
|
|
621
|
-
arguments:
|
|
847
|
+
arguments: stringify_tool_arguments(
|
|
848
|
+
arguments: tc.dig(:function, :arguments) || tc[:arguments]
|
|
849
|
+
)
|
|
622
850
|
}
|
|
623
851
|
}
|
|
624
852
|
end
|
|
@@ -628,6 +856,17 @@ module PWN
|
|
|
628
856
|
# original tool_use block to precede a tool_result).
|
|
629
857
|
out[:_native_content] = msg[:_native_content] if msg[:_native_content]
|
|
630
858
|
out[:thinking] = msg[:thinking] if msg[:thinking]
|
|
859
|
+
# Local/abliterated models sometimes emit shell(...) as plain content
|
|
860
|
+
# with empty tool_calls. Coerce registered call-shaped text into
|
|
861
|
+
# structured tool_calls so Loop dispatches instead of FINAL-answering.
|
|
862
|
+
if out[:tool_calls].empty? && defined?(Dispatch) && Dispatch.respond_to?(:tool_calls_from_text)
|
|
863
|
+
coerced = Dispatch.tool_calls_from_text(text: out[:content].to_s)
|
|
864
|
+
if coerced.any?
|
|
865
|
+
out[:tool_calls] = coerced.map { |tc| openai_wire_tool_call(tool_call: tc) }
|
|
866
|
+
out[:content] = nil
|
|
867
|
+
out[:_text_tool_coerced] = true
|
|
868
|
+
end
|
|
869
|
+
end
|
|
631
870
|
out
|
|
632
871
|
end
|
|
633
872
|
|
|
@@ -652,11 +891,50 @@ module PWN
|
|
|
652
891
|
|
|
653
892
|
mod = Object.const_get(mod_name)
|
|
654
893
|
if mod.respond_to?(:chat_with_tools)
|
|
655
|
-
|
|
656
|
-
|
|
894
|
+
# xAI/OpenAI reject Hash function.arguments / Hash content (422 map→string).
|
|
895
|
+
# Ollama / Open WebUI reject *string* function.arguments (HTTP 400
|
|
896
|
+
# "can't find closing '}' symbol") — opposite of OpenAI wire form.
|
|
897
|
+
wire_msgs = if %i[grok openai].include?(engine)
|
|
898
|
+
openai_wire_messages(messages: messages)
|
|
899
|
+
elsif local_engine?(engine: engine)
|
|
900
|
+
ollama_wire_messages(messages: messages)
|
|
901
|
+
else
|
|
902
|
+
messages
|
|
903
|
+
end
|
|
904
|
+
cwt_opts = {
|
|
905
|
+
messages: wire_msgs,
|
|
657
906
|
tools: tools,
|
|
658
907
|
spinner: true
|
|
659
|
-
|
|
908
|
+
}
|
|
909
|
+
# Ollama + abliterated / weak chat-templates often ignore tools: and
|
|
910
|
+
# answer in prose (or print shell(...) as text). Force native
|
|
911
|
+
# tool_calls until at least one tool result is already in history;
|
|
912
|
+
# after that, auto so the model can emit a real final answer.
|
|
913
|
+
# Respect explicit PWN::Env[:ai][:ollama][:tool_choice] override.
|
|
914
|
+
if local_engine?(engine: engine) && tools && !tools.empty?
|
|
915
|
+
env_tc = begin
|
|
916
|
+
PWN::Env.dig(:ai, engine, :tool_choice)
|
|
917
|
+
rescue StandardError
|
|
918
|
+
nil
|
|
919
|
+
end
|
|
920
|
+
if env_tc && !env_tc.to_s.empty?
|
|
921
|
+
cwt_opts[:tool_choice] = env_tc
|
|
922
|
+
else
|
|
923
|
+
# Weak chat templates (TEMPLATE {{ .Prompt }} on abliterated
|
|
924
|
+
# Gemma etc.) ignore tools: under tool_choice=auto and dump
|
|
925
|
+
# monologue as content. Stay on required until a tool result
|
|
926
|
+
# exists AND the last assistant turn already looks like a
|
|
927
|
+
# genuine final (no monologue / handoff markers). That keeps
|
|
928
|
+
# pressure on native tool_calls through the mid-loop thrash
|
|
929
|
+
# that previously returned "Wait, let's try hping3…" as FINAL.
|
|
930
|
+
has_tool_result = Array(messages).any? { |m| m[:role].to_s == 'tool' }
|
|
931
|
+
last_asst = Array(messages).reverse.find { |m| m[:role].to_s == 'assistant' }
|
|
932
|
+
last_txt = last_asst.is_a?(Hash) ? last_asst[:content].to_s : ''
|
|
933
|
+
still_acting = last_txt.strip.empty? || incomplete_final?(text: last_txt, last_iter: false)
|
|
934
|
+
cwt_opts[:tool_choice] = has_tool_result && !still_acting ? 'auto' : 'required'
|
|
935
|
+
end
|
|
936
|
+
end
|
|
937
|
+
response = mod.chat_with_tools(**cwt_opts)
|
|
660
938
|
publish_usage(response: response, engine: engine)
|
|
661
939
|
normalize_llm(response: response)
|
|
662
940
|
else
|
|
@@ -842,7 +1120,7 @@ module PWN
|
|
|
842
1120
|
# Live coalesced "what am I doing" lines for the TUI (not a model tool).
|
|
843
1121
|
ts_state = (TaskSummarizer.fresh(request: request) if defined?(TaskSummarizer) && TaskSummarizer.enabled? && Thread.current[:pwn_reflect_depth].to_i.zero?)
|
|
844
1122
|
engine = active_engine
|
|
845
|
-
local = engine
|
|
1123
|
+
local = local_engine?(engine: engine)
|
|
846
1124
|
system_role_content = opts[:system_role_content] ||= PWN::AI::Agent::PromptBuilder.build(session_id: session_id, request: request)
|
|
847
1125
|
|
|
848
1126
|
Registry.discover
|
|
@@ -889,7 +1167,7 @@ module PWN
|
|
|
889
1167
|
inject_task_focus!(messages: messages, state: ts_state, force: true)
|
|
890
1168
|
end
|
|
891
1169
|
if budget_exhaustion_hot?
|
|
892
|
-
hot_hint = if
|
|
1170
|
+
hot_hint = if local_engine?
|
|
893
1171
|
'[pwn-ai/p17] Budget-exhaustion is the top open failure on this host. ' \
|
|
894
1172
|
'Prefer the SHORTEST plan that finishes the ask (≤3 tool calls). ' \
|
|
895
1173
|
'Emit a final answer as soon as you have evidence — do not explore.'
|
|
@@ -949,7 +1227,7 @@ module PWN
|
|
|
949
1227
|
# Plan-faithful: delay the text-only strip when a plan is executing
|
|
950
1228
|
# cleanly. Remote hot allows longer plans (runway 25); local hot
|
|
951
1229
|
# still favors short plans under the 8-iter cap.
|
|
952
|
-
plan_step_limit =
|
|
1230
|
+
plan_step_limit = local_engine? ? 3 : 12
|
|
953
1231
|
plan_faithful = hot && plan_steps.positive? && plan_steps <= plan_step_limit &&
|
|
954
1232
|
turn_fails['empty_final'].to_i.zero? &&
|
|
955
1233
|
turn_fails.values.sum < 2
|
|
@@ -957,7 +1235,7 @@ module PWN
|
|
|
957
1235
|
if plan_faithful
|
|
958
1236
|
1
|
|
959
1237
|
else
|
|
960
|
-
(
|
|
1238
|
+
(local_engine? ? 3 : 2)
|
|
961
1239
|
end
|
|
962
1240
|
else
|
|
963
1241
|
1
|
|
@@ -984,6 +1262,20 @@ module PWN
|
|
|
984
1262
|
calls = Array(msg[:tool_calls])
|
|
985
1263
|
text = msg[:content].to_s
|
|
986
1264
|
|
|
1265
|
+
# Belt-and-suspenders: plain-text shell(...) / tool forms from local
|
|
1266
|
+
# models under weak TEMPLATE {{ .Prompt }} become real tool_calls.
|
|
1267
|
+
if calls.empty? && !text.strip.empty? && !last_iter &&
|
|
1268
|
+
defined?(Dispatch) && Dispatch.respond_to?(:tool_calls_from_text)
|
|
1269
|
+
coerced = Dispatch.tool_calls_from_text(text: text)
|
|
1270
|
+
if coerced.any?
|
|
1271
|
+
wired = coerced.map { |tc| openai_wire_tool_call(tool_call: tc) }
|
|
1272
|
+
msg = msg.merge(tool_calls: wired, content: nil, _text_tool_coerced: true)
|
|
1273
|
+
calls = wired
|
|
1274
|
+
text = ''
|
|
1275
|
+
warn "[pwn-ai/loop] coerced #{wired.length} text tool call(s) on iter=#{i}" if local
|
|
1276
|
+
end
|
|
1277
|
+
end
|
|
1278
|
+
|
|
987
1279
|
# Empty-final guard (local/thinking models): Ollama sometimes
|
|
988
1280
|
# returns done_reason=stop with eval_count<=1, empty content, no
|
|
989
1281
|
# tool_calls — historically surface as a blank TUI reply. Do NOT
|
|
@@ -1005,15 +1297,16 @@ module PWN
|
|
|
1005
1297
|
|
|
1006
1298
|
if calls.empty?
|
|
1007
1299
|
# P28 — refuse polite mid-goal handoffs so multi-step tasks stay autonomous.
|
|
1008
|
-
if incomplete_final?(text: text, last_iter: last_iter) && turn_fails['incomplete_final'].to_i <
|
|
1300
|
+
if incomplete_final?(text: text, last_iter: last_iter) && turn_fails['incomplete_final'].to_i < 4
|
|
1009
1301
|
turn_fails['incomplete_final'] += 1
|
|
1010
1302
|
warn "[pwn-ai/loop] incomplete final on iter=#{i}; continuing autonomously"
|
|
1011
1303
|
messages << {
|
|
1012
1304
|
role: 'user',
|
|
1013
|
-
content: '[pwn-ai/p28] That reply
|
|
1014
|
-
'Do NOT
|
|
1015
|
-
'
|
|
1016
|
-
'
|
|
1305
|
+
content: '[pwn-ai/p28] That reply was incomplete (handoff or narrated next step). ' \
|
|
1306
|
+
'Do NOT monologue about what you will try. Do NOT ask the user to ' \
|
|
1307
|
+
'confirm. Emit NATIVE tool_calls NOW (e.g. shell with a concrete ' \
|
|
1308
|
+
'command). Never print shell(...) as plain text. Only emit a final ' \
|
|
1309
|
+
'answer when the request is complete or truly blocked with evidence.'
|
|
1017
1310
|
}
|
|
1018
1311
|
next
|
|
1019
1312
|
end
|
|
@@ -1191,7 +1484,7 @@ module PWN
|
|
|
1191
1484
|
Set PWN::Env[:ai][:active] to choose; PWN::Env[:ai][:agent][:max_iters] to bound.
|
|
1192
1485
|
|
|
1193
1486
|
Local-model scaffolding (PWN::Env[:ai][:agent]):
|
|
1194
|
-
:plan_first - Boolean, plan-then-act pre-pass (default: engine
|
|
1487
|
+
:plan_first - Boolean, plan-then-act pre-pass (default: local engine :ollama/:openwebui)
|
|
1195
1488
|
:tool_router - Boolean/nil, slim Registry.definitions (nil=auto on for ollama)
|
|
1196
1489
|
:escalation_persona - Swarm persona name for frontier corrective hints when stuck
|
|
1197
1490
|
:critic - S3 constitutional critic before every final (Boolean)
|
|
@@ -39,7 +39,9 @@ module PWN
|
|
|
39
39
|
b = budget
|
|
40
40
|
base = (PWN::Env.dig(:ai, engine, :system_role_content) if defined?(PWN::Env)) || 'You are a world-class introspective offensive cyber security and research engineer. You specialize in discovering zero day vulnerabilities focused on responsible disclosure prior to threat actors discovering and exploiting. You are self-aware of your harness, pwn which begins with the ruby namespace `PWN` operating inside the pwn REPL. For every request you first begin by determining if PWN has a module capable of satisfying the request.'
|
|
41
41
|
|
|
42
|
-
"
|
|
42
|
+
# Heredoc (not a "..." literal): an unescaped "..." inside a
|
|
43
|
+
# double-quoted string is parsed as Range (begin..."...end).
|
|
44
|
+
<<~PROMPT
|
|
43
45
|
#{base}
|
|
44
46
|
|
|
45
47
|
ENVIRONMENT
|
|
@@ -50,8 +52,13 @@ module PWN
|
|
|
50
52
|
session_id : #{session_id || '(none)'}
|
|
51
53
|
|
|
52
54
|
#{memory_block(limit: b[:memory], request: request)}#{skills_block}#{learning_block(limit: b[:learning])}#{mistakes_block(limit: b[:mistakes], request: request)}#{metrics_block(limit: b[:metrics], engine: engine)}#{extrospection_block if b[:extro]}TOOL USE
|
|
53
|
-
Use the provided function tools to act on the host
|
|
54
|
-
|
|
55
|
+
Use the provided function tools to act on the host via NATIVE
|
|
56
|
+
tool_calls / function calling — never print tool invocations as
|
|
57
|
+
plain text (e.g. do NOT write shell(command="...") as your answer).
|
|
58
|
+
Never narrate the next step in prose ("Wait, let's try hping3…",
|
|
59
|
+
"I will run…", "one more thing…") — that is treated as an incomplete
|
|
60
|
+
reply. Emit a real tool_call instead, or a complete final answer
|
|
61
|
+
with evidence. A reply with no tool_calls is your FINAL answer to the user.
|
|
55
62
|
Prefer `pwn_eval` for anything in the PWN:: namespace and `shell`
|
|
56
63
|
for OS commands. Save durable facts with `memory_remember`.
|
|
57
64
|
|
|
@@ -63,7 +70,7 @@ module PWN
|
|
|
63
70
|
irreversible destructive action, or missing external decision is
|
|
64
71
|
strictly required. Partial progress reports without completing the
|
|
65
72
|
goal are incorrect behavior.
|
|
66
|
-
|
|
73
|
+
PROMPT
|
|
67
74
|
end
|
|
68
75
|
|
|
69
76
|
# Supported Method Parameters::
|
|
@@ -77,7 +84,7 @@ module PWN
|
|
|
77
84
|
public_class_method def self.budget
|
|
78
85
|
eng = active_engine
|
|
79
86
|
b = (PWN::Env.dig(:ai, eng, :prompt_budget) if defined?(PWN::Env)) || {}
|
|
80
|
-
local = eng
|
|
87
|
+
local = %i[ollama openwebui].include?(eng)
|
|
81
88
|
{
|
|
82
89
|
memory: (b[:memory] || (local ? 6 : 25)).to_i,
|
|
83
90
|
metrics: (b[:metrics] || (local ? 3 : 8)).to_i,
|
data/lib/pwn/ai/agent/reflect.rb
CHANGED
|
@@ -215,7 +215,7 @@ module PWN
|
|
|
215
215
|
# schema tokens/turn); off for frontier unless explicitly enabled.
|
|
216
216
|
return v ? true : false unless v.nil?
|
|
217
217
|
|
|
218
|
-
PWN::Env.dig(:ai, :active).to_s.downcase.to_sym
|
|
218
|
+
%i[ollama openwebui].include?(PWN::Env.dig(:ai, :active).to_s.downcase.to_sym)
|
|
219
219
|
rescue StandardError
|
|
220
220
|
false
|
|
221
221
|
end
|
data/lib/pwn/ai/agent/result.rb
CHANGED
|
@@ -47,9 +47,9 @@ module PWN
|
|
|
47
47
|
# a 24k tool dump on every call eats the window before useful work.
|
|
48
48
|
# Override via PWN::Env[:ai][:ollama][:result_max].
|
|
49
49
|
public_class_method def self.default_max
|
|
50
|
-
eng = (PWN::Env.dig(:ai, :active) if defined?(PWN::Env)).to_s.downcase
|
|
51
|
-
if eng
|
|
52
|
-
v = (PWN::Env.dig(:ai,
|
|
50
|
+
eng = (PWN::Env.dig(:ai, :active) if defined?(PWN::Env)).to_s.downcase.to_sym
|
|
51
|
+
if %i[ollama openwebui].include?(eng)
|
|
52
|
+
v = (PWN::Env.dig(:ai, eng, :result_max) if defined?(PWN::Env))
|
|
53
53
|
return v.to_i if v.to_i.positive?
|
|
54
54
|
|
|
55
55
|
return LOCAL_DEFAULT_MAX
|
data/lib/pwn/ai/agent/swarm.rb
CHANGED
|
@@ -60,7 +60,7 @@ module PWN
|
|
|
60
60
|
# name: 'required - persona name (snake_case)',
|
|
61
61
|
# role: 'required - system_role_content overlay for this persona',
|
|
62
62
|
# toolsets: 'optional - Array of Registry toolset names',
|
|
63
|
-
# engine: 'optional - :openai / :anthropic / :grok / :gemini / :ollama',
|
|
63
|
+
# engine: 'optional - :openai / :anthropic / :grok / :gemini / :ollama / :openwebui',
|
|
64
64
|
# max_iters: 'optional - per-turn iteration cap for this persona'
|
|
65
65
|
# )
|
|
66
66
|
|
|
@@ -39,19 +39,33 @@ PWN::AI::Agent::Registry.register(
|
|
|
39
39
|
toolset: 'memory',
|
|
40
40
|
schema: {
|
|
41
41
|
name: 'memory_recall',
|
|
42
|
-
description: '
|
|
42
|
+
description: 'Recall context for the active session: starts at the previous ' \
|
|
43
|
+
'assistant response and walks backward through the current ' \
|
|
44
|
+
'session only (skipping empty/invalid turns), then fills with ' \
|
|
45
|
+
'matching durable ~/.pwn/memory.json entries (newest first).',
|
|
43
46
|
parameters: {
|
|
44
47
|
type: 'object',
|
|
45
48
|
properties: {
|
|
46
|
-
query: {
|
|
47
|
-
|
|
49
|
+
query: {
|
|
50
|
+
type: 'string',
|
|
51
|
+
description: 'Substring to match against session content / durable keys and values. Omit for all.'
|
|
52
|
+
},
|
|
53
|
+
limit: { type: 'integer', default: 20 },
|
|
54
|
+
session_id: {
|
|
55
|
+
type: 'string',
|
|
56
|
+
description: 'Optional session id (default: active PWN::Env / current session).'
|
|
57
|
+
}
|
|
48
58
|
},
|
|
49
59
|
required: []
|
|
50
60
|
}
|
|
51
61
|
},
|
|
52
62
|
check: -> { defined?(PWN::Memory) },
|
|
53
63
|
handler: lambda { |args|
|
|
54
|
-
PWN::Memory.recall(
|
|
64
|
+
PWN::Memory.recall(
|
|
65
|
+
query: args[:query],
|
|
66
|
+
limit: args[:limit] || 20,
|
|
67
|
+
session_id: args[:session_id]
|
|
68
|
+
)
|
|
55
69
|
}
|
|
56
70
|
)
|
|
57
71
|
|
data/lib/pwn/ai/grok.rb
CHANGED
|
@@ -513,6 +513,9 @@ module PWN
|
|
|
513
513
|
messages = opts[:messages]
|
|
514
514
|
raise 'ERROR: messages array is required' if messages.nil? || messages.empty?
|
|
515
515
|
|
|
516
|
+
# xAI rejects Hash function.arguments / Hash content (422 map → string).
|
|
517
|
+
messages = PWN::AI::Agent::Loop.openai_wire_messages(messages: messages) if defined?(PWN::AI::Agent::Loop) && PWN::AI::Agent::Loop.respond_to?(:openai_wire_messages)
|
|
518
|
+
|
|
516
519
|
model = opts[:model] ||= engine[:model]
|
|
517
520
|
raise 'ERROR: Model is required. Call #get_models method for details' if model.nil?
|
|
518
521
|
|