pwn 0.5.666 → 0.5.668

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require 'json'
4
+ require 'securerandom'
4
5
  require 'digest'
5
6
  require 'pwn/ai/agent/mistakes'
6
7
 
@@ -60,6 +61,7 @@ module PWN
60
61
  openai: 'PWN::AI::OpenAI',
61
62
  grok: 'PWN::AI::Grok',
62
63
  ollama: 'PWN::AI::Ollama',
64
+ openwebui: 'PWN::AI::OpenWebUI',
63
65
  anthropic: 'PWN::AI::Anthropic',
64
66
  gemini: 'PWN::AI::Gemini'
65
67
  }.freeze
@@ -237,6 +239,11 @@ module PWN
237
239
  # goal was done ("shall I proceed?", "next step:", "want me to…").
238
240
  # Loop.run treats no-tool_calls as FINAL; this detector lets us refuse
239
241
  # that handoff and keep the tool loop alive for multi-step autonomy.
242
+ #
243
+ # Local/thinking models (gemma/Qwen abliterated etc.) often emit a
244
+ # monologue that NARRATES the next tool ("Wait, let's try hping3…")
245
+ # without producing native tool_calls or shell(...). Treat that as
246
+ # incomplete too so the loop re-pressures tools instead of FINAL.
240
247
  INCOMPLETE_FINAL_RX = /
241
248
  \b(shall\s+i|should\s+i|may\s+i|can\s+i|want\s+me\s+to|do\s+you\s+want\s+me|
242
249
  next\s+single\s+step|next\s+step\s*:|awaiting\s+your\s+(ok|approval|go-ahead|confirmation)|
@@ -248,6 +255,21 @@ module PWN
248
255
  )\b
249
256
  /ix
250
257
 
258
+ # Narrated-intent monologue without a structured tool call.
259
+ # Distinct from INCOMPLETE_FINAL_RX (polite handoff to the human).
260
+ MONOLOGUE_TOOL_INTENT_RX = /
261
+ \b(
262
+ wait[,\s]+let'?s\s+try|
263
+ let'?s\s+try\s+(one|to|again|hping|nmap|ping|sudo|shell|running|checking)|
264
+ i\s+(?:will|'ll)\s+(?:just\s+)?(?:try|run|check|probe|scan|use)\b|
265
+ actually,?\s+i\s+will\b|
266
+ one\s+more\s+thing\b|
267
+ if\s+it\s+fails\b.{0,80}\bthen\s+we\s+can\b|
268
+ verification\s+complete\b|
269
+ report\s+that\s+(?:the\s+)?verification\s+failed\b
270
+ )
271
+ /ix
272
+
251
273
  private_class_method def self.incomplete_final?(opts = {})
252
274
  text = opts[:text].to_s
253
275
  return false if text.strip.empty?
@@ -258,6 +280,22 @@ module PWN
258
280
  return true if text.include?('?') && text.length < 900 &&
259
281
  text.match?(/\b(proceed|continue|confirm|apply|next)\b/i)
260
282
 
283
+ # Thinking/monologue that plans tools in prose (esp. Ollama gemma).
284
+ # Cap length so a real long final answer is not bounced forever.
285
+ if text.length < 12_000 && text.match?(MONOLOGUE_TOOL_INTENT_RX)
286
+ # Prefer bounce when no concrete shell(...) / tool call is embedded.
287
+ return true unless defined?(Dispatch) && Dispatch.respond_to?(:tool_calls_from_text)
288
+ return true if Dispatch.tool_calls_from_text(text: text).empty?
289
+
290
+ end
291
+
292
+ # Degenerate repetition ("Wait, let's try…" loop) always incomplete.
293
+ lines = text.lines.map { |l| l.strip.downcase }.reject(&:empty?)
294
+ if lines.length >= 6
295
+ uniq_ratio = lines.uniq.length.to_f / lines.length
296
+ return true if uniq_ratio < 0.35
297
+ end
298
+
261
299
  false
262
300
  rescue StandardError
263
301
  false
@@ -268,7 +306,7 @@ module PWN
268
306
  n = v.to_i.positive? ? v.to_i : DEFAULT_MAX_ITERS
269
307
  # 0.3 — frontier leakage: live max_iters=80 burns local models.
270
308
  # Cap ollama at 25 unless the operator set an explicit lower value.
271
- n = 75 if active_engine == :ollama && n > 75
309
+ n = 75 if local_engine? && n > 75
272
310
  # P7 — W3 controller: when this engine is badly overconfident,
273
311
  # shrink the tool budget so thrash can't compound on bad plans.
274
312
  cal = calibration_state
@@ -282,7 +320,7 @@ module PWN
282
320
  # refuse polite handoffs and finish early when evidence is enough.
283
321
  # CF / red_team forks stay suppressed while hot.
284
322
  if budget_exhaustion_hot?
285
- hot_cap = active_engine == :ollama ? 24 : 75
323
+ hot_cap = local_engine? ? 24 : 75
286
324
  n = [n, hot_cap].min
287
325
  end
288
326
  n
@@ -311,7 +349,7 @@ module PWN
311
349
  remote_cap = 120
312
350
  local_cap = 24
313
351
  cap = if bad
314
- (eng == :ollama ? local_cap : remote_cap)
352
+ (local_engine?(engine: eng) ? local_cap : remote_cap)
315
353
  else
316
354
  75
317
355
  end
@@ -326,6 +364,16 @@ module PWN
326
364
  { overconfident: false, force_plan: false, force_critic: false, max_iters_cap: 75 }
327
365
  end
328
366
 
367
+ # Local/on-box engines share tight-context scaffolding (plan_first,
368
+ # tool_router, result caps, tool_choice pressure). Both :ollama
369
+ # (native server) and :openwebui (gateway in front of models) count.
370
+ private_class_method def self.local_engine?(opts = {})
371
+ eng = opts.key?(:engine) ? opts[:engine].to_s.downcase.to_sym : active_engine
372
+ %i[ollama openwebui].include?(eng)
373
+ rescue StandardError
374
+ false
375
+ end
376
+
329
377
  private_class_method def self.active_engine
330
378
  e = (PWN::Env.dig(:ai, :active) if defined?(PWN::Env)).to_s.downcase.to_sym
331
379
  e == :'' ? :openai : e
@@ -452,7 +500,7 @@ module PWN
452
500
  rescue StandardError
453
501
  false
454
502
  end
455
- plan_prompt = if hot && active_engine == :ollama
503
+ plan_prompt = if hot && local_engine?
456
504
  # Local hot: ultra-short plan — finish-under-8 is the skill gap.
457
505
  'Before acting: write AT MOST 3 numbered tool calls (name + key args) that finish the ask. Prefer fewer. LAST line: "p(success)=<0.0-1.0>". Reply ONLY with the plan + that line — no tools, no prose.'
458
506
  elsif hot
@@ -587,6 +635,184 @@ module PWN
587
635
  warn "[pwn-ai/loop] publish_usage swallowed: #{e.class}: #{e.message}"
588
636
  end
589
637
 
638
+ # Ollama / Open WebUI proxied /ollama/api/chat expect function.arguments
639
+ # as a JSON *object* (map), not a JSON-encoded string. Loop.normalize_llm
640
+ # and openai_wire_tool_call stringify for OpenAI/xAI; replaying that
641
+ # history into local engines produces HTTP 400:
642
+ # {"detail":"Value looks like object, but can't find closing '}' symbol"}
643
+ # because the gateway parses the string as if it were raw object text.
644
+ private_class_method def self.parse_tool_arguments(opts = {})
645
+ raw = opts[:arguments]
646
+ case raw
647
+ when Hash, Array
648
+ # Object form (Hash) is Ollama-native; Array is rare but keep as-is.
649
+ raw
650
+ when nil
651
+ {}
652
+ when String
653
+ s = raw.strip
654
+ return {} if s.empty?
655
+
656
+ begin
657
+ parsed = JSON.parse(s, symbolize_names: true)
658
+ return parsed if parsed.is_a?(Hash) || parsed.is_a?(Array)
659
+ rescue JSON::ParserError
660
+ # fall through
661
+ end
662
+ # Non-JSON bare string — wrap so the schema still gets an object.
663
+ { value: s }
664
+ else
665
+ { value: raw.to_s }
666
+ end
667
+ end
668
+
669
+ private_class_method def self.ollama_wire_tool_call(opts = {})
670
+ tc = opts[:tool_call]
671
+ return nil unless tc.is_a?(Hash)
672
+
673
+ fn = tc[:function] || tc['function'] || {}
674
+ name = fn[:name] || fn['name'] || tc[:name] || tc['name']
675
+ args = fn[:arguments] || fn['arguments'] || tc[:arguments] || tc['arguments']
676
+ {
677
+ id: (tc[:id] || tc['id'] || "call_#{SecureRandom.hex(4)}").to_s,
678
+ type: (tc[:type] || tc['type'] || 'function').to_s,
679
+ function: {
680
+ name: name.to_s,
681
+ arguments: parse_tool_arguments(arguments: args)
682
+ }
683
+ }
684
+ end
685
+
686
+ # Supported Method Parameters::
687
+ # wire = PWN::AI::Agent::Loop.ollama_wire_messages(
688
+ # messages: 'required - in-memory OpenAI-ish messages (may have String args)'
689
+ # )
690
+ #
691
+ # Returns a deep-copied array safe for Ollama / Open WebUI ollama/api/chat:
692
+ # - parses JSON-string function.arguments into Hash/Array objects
693
+ # - coerces nil assistant content to '' when tool_calls present
694
+ # (Open WebUI GenerateChatCompletionForm rejects content:null alone)
695
+ # - drops _native_content / _text_tool_coerced / thinking private keys
696
+ # - stringifies Hash/Array message content (tool results) to JSON text
697
+
698
+ public_class_method def self.ollama_wire_messages(opts = {})
699
+ messages = opts[:messages]
700
+ Array(messages).filter_map do |m|
701
+ next unless m.is_a?(Hash)
702
+
703
+ role = (m[:role] || m['role']).to_s
704
+ out = { role: role }
705
+
706
+ tcs = m[:tool_calls] || m['tool_calls']
707
+ wired_tcs = nil
708
+ if tcs
709
+ wired_tcs = Array(tcs).filter_map { |tc| ollama_wire_tool_call(tool_call: tc) }
710
+ out[:tool_calls] = wired_tcs unless wired_tcs.empty?
711
+ end
712
+
713
+ if m.key?(:content) || m.key?('content')
714
+ content = m.key?(:content) ? m[:content] : m['content']
715
+ out[:content] = case content
716
+ when nil
717
+ # Open WebUI: null content without tool_calls 400s;
718
+ # with tool_calls prefer "" over null.
719
+ wired_tcs && !wired_tcs.empty? ? '' : nil
720
+ when String then content
721
+ when Hash, Array then JSON.generate(content)
722
+ else content.to_s
723
+ end
724
+ elsif wired_tcs && !wired_tcs.empty?
725
+ out[:content] = ''
726
+ end
727
+
728
+ name = m[:name] || m['name']
729
+ out[:name] = name.to_s if name && !name.to_s.empty?
730
+
731
+ tcid = m[:tool_call_id] || m['tool_call_id']
732
+ out[:tool_call_id] = tcid.to_s if tcid && !tcid.to_s.empty?
733
+
734
+ out
735
+ end
736
+ end
737
+
738
+ # OpenAI-compatible wire format (xAI Grok / OpenAI chat.completions):
739
+ # function.arguments MUST be a JSON string, not a map; message.content
740
+ # MUST be a string (or null). Internal Hash arguments from Ollama or
741
+ # Dispatch.tool_calls_from_text caused:
742
+ # 422 Unprocessable Entity: messages[N]: invalid type: map, expected a string
743
+ private_class_method def self.stringify_tool_arguments(opts = {})
744
+ raw = opts[:arguments]
745
+ case raw
746
+ when String then raw.empty? ? '{}' : raw
747
+ when Hash, Array then JSON.generate(raw)
748
+ when nil then '{}'
749
+ else
750
+ s = raw.to_s
751
+ s.empty? ? '{}' : s
752
+ end
753
+ end
754
+
755
+ private_class_method def self.openai_wire_tool_call(opts = {})
756
+ tc = opts[:tool_call]
757
+ return nil unless tc.is_a?(Hash)
758
+
759
+ fn = tc[:function] || tc['function'] || {}
760
+ name = fn[:name] || fn['name'] || tc[:name] || tc['name']
761
+ args = fn[:arguments] || fn['arguments'] || tc[:arguments] || tc['arguments']
762
+ {
763
+ id: (tc[:id] || tc['id'] || "call_#{SecureRandom.hex(4)}").to_s,
764
+ type: (tc[:type] || tc['type'] || 'function').to_s,
765
+ function: {
766
+ name: name.to_s,
767
+ arguments: stringify_tool_arguments(arguments: args)
768
+ }
769
+ }
770
+ end
771
+
772
+ # Supported Method Parameters::
773
+ # wire = PWN::AI::Agent::Loop.openai_wire_messages(
774
+ # messages: 'required - in-memory OpenAI-ish messages (may have Hash args / internal keys)'
775
+ # )
776
+ #
777
+ # Returns a deep-copied array safe for OpenAI / xAI chat.completions:
778
+ # - drops _native_content / _text_tool_coerced / thinking private keys
779
+ # - stringifies function.arguments maps
780
+ # - coerces Hash/non-string content to JSON/string (nil kept for assistant tool turns)
781
+
782
+ public_class_method def self.openai_wire_messages(opts = {})
783
+ messages = opts[:messages]
784
+ Array(messages).filter_map do |m|
785
+ next unless m.is_a?(Hash)
786
+
787
+ role = (m[:role] || m['role']).to_s
788
+ out = { role: role }
789
+
790
+ if m.key?(:content) || m.key?('content')
791
+ content = m.key?(:content) ? m[:content] : m['content']
792
+ out[:content] = case content
793
+ when nil then nil
794
+ when String then content
795
+ when Hash, Array then JSON.generate(content)
796
+ else content.to_s
797
+ end
798
+ end
799
+
800
+ name = m[:name] || m['name']
801
+ out[:name] = name.to_s if name && !name.to_s.empty?
802
+
803
+ tcid = m[:tool_call_id] || m['tool_call_id']
804
+ out[:tool_call_id] = tcid.to_s if tcid && !tcid.to_s.empty?
805
+
806
+ tcs = m[:tool_calls] || m['tool_calls']
807
+ if tcs
808
+ wired = Array(tcs).filter_map { |tc| openai_wire_tool_call(tool_call: tc) }
809
+ out[:tool_calls] = wired unless wired.empty?
810
+ end
811
+
812
+ out
813
+ end
814
+ end
815
+
590
816
  # Supported Method Parameters::
591
817
  # msg = PWN::AI::Agent::Loop.normalize_llm(
592
818
  # response: 'required - chat_with_tools response Hash from any provider'
@@ -618,7 +844,9 @@ module PWN
618
844
  type: 'function',
619
845
  function: {
620
846
  name: tc.dig(:function, :name) || tc[:name],
621
- arguments: tc.dig(:function, :arguments) || tc[:arguments]
847
+ arguments: stringify_tool_arguments(
848
+ arguments: tc.dig(:function, :arguments) || tc[:arguments]
849
+ )
622
850
  }
623
851
  }
624
852
  end
@@ -628,6 +856,17 @@ module PWN
628
856
  # original tool_use block to precede a tool_result).
629
857
  out[:_native_content] = msg[:_native_content] if msg[:_native_content]
630
858
  out[:thinking] = msg[:thinking] if msg[:thinking]
859
+ # Local/abliterated models sometimes emit shell(...) as plain content
860
+ # with empty tool_calls. Coerce registered call-shaped text into
861
+ # structured tool_calls so Loop dispatches instead of FINAL-answering.
862
+ if out[:tool_calls].empty? && defined?(Dispatch) && Dispatch.respond_to?(:tool_calls_from_text)
863
+ coerced = Dispatch.tool_calls_from_text(text: out[:content].to_s)
864
+ if coerced.any?
865
+ out[:tool_calls] = coerced.map { |tc| openai_wire_tool_call(tool_call: tc) }
866
+ out[:content] = nil
867
+ out[:_text_tool_coerced] = true
868
+ end
869
+ end
631
870
  out
632
871
  end
633
872
 
@@ -652,11 +891,50 @@ module PWN
652
891
 
653
892
  mod = Object.const_get(mod_name)
654
893
  if mod.respond_to?(:chat_with_tools)
655
- response = mod.chat_with_tools(
656
- messages: messages,
894
+ # xAI/OpenAI reject Hash function.arguments / Hash content (422 map→string).
895
+ # Ollama / Open WebUI reject *string* function.arguments (HTTP 400
896
+ # "can't find closing '}' symbol") — opposite of OpenAI wire form.
897
+ wire_msgs = if %i[grok openai].include?(engine)
898
+ openai_wire_messages(messages: messages)
899
+ elsif local_engine?(engine: engine)
900
+ ollama_wire_messages(messages: messages)
901
+ else
902
+ messages
903
+ end
904
+ cwt_opts = {
905
+ messages: wire_msgs,
657
906
  tools: tools,
658
907
  spinner: true
659
- )
908
+ }
909
+ # Ollama + abliterated / weak chat-templates often ignore tools: and
910
+ # answer in prose (or print shell(...) as text). Force native
911
+ # tool_calls until at least one tool result is already in history;
912
+ # after that, auto so the model can emit a real final answer.
913
+ # Respect explicit PWN::Env[:ai][:ollama][:tool_choice] override.
914
+ if local_engine?(engine: engine) && tools && !tools.empty?
915
+ env_tc = begin
916
+ PWN::Env.dig(:ai, engine, :tool_choice)
917
+ rescue StandardError
918
+ nil
919
+ end
920
+ if env_tc && !env_tc.to_s.empty?
921
+ cwt_opts[:tool_choice] = env_tc
922
+ else
923
+ # Weak chat templates (TEMPLATE {{ .Prompt }} on abliterated
924
+ # Gemma etc.) ignore tools: under tool_choice=auto and dump
925
+ # monologue as content. Stay on required until a tool result
926
+ # exists AND the last assistant turn already looks like a
927
+ # genuine final (no monologue / handoff markers). That keeps
928
+ # pressure on native tool_calls through the mid-loop thrash
929
+ # that previously returned "Wait, let's try hping3…" as FINAL.
930
+ has_tool_result = Array(messages).any? { |m| m[:role].to_s == 'tool' }
931
+ last_asst = Array(messages).reverse.find { |m| m[:role].to_s == 'assistant' }
932
+ last_txt = last_asst.is_a?(Hash) ? last_asst[:content].to_s : ''
933
+ still_acting = last_txt.strip.empty? || incomplete_final?(text: last_txt, last_iter: false)
934
+ cwt_opts[:tool_choice] = has_tool_result && !still_acting ? 'auto' : 'required'
935
+ end
936
+ end
937
+ response = mod.chat_with_tools(**cwt_opts)
660
938
  publish_usage(response: response, engine: engine)
661
939
  normalize_llm(response: response)
662
940
  else
@@ -842,7 +1120,7 @@ module PWN
842
1120
  # Live coalesced "what am I doing" lines for the TUI (not a model tool).
843
1121
  ts_state = (TaskSummarizer.fresh(request: request) if defined?(TaskSummarizer) && TaskSummarizer.enabled? && Thread.current[:pwn_reflect_depth].to_i.zero?)
844
1122
  engine = active_engine
845
- local = engine == :ollama
1123
+ local = local_engine?(engine: engine)
846
1124
  system_role_content = opts[:system_role_content] ||= PWN::AI::Agent::PromptBuilder.build(session_id: session_id, request: request)
847
1125
 
848
1126
  Registry.discover
@@ -889,7 +1167,7 @@ module PWN
889
1167
  inject_task_focus!(messages: messages, state: ts_state, force: true)
890
1168
  end
891
1169
  if budget_exhaustion_hot?
892
- hot_hint = if active_engine == :ollama
1170
+ hot_hint = if local_engine?
893
1171
  '[pwn-ai/p17] Budget-exhaustion is the top open failure on this host. ' \
894
1172
  'Prefer the SHORTEST plan that finishes the ask (≤3 tool calls). ' \
895
1173
  'Emit a final answer as soon as you have evidence — do not explore.'
@@ -949,7 +1227,7 @@ module PWN
949
1227
  # Plan-faithful: delay the text-only strip when a plan is executing
950
1228
  # cleanly. Remote hot allows longer plans (runway 25); local hot
951
1229
  # still favors short plans under the 8-iter cap.
952
- plan_step_limit = active_engine == :ollama ? 3 : 12
1230
+ plan_step_limit = local_engine? ? 3 : 12
953
1231
  plan_faithful = hot && plan_steps.positive? && plan_steps <= plan_step_limit &&
954
1232
  turn_fails['empty_final'].to_i.zero? &&
955
1233
  turn_fails.values.sum < 2
@@ -957,7 +1235,7 @@ module PWN
957
1235
  if plan_faithful
958
1236
  1
959
1237
  else
960
- (active_engine == :ollama ? 3 : 2)
1238
+ (local_engine? ? 3 : 2)
961
1239
  end
962
1240
  else
963
1241
  1
@@ -984,6 +1262,20 @@ module PWN
984
1262
  calls = Array(msg[:tool_calls])
985
1263
  text = msg[:content].to_s
986
1264
 
1265
+ # Belt-and-suspenders: plain-text shell(...) / tool forms from local
1266
+ # models under weak TEMPLATE {{ .Prompt }} become real tool_calls.
1267
+ if calls.empty? && !text.strip.empty? && !last_iter &&
1268
+ defined?(Dispatch) && Dispatch.respond_to?(:tool_calls_from_text)
1269
+ coerced = Dispatch.tool_calls_from_text(text: text)
1270
+ if coerced.any?
1271
+ wired = coerced.map { |tc| openai_wire_tool_call(tool_call: tc) }
1272
+ msg = msg.merge(tool_calls: wired, content: nil, _text_tool_coerced: true)
1273
+ calls = wired
1274
+ text = ''
1275
+ warn "[pwn-ai/loop] coerced #{wired.length} text tool call(s) on iter=#{i}" if local
1276
+ end
1277
+ end
1278
+
987
1279
  # Empty-final guard (local/thinking models): Ollama sometimes
988
1280
  # returns done_reason=stop with eval_count<=1, empty content, no
989
1281
  # tool_calls — historically surface as a blank TUI reply. Do NOT
@@ -1005,15 +1297,16 @@ module PWN
1005
1297
 
1006
1298
  if calls.empty?
1007
1299
  # P28 — refuse polite mid-goal handoffs so multi-step tasks stay autonomous.
1008
- if incomplete_final?(text: text, last_iter: last_iter) && turn_fails['incomplete_final'].to_i < 2
1300
+ if incomplete_final?(text: text, last_iter: last_iter) && turn_fails['incomplete_final'].to_i < 4
1009
1301
  turn_fails['incomplete_final'] += 1
1010
1302
  warn "[pwn-ai/loop] incomplete final on iter=#{i}; continuing autonomously"
1011
1303
  messages << {
1012
1304
  role: 'user',
1013
- content: '[pwn-ai/p28] That reply handed control back before the goal was done. ' \
1014
- 'Do NOT ask the user to confirm the next step. Continue with the ' \
1015
- 'necessary tool calls now and finish the goal autonomously. Only ' \
1016
- 'emit a final answer when the request is complete or truly blocked.'
1305
+ content: '[pwn-ai/p28] That reply was incomplete (handoff or narrated next step). ' \
1306
+ 'Do NOT monologue about what you will try. Do NOT ask the user to ' \
1307
+ 'confirm. Emit NATIVE tool_calls NOW (e.g. shell with a concrete ' \
1308
+ 'command). Never print shell(...) as plain text. Only emit a final ' \
1309
+ 'answer when the request is complete or truly blocked with evidence.'
1017
1310
  }
1018
1311
  next
1019
1312
  end
@@ -1191,7 +1484,7 @@ module PWN
1191
1484
  Set PWN::Env[:ai][:active] to choose; PWN::Env[:ai][:agent][:max_iters] to bound.
1192
1485
 
1193
1486
  Local-model scaffolding (PWN::Env[:ai][:agent]):
1194
- :plan_first - Boolean, plan-then-act pre-pass (default: engine == :ollama)
1487
+ :plan_first - Boolean, plan-then-act pre-pass (default: local engine :ollama/:openwebui)
1195
1488
  :tool_router - Boolean/nil, slim Registry.definitions (nil=auto on for ollama)
1196
1489
  :escalation_persona - Swarm persona name for frontier corrective hints when stuck
1197
1490
  :critic - S3 constitutional critic before every final (Boolean)
@@ -39,7 +39,9 @@ module PWN
39
39
  b = budget
40
40
  base = (PWN::Env.dig(:ai, engine, :system_role_content) if defined?(PWN::Env)) || 'You are a world-class introspective offensive cyber security and research engineer. You specialize in discovering zero day vulnerabilities focused on responsible disclosure prior to threat actors discovering and exploiting. You are self-aware of your harness, pwn which begins with the ruby namespace `PWN` operating inside the pwn REPL. For every request you first begin by determining if PWN has a module capable of satisfying the request.'
41
41
 
42
- "
42
+ # Heredoc (not a "..." literal): an unescaped "..." inside a
43
+ # double-quoted string is parsed as Range (begin..."...end).
44
+ <<~PROMPT
43
45
  #{base}
44
46
 
45
47
  ENVIRONMENT
@@ -50,8 +52,13 @@ module PWN
50
52
  session_id : #{session_id || '(none)'}
51
53
 
52
54
  #{memory_block(limit: b[:memory], request: request)}#{skills_block}#{learning_block(limit: b[:learning])}#{mistakes_block(limit: b[:mistakes], request: request)}#{metrics_block(limit: b[:metrics], engine: engine)}#{extrospection_block if b[:extro]}TOOL USE
53
- Use the provided function tools to act on the host. A reply with
54
- no tool_calls is treated as your FINAL answer to the user.
55
+ Use the provided function tools to act on the host via NATIVE
56
+ tool_calls / function calling — never print tool invocations as
57
+ plain text (e.g. do NOT write shell(command="...") as your answer).
58
+ Never narrate the next step in prose ("Wait, let's try hping3…",
59
+ "I will run…", "one more thing…") — that is treated as an incomplete
60
+ reply. Emit a real tool_call instead, or a complete final answer
61
+ with evidence. A reply with no tool_calls is your FINAL answer to the user.
55
62
  Prefer `pwn_eval` for anything in the PWN:: namespace and `shell`
56
63
  for OS commands. Save durable facts with `memory_remember`.
57
64
 
@@ -63,7 +70,7 @@ module PWN
63
70
  irreversible destructive action, or missing external decision is
64
71
  strictly required. Partial progress reports without completing the
65
72
  goal are incorrect behavior.
66
- "
73
+ PROMPT
67
74
  end
68
75
 
69
76
  # Supported Method Parameters::
@@ -77,7 +84,7 @@ module PWN
77
84
  public_class_method def self.budget
78
85
  eng = active_engine
79
86
  b = (PWN::Env.dig(:ai, eng, :prompt_budget) if defined?(PWN::Env)) || {}
80
- local = eng == :ollama
87
+ local = %i[ollama openwebui].include?(eng)
81
88
  {
82
89
  memory: (b[:memory] || (local ? 6 : 25)).to_i,
83
90
  metrics: (b[:metrics] || (local ? 3 : 8)).to_i,
@@ -44,6 +44,7 @@ module PWN
44
44
  openai: 'PWN::AI::OpenAI',
45
45
  grok: 'PWN::AI::Grok',
46
46
  ollama: 'PWN::AI::Ollama',
47
+ openwebui: 'PWN::AI::OpenWebUI',
47
48
  anthropic: 'PWN::AI::Anthropic',
48
49
  gemini: 'PWN::AI::Gemini'
49
50
  }.freeze
@@ -215,7 +215,7 @@ module PWN
215
215
  # schema tokens/turn); off for frontier unless explicitly enabled.
216
216
  return v ? true : false unless v.nil?
217
217
 
218
- PWN::Env.dig(:ai, :active).to_s.downcase.to_sym == :ollama
218
+ %i[ollama openwebui].include?(PWN::Env.dig(:ai, :active).to_s.downcase.to_sym)
219
219
  rescue StandardError
220
220
  false
221
221
  end
@@ -47,9 +47,9 @@ module PWN
47
47
  # a 24k tool dump on every call eats the window before useful work.
48
48
  # Override via PWN::Env[:ai][:ollama][:result_max].
49
49
  public_class_method def self.default_max
50
- eng = (PWN::Env.dig(:ai, :active) if defined?(PWN::Env)).to_s.downcase
51
- if eng == 'ollama'
52
- v = (PWN::Env.dig(:ai, :ollama, :result_max) if defined?(PWN::Env))
50
+ eng = (PWN::Env.dig(:ai, :active) if defined?(PWN::Env)).to_s.downcase.to_sym
51
+ if %i[ollama openwebui].include?(eng)
52
+ v = (PWN::Env.dig(:ai, eng, :result_max) if defined?(PWN::Env))
53
53
  return v.to_i if v.to_i.positive?
54
54
 
55
55
  return LOCAL_DEFAULT_MAX
@@ -60,7 +60,7 @@ module PWN
60
60
  # name: 'required - persona name (snake_case)',
61
61
  # role: 'required - system_role_content overlay for this persona',
62
62
  # toolsets: 'optional - Array of Registry toolset names',
63
- # engine: 'optional - :openai / :anthropic / :grok / :gemini / :ollama',
63
+ # engine: 'optional - :openai / :anthropic / :grok / :gemini / :ollama / :openwebui',
64
64
  # max_iters: 'optional - per-turn iteration cap for this persona'
65
65
  # )
66
66
 
@@ -40,6 +40,7 @@ module PWN
40
40
  openai: 'PWN::AI::OpenAI',
41
41
  grok: 'PWN::AI::Grok',
42
42
  ollama: 'PWN::AI::Ollama',
43
+ openwebui: 'PWN::AI::OpenWebUI',
43
44
  anthropic: 'PWN::AI::Anthropic',
44
45
  gemini: 'PWN::AI::Gemini'
45
46
  }.freeze
@@ -39,19 +39,33 @@ PWN::AI::Agent::Registry.register(
39
39
  toolset: 'memory',
40
40
  schema: {
41
41
  name: 'memory_recall',
42
- description: 'Search persistent memory for entries matching a substring query.',
42
+ description: 'Recall context for the active session: starts at the previous ' \
43
+ 'assistant response and walks backward through the current ' \
44
+ 'session only (skipping empty/invalid turns), then fills with ' \
45
+ 'matching durable ~/.pwn/memory.json entries (newest first).',
43
46
  parameters: {
44
47
  type: 'object',
45
48
  properties: {
46
- query: { type: 'string', description: 'Substring to match against keys and values. Omit for all.' },
47
- limit: { type: 'integer', default: 20 }
49
+ query: {
50
+ type: 'string',
51
+ description: 'Substring to match against session content / durable keys and values. Omit for all.'
52
+ },
53
+ limit: { type: 'integer', default: 20 },
54
+ session_id: {
55
+ type: 'string',
56
+ description: 'Optional session id (default: active PWN::Env / current session).'
57
+ }
48
58
  },
49
59
  required: []
50
60
  }
51
61
  },
52
62
  check: -> { defined?(PWN::Memory) },
53
63
  handler: lambda { |args|
54
- PWN::Memory.recall(query: args[:query], limit: args[:limit] || 20)
64
+ PWN::Memory.recall(
65
+ query: args[:query],
66
+ limit: args[:limit] || 20,
67
+ session_id: args[:session_id]
68
+ )
55
69
  }
56
70
  )
57
71
 
data/lib/pwn/ai/grok.rb CHANGED
@@ -513,6 +513,9 @@ module PWN
513
513
  messages = opts[:messages]
514
514
  raise 'ERROR: messages array is required' if messages.nil? || messages.empty?
515
515
 
516
+ # xAI rejects Hash function.arguments / Hash content (422 map → string).
517
+ messages = PWN::AI::Agent::Loop.openai_wire_messages(messages: messages) if defined?(PWN::AI::Agent::Loop) && PWN::AI::Agent::Loop.respond_to?(:openai_wire_messages)
518
+
516
519
  model = opts[:model] ||= engine[:model]
517
520
  raise 'ERROR: Model is required. Call #get_models method for details' if model.nil?
518
521