pwn 0.5.666 → 0.5.668

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/lib/pwn/ai/ollama.rb CHANGED
@@ -7,17 +7,31 @@ require 'tty-spinner'
7
7
 
8
8
  module PWN
9
9
  module AI
10
- # This plugin is used for interacting w/ Ollama's REST API using
11
- # the 'rest' browser type of PWN::Plugins::TransparentBrowser.
12
- # This is based on the following Ollama API Specification:
13
- # https://api.openai.com/v1
10
+ # Direct client for a local/remote Ollama server REST API.
11
+ # No API key is required for a stock ollama serve (http://127.0.0.1:11434).
12
+ # Paths are native Ollama:
13
+ # GET /api/tags
14
+ # POST /api/chat (native tool_calls, options.num_ctx / num_predict)
15
+ # POST /api/embed
16
+ # POST /v1/chat/completions (OpenAI-compat shim)
17
+ # Spec: https://github.com/ollama/ollama/blob/main/docs/api.md
14
18
  module Ollama
19
+ DEFAULT_BASE_URI = 'http://127.0.0.1:11434'
20
+
21
+ private_class_method def self.real_config_value?(opts = {})
22
+ s = opts[:value].to_s.strip
23
+ return false if s.empty?
24
+ return false if s.match?(/\A(optional|required)\b/i)
25
+ return false if s.match?(/REDACTED/i)
26
+ return false if s.match?(/\A<{3}.*>{3}\z/)
27
+
28
+ true
29
+ end
30
+
15
31
  # Supported Method Parameters::
16
32
  # ollama_rest_call(
17
- # base_uri: 'required - base URI for the Ollama API',
18
- # token: 'required - ollama bearer token',
19
- # http_method: 'optional HTTP method (defaults to GET)
20
- # rest_call: 'required rest call to make per the schema',
33
+ # http_method: 'optional HTTP method (defaults to GET)',
34
+ # rest_call: 'required rest call path relative to base_uri (e.g. api/chat)',
21
35
  # params: 'optional params passed in the URI or HTTP Headers',
22
36
  # http_body: 'optional HTTP body sent in HTTP methods that support it e.g. POST',
23
37
  # timeout: 'optional timeout in seconds (defaults to 900)',
@@ -26,24 +40,28 @@ module PWN
26
40
 
27
41
  private_class_method def self.ollama_rest_call(opts = {})
28
42
  engine = PWN::Env[:ai][:ollama]
29
- raise 'ERROR: Jira Server Hash not found in PWN::Env. Run i`pwn -Y default.yaml`, then `PWN::Env` for usage.' if engine.nil?
43
+ raise 'ERROR: Ollama engine hash not found in PWN::Env[:ai][:ollama]. Run `pwn -Y default.yaml`, then `PWN::Env` for usage.' if engine.nil?
30
44
 
31
- base_uri = engine[:base_uri]
32
- raise 'ERROR: base_uri must be provided in PWN::Env[:ai][:ollama][:base_uri]' if base_uri.nil?
45
+ base_uri = engine[:base_uri].to_s.strip
46
+ base_uri = DEFAULT_BASE_URI if base_uri.empty? || base_uri.match?(/\A(optional|required)\b/i)
47
+ base_uri = base_uri.chomp('/')
48
+
49
+ # Stock ollama needs no key. Only attach Authorization when the operator
50
+ # set a real key (e.g. reverse-proxy basic/JWT in front of ollama).
51
+ token = real_config_value?(value: engine[:key]) ? engine[:key].to_s.strip : nil
33
52
 
34
- token = engine[:key] ||= PWN::Plugins::AuthenticationHelper.mask_password(prompt: 'Ollama (i.e. OpenAPI) Key')
35
53
  http_method = if opts[:http_method].nil?
36
54
  :get
37
55
  else
38
56
  opts[:http_method].to_s.scrub.to_sym
39
57
  end
40
- rest_call = opts[:rest_call].to_s.scrub
58
+ rest_call = opts[:rest_call].to_s.scrub.sub(%r{\A/}, '')
41
59
  params = opts[:params]
42
60
 
43
61
  headers = {
44
- content_type: 'application/json; charset=UTF-8',
45
- authorization: "Bearer #{token}"
62
+ content_type: 'application/json; charset=UTF-8'
46
63
  }
64
+ headers[:authorization] = "Bearer #{token}" if token
47
65
 
48
66
  http_body = opts[:http_body]
49
67
  http_body ||= {}
@@ -101,9 +119,8 @@ module PWN
101
119
  #
102
120
  # ABSOLUTE DEADLINE: Net::HTTP read_timeout only fires on *idle*
103
121
  # gaps between chunks. A thinking model that dribbles tokens
104
- # forever never idles out — the hung pwn-ai spinner on pts/8
105
- # (2026-07-23) had already received 17MB+ after 25 minutes.
106
- # Enforce a wall-clock deadline independent of chunk cadence.
122
+ # forever never idles out. Enforce a wall-clock deadline
123
+ # independent of chunk cadence.
107
124
  assembled = nil
108
125
  stream_err = nil
109
126
  deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + timeout.to_f
@@ -123,7 +140,6 @@ module PWN
123
140
  now = Process.clock_gettime(Process::CLOCK_MONOTONIC)
124
141
  if now > deadline
125
142
  stream_err = "ERROR: Ollama stream absolute timeout after #{timeout}s for #{rest_call} (received #{bytes} bytes - model likely stuck in unbounded thinking; lower num_ctx / set num_predict)"
126
- # Stop reading; RestClient will tear down the socket.
127
143
  raise stream_err
128
144
  end
129
145
  buf << chunk
@@ -157,7 +173,7 @@ module PWN
157
173
  end
158
174
 
159
175
  else
160
- raise @@logger.error("Unsupported HTTP Method #{http_method} for #{self} Plugin")
176
+ raise "Unsupported HTTP Method #{http_method} for #{self} Plugin"
161
177
  end
162
178
  response
163
179
  rescue RestClient::TooManyRequests => e
@@ -168,9 +184,6 @@ module PWN
168
184
  retry
169
185
  end
170
186
  rescue RestClient::ExceptionWithResponse => e
171
- # Never return nil here — chat_with_tools used to `return nil if response.nil?`
172
- # which made the agent loop print "[pwn-ai] engine returned no message"
173
- # with no actionable error. Raise so the caller / REPL surfaces the body.
174
187
  body = begin
175
188
  e.response.to_s[0, 800]
176
189
  rescue StandardError
@@ -195,13 +208,11 @@ module PWN
195
208
  thinking = opts[:thinking].to_s
196
209
  return '' if thinking.strip.empty?
197
210
 
198
- # Common end-of-reasoning markers across distill/abliterated builds.
199
211
  markers = [
200
212
  %r{</think>}i,
201
213
  %r{</thinking>}i,
202
214
  /(?:\A|\n)\s*Final\s*Answer\s*:\s*/i,
203
215
  /(?:\A|\n)\s*Answer\s*:\s*/i,
204
- # Inline form: "... Final Answer: <text>"
205
216
  /Final\s*Answer\s*:\s*/i
206
217
  ]
207
218
  markers.each do |rx|
@@ -221,7 +232,6 @@ module PWN
221
232
  rest_call = opts[:rest_call].to_s
222
233
 
223
234
  lines = body.each_line.map(&:strip).reject(&:empty?)
224
- # Strip SSE "data: " prefix when present; drop the terminal [DONE].
225
235
  payloads = lines.filter_map do |line|
226
236
  line = line.sub(/\Adata:\s*/, '')
227
237
  next if line.empty? || line == '[DONE]'
@@ -233,9 +243,6 @@ module PWN
233
243
  end
234
244
  end
235
245
  if payloads.empty?
236
- # Preserve a parseable shape so callers don't NPE, but mark the
237
- # failure explicitly — blank NDJSON previously became {} → nil msg
238
- # → silent "[pwn-ai] engine returned no message".
239
246
  return {
240
247
  message: {
241
248
  role: 'assistant',
@@ -248,6 +255,7 @@ module PWN
248
255
  end
249
256
 
250
257
  openai_compat = rest_call.include?('/v1/') ||
258
+ rest_call.start_with?('v1/') ||
251
259
  payloads.any? { |p| p.key?(:choices) }
252
260
 
253
261
  if openai_compat
@@ -258,8 +266,6 @@ module PWN
258
266
  end
259
267
 
260
268
  # Native Ollama /api/chat NDJSON → single-object JSON string.
261
- # Each chunk carries message.content delta (and optionally tool_calls);
262
- # the final chunk has done:true plus timing/usage fields.
263
269
  private_class_method def self.assemble_native_chat_stream(opts = {})
264
270
  payloads = opts[:payloads]
265
271
  final = payloads.reverse.find { |p| p[:done] } || payloads.last
@@ -274,16 +280,10 @@ module PWN
274
280
  content << msg[:content].to_s if msg[:content]
275
281
  thinking << msg[:thinking].to_s if msg[:thinking]
276
282
  Array(msg[:tool_calls]).each do |tc|
277
- # Tool calls typically arrive complete in one chunk; append uniques.
278
283
  tool_calls << tc unless tool_calls.any? { |existing| existing == tc }
279
284
  end
280
285
  end
281
286
 
282
- # Thinking-only models (Qwen3 / DeepSeek-R1 style via Ollama) often
283
- # emit message.thinking deltas with message.content left blank. The
284
- # agent loop only displays msg[:content], so that looked like "no
285
- # response". Promote thinking → content when there is nothing else
286
- # visible and no tool_calls to act on.
287
287
  content = visible_from_thinking(thinking: thinking) if content.empty? && !thinking.empty? && tool_calls.empty?
288
288
 
289
289
  message = { role: role, content: content }
@@ -367,13 +367,92 @@ module PWN
367
367
  # response = PWN::AI::Ollama.get_models
368
368
 
369
369
  public_class_method def self.get_models
370
- models = ollama_rest_call(rest_call: 'ollama/api/tags')
370
+ models = ollama_rest_call(rest_call: 'api/tags')
371
371
 
372
372
  JSON.parse(models, symbolize_names: true)[:models]
373
373
  rescue StandardError => e
374
374
  raise e
375
375
  end
376
376
 
377
+ # Coerce OpenAI-wire message history into Ollama-native shapes before
378
+ # POST /api/chat (or Open WebUI /ollama/api/chat):
379
+ # - function.arguments must be a Hash/Array object, not a JSON string
380
+ # (string args → HTTP 400 "can't find closing '}' symbol")
381
+ # - assistant content nil + tool_calls → "" (Open WebUI form validation)
382
+ private_class_method def self.parse_tool_arguments_object(opts = {})
383
+ raw = opts[:arguments]
384
+ case raw
385
+ when Hash, Array then raw
386
+ when nil then {}
387
+ when String
388
+ s = raw.strip
389
+ return {} if s.empty?
390
+
391
+ begin
392
+ parsed = JSON.parse(s, symbolize_names: true)
393
+ return parsed if parsed.is_a?(Hash) || parsed.is_a?(Array)
394
+ rescue JSON::ParserError
395
+ nil
396
+ end
397
+ { value: s }
398
+ else
399
+ { value: raw.to_s }
400
+ end
401
+ end
402
+
403
+ private_class_method def self.normalize_messages_for_ollama(opts = {})
404
+ Array(opts[:messages]).filter_map do |m|
405
+ next unless m.is_a?(Hash)
406
+
407
+ role = (m[:role] || m['role']).to_s
408
+ out = { role: role }
409
+
410
+ tcs = m[:tool_calls] || m['tool_calls']
411
+ has_tcs = false
412
+ if tcs
413
+ wired = Array(tcs).filter_map do |tc|
414
+ next unless tc.is_a?(Hash)
415
+
416
+ fn = tc[:function] || tc['function'] || {}
417
+ name = fn[:name] || fn['name'] || tc[:name] || tc['name']
418
+ args = fn[:arguments] || fn['arguments'] || tc[:arguments] || tc['arguments']
419
+ {
420
+ id: (tc[:id] || tc['id'] || "call_#{SecureRandom.hex(4)}").to_s,
421
+ type: (tc[:type] || tc['type'] || 'function').to_s,
422
+ function: {
423
+ name: name.to_s,
424
+ arguments: parse_tool_arguments_object(arguments: args)
425
+ }
426
+ }
427
+ end
428
+ unless wired.empty?
429
+ out[:tool_calls] = wired
430
+ has_tcs = true
431
+ end
432
+ end
433
+
434
+ if m.key?(:content) || m.key?('content')
435
+ content = m.key?(:content) ? m[:content] : m['content']
436
+ out[:content] = case content
437
+ when nil then has_tcs ? '' : nil
438
+ when String then content
439
+ when Hash, Array then JSON.generate(content)
440
+ else content.to_s
441
+ end
442
+ elsif has_tcs
443
+ out[:content] = ''
444
+ end
445
+
446
+ name = m[:name] || m['name']
447
+ out[:name] = name.to_s if name && !name.to_s.empty?
448
+
449
+ tcid = m[:tool_call_id] || m['tool_call_id']
450
+ out[:tool_call_id] = tcid.to_s if tcid && !tcid.to_s.empty?
451
+
452
+ out
453
+ end
454
+ end
455
+
377
456
  # Supported Method Parameters::
378
457
  # response = PWN::AI::Ollama.chat_with_tools(
379
458
  # messages: 'required - full OpenAI-format messages array (system/user/assistant/tool)',
@@ -385,32 +464,13 @@ module PWN
385
464
  # spinner: 'optional - display spinner (default false)'
386
465
  # )
387
466
  #
388
- # Returns a Hash with :choices / :assistant_message intact (including
389
- # :message[:tool_calls]) — used by PWN::AI::Agent::Loop.
390
- #
391
- # LOCAL-MODEL SCAFFOLDING
392
- # -----------------------
393
- # This hits Ollama's NATIVE /api/chat (not the OpenAI-compat shim) so
394
- # the following actually take effect:
395
- # options.num_ctx - Ollama defaults to 2048; the pwn-ai system
396
- # prompt alone blows that. Defaults here to
397
- # PWN::Env[:ai][:ollama][:num_ctx] || 32768.
398
- # options.temperature - forced to 0.1 on tool-bearing turns for
399
- # deterministic tool selection; engine[:temp]
400
- # (creative) on the final text-only turn.
401
- # format - ONLY set when engine[:format] is explicitly
402
- # configured. Never default to 'json' when
403
- # tools: are present — that fights native
404
- # tool_calls and kills mid-loop tool use.
405
- # keep_alive: '30m' - avoids reload latency between iterations.
406
- # tool_calls come back with function.arguments as a Hash (not a JSON
407
- # string), which PWN::AI::Agent::Dispatch.parse_args handles.
408
- # Streaming is ON (stream: true): ollama_rest_call assembles NDJSON
409
- # chunks back into a single response so the return shape is unchanged.
467
+ # Hits Ollama NATIVE POST /api/chat so options.num_ctx / num_predict /
468
+ # keep_alive take effect. Streaming is ON; ollama_rest_call assembles
469
+ # NDJSON chunks back into a single response.
410
470
 
411
471
  public_class_method def self.chat_with_tools(opts = {})
412
472
  engine = PWN::Env[:ai][:ollama]
413
- messages = opts[:messages]
473
+ messages = normalize_messages_for_ollama(messages: opts[:messages])
414
474
  raise 'ERROR: messages array is required' if messages.nil? || messages.empty?
415
475
 
416
476
  model = opts[:model] ||= engine[:model]
@@ -422,10 +482,6 @@ module PWN
422
482
  tools_present = opts[:tools] && !opts[:tools].empty?
423
483
  tool_temp = (engine[:tool_temp] || 0.1).to_f
424
484
  num_ctx = (engine[:num_ctx] || 32_768).to_i
425
- # Hard cap generation length. Thinking models (Qwen3 / R1-style)
426
- # otherwise stream unbounded message.thinking tokens until the
427
- # idle read_timeout (default 900s) — which looks like a 15–25 min
428
- # "stuck spinner" while bytes keep arriving on the socket.
429
485
  num_predict = (engine[:num_predict] || 4_096).to_i
430
486
  keep_alive = engine[:keep_alive] || '30m'
431
487
 
@@ -442,10 +498,6 @@ module PWN
442
498
  }
443
499
  if tools_present
444
500
  http_body[:tools] = opts[:tools]
445
- # 0.2 — omit format when tools are present unless the operator
446
- # explicitly set PWN::Env[:ai][:ollama][:format]. Forcing 'json'
447
- # races the native tool_calls sampler and produces "can't call tools"
448
- # mid-loop on many local models.
449
501
  fmt = engine[:format]
450
502
  http_body[:format] = fmt unless fmt.nil? || fmt.to_s.empty?
451
503
  end
@@ -453,7 +505,7 @@ module PWN
453
505
 
454
506
  response = ollama_rest_call(
455
507
  http_method: :post,
456
- rest_call: 'ollama/api/chat',
508
+ rest_call: 'api/chat',
457
509
  http_body: http_body,
458
510
  timeout: opts[:timeout],
459
511
  spinner: opts[:spinner]
@@ -461,7 +513,6 @@ module PWN
461
513
  raise 'ERROR: Ollama chat_with_tools received empty response from ollama_rest_call' if response.nil? || (response.respond_to?(:empty?) && response.empty?)
462
514
 
463
515
  json_resp = JSON.parse(response, symbolize_names: true)
464
- # Normalise native /api/chat shape to what Loop.normalize_llm expects.
465
516
  msg = json_resp[:message] || json_resp.dig(:choices, 0, :message)
466
517
  if msg.is_a?(Hash)
467
518
  content = msg[:content].to_s
@@ -482,7 +533,7 @@ module PWN
482
533
  # response = PWN::AI::Ollama.chat(
483
534
  # request: 'required - message to Ollama'
484
535
  # model: 'optional - model to use for text generation (defaults to PWN::Env[:ai][:ollama][:model])',
485
- # temp: 'optional - creative response float (deafults to PWN::Env[:ai][:ollama][:temp])',
536
+ # temp: 'optional - creative response float (defaults to PWN::Env[:ai][:ollama][:temp])',
486
537
  # system_role_content: 'optional - context to set up the model behavior for conversation (Default: PWN::Env[:ai][:ollama][:system_role_content])',
487
538
  # response_history: 'optional - pass response back in to have a conversation',
488
539
  # speak_answer: 'optional speak answer using PWN::Plugins::Voice.text_to_speech (Default: nil)',
@@ -503,12 +554,11 @@ module PWN
503
554
  temp = opts[:temp].to_f ||= engine[:temp].to_f
504
555
  temp = 1 if temp.zero?
505
556
 
506
- rest_call = 'ollama/v1/chat/completions'
557
+ # OpenAI-compat shim on the ollama server (no api-key needed).
558
+ rest_call = 'v1/chat/completions'
507
559
 
508
560
  response_history = opts[:response_history]
509
561
 
510
- max_tokens = response_history[:usage][:total_tokens] unless response_history.nil?
511
-
512
562
  system_role_content = opts[:system_role_content] ||= engine[:system_role_content]
513
563
 
514
564
  system_role = {
@@ -522,7 +572,6 @@ module PWN
522
572
  }
523
573
 
524
574
  response_history ||= { choices: [system_role] }
525
- choices_len = response_history[:choices].length
526
575
 
527
576
  http_body = {
528
577
  model: model,
@@ -560,8 +609,6 @@ module PWN
560
609
  if speak_answer
561
610
  answer = assistant_resp[:content]
562
611
  text_path = "/tmp/#{SecureRandom.hex}.pwn_voice"
563
- # answer = json_resp[:choices].last[:text]
564
- # answer = json_resp[:choices].last[:content] if gpt
565
612
  File.write(text_path, answer)
566
613
  PWN::Plugins::Voice.text_to_speech(text_path: text_path)
567
614
  File.unlink(text_path)
@@ -597,6 +644,12 @@ module PWN
597
644
  spinner: 'optional - display spinner (defaults to false)'
598
645
  )
599
646
 
647
+ response = #{self}.chat_with_tools(
648
+ messages: 'required - messages array',
649
+ tools: 'optional - OpenAI tools array',
650
+ model: 'optional - overrides PWN::Env[:ai][:ollama][:model]'
651
+ )
652
+
600
653
  #{self}.authors
601
654
  "
602
655
  end
@@ -555,6 +555,9 @@ module PWN
555
555
  messages = opts[:messages]
556
556
  raise 'ERROR: messages array is required' if messages.nil? || messages.empty?
557
557
 
558
+ # OpenAI rejects Hash function.arguments / Hash content (422 map → string).
559
+ messages = PWN::AI::Agent::Loop.openai_wire_messages(messages: messages) if defined?(PWN::AI::Agent::Loop) && PWN::AI::Agent::Loop.respond_to?(:openai_wire_messages)
560
+
558
561
  model = opts[:model] ||= engine[:model]
559
562
 
560
563
  reasoning = reasoning_model?(model: model)