pwn 0.5.705 → 0.5.707

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. checksums.yaml +4 -4
  2. data/Gemfile +1 -1
  3. data/documentation/Reporting.md +1 -0
  4. data/etc/default_skills/pwn/reports/SKILL.md +4 -2
  5. data/etc/default_skills/pwn/reports/csv/SKILL.md +47 -0
  6. data/etc/default_skills/pwn/reports/html/SKILL.md +47 -0
  7. data/etc/default_skills/pwn/reports/json/SKILL.md +47 -0
  8. data/etc/default_skills/pwn/reports/markdown/SKILL.md +47 -0
  9. data/etc/default_skills/pwn/reports/pdf/SKILL.md +47 -0
  10. data/etc/default_skills/pwn/reports/xml/SKILL.md +47 -0
  11. data/lib/pwn/ai/agent/curriculum.rb +22 -20
  12. data/lib/pwn/ai/agent/learning.rb +3 -10
  13. data/lib/pwn/ai/agent/loop.rb +243 -52
  14. data/lib/pwn/ai/agent/policy.rb +23 -5
  15. data/lib/pwn/ai/agent/reward.rb +11 -23
  16. data/lib/pwn/ai/agent/turn_finalizer.rb +0 -1
  17. data/lib/pwn/reports/ai_red_team.rb +1 -1
  18. data/lib/pwn/reports/csv.rb +38 -0
  19. data/lib/pwn/reports/fuzz.rb +1 -1
  20. data/lib/pwn/reports/html.rb +58 -0
  21. data/lib/pwn/reports/json.rb +32 -0
  22. data/lib/pwn/reports/markdown.rb +40 -0
  23. data/lib/pwn/reports/pdf.rb +93 -0
  24. data/lib/pwn/reports/phone.rb +1 -1
  25. data/lib/pwn/reports/sast.rb +1 -1
  26. data/lib/pwn/reports/uri_buster.rb +1 -1
  27. data/lib/pwn/reports/xml.rb +44 -0
  28. data/lib/pwn/reports.rb +54 -6
  29. data/lib/pwn/version.rb +1 -1
  30. data/spec/integration/reinforced_feedback_loop_spec.rb +19 -4
  31. data/spec/lib/pwn/ai/agent/loop_spec.rb +166 -10
  32. data/spec/lib/pwn/ai/agent/policy_spec.rb +12 -2
  33. data/spec/lib/pwn/ai/agent/reward_spec.rb +72 -0
  34. data/spec/lib/pwn/reports/csv_spec.rb +19 -0
  35. data/spec/lib/pwn/reports/formats_spec.rb +90 -0
  36. data/spec/lib/pwn/reports/html_spec.rb +19 -0
  37. data/spec/lib/pwn/reports/json_spec.rb +19 -0
  38. data/spec/lib/pwn/reports/markdown_spec.rb +19 -0
  39. data/spec/lib/pwn/reports/pdf_spec.rb +19 -0
  40. data/spec/lib/pwn/reports/xml_spec.rb +19 -0
  41. data/third_party/pwn_rdoc.jsonl +46 -1
  42. metadata +22 -3
@@ -3,6 +3,8 @@
3
3
  require 'json'
4
4
  require 'securerandom'
5
5
  require 'digest'
6
+ require 'fileutils'
7
+ require 'tmpdir'
6
8
  require 'pwn/ai/agent/mistakes'
7
9
 
8
10
  module PWN
@@ -324,7 +326,12 @@ module PWN
324
326
 
325
327
  top = Mistakes.top(limit: 24, unresolved_only: true)
326
328
  now = Time.now
327
- budget = top.select { |mistake| budget_hit?(mistake: mistake) && !mistake[:parked] }
329
+ budget = top.select do |mistake|
330
+ next false if mistake[:parked]
331
+ next false if %w[agent_loop assistant_answer].include?(mistake[:tool].to_s)
332
+
333
+ budget_hit?(mistake: mistake)
334
+ end
328
335
  budget.each do |mistake|
329
336
  stamp = mistake_ts(mistake: mistake)
330
337
  next if stamp && (now - stamp) <= PARK_COOL_SECS
@@ -584,6 +591,10 @@ module PWN
584
591
  live = effects.reject { |fx| %i[recall store].include?(fx) }
585
592
  return true if live.empty?
586
593
  return true if duration_unsatisfied?(request: request)
594
+ return true if declared_contract_unsatisfied?(
595
+ request: request,
596
+ messages: opts[:messages]
597
+ )
587
598
  return true if need == :write && !write_verified?(effects: effects)
588
599
  return true if need == :browse && !effects.include?(:browse)
589
600
  return true if need == :any && !effects.intersect?(%i[write browse eval])
@@ -613,23 +624,166 @@ module PWN
613
624
  }.freeze
614
625
 
615
626
  private_class_method def self.duration_unsatisfied?(opts = {})
616
- req = opts[:request].to_s
617
- hours = nil
618
- if (m = req.match(/\b(\d+)\s*(?:hours?|hrs?)\b/i))
619
- hours = m[1].to_i
620
- elsif (m = req.match(/\b(twenty-four|one|two|three|four|five|six|seven|eight|nine|ten|eleven|twelve|thirteen|fourteen|fifteen|sixteen|seventeen|eighteen|nineteen|twenty)\s*(?:hours?|hrs?)\b/i))
621
- hours = HOUR_WORDS[m[1].downcase]
627
+ secs = declared_min_seconds(request: opts[:request])
628
+ if secs <= 0
629
+ req = opts[:request].to_s
630
+ hours = nil
631
+ if (m = req.match(/\b(\d+)\s*(?:hours?|hrs?)\b/i))
632
+ hours = m[1].to_i
633
+ elsif (m = req.match(/\b(twenty-four|one|two|three|four|five|six|seven|eight|nine|ten|eleven|twelve|thirteen|fourteen|fifteen|sixteen|seventeen|eighteen|nineteen|twenty)\s*(?:hours?|hrs?)\b/i))
634
+ hours = HOUR_WORDS[m[1].downcase]
635
+ end
636
+ secs = hours.to_i * 3600
622
637
  end
623
- return false if hours.to_i <= 0
638
+ return false if secs <= 0
624
639
 
625
640
  t0 = Thread.current[:pwn_loop_t0]
626
641
  return false unless t0
627
642
 
628
- (Time.now - t0) < (hours * 3600)
643
+ (Time.now - t0) < secs
644
+ rescue StandardError
645
+ false
646
+ end
647
+
648
+ EMPTY_CONTRACT = {
649
+ paths: [],
650
+ min_seconds: 0,
651
+ skills: [],
652
+ proofs: [],
653
+ hosts: [],
654
+ techniques: [],
655
+ issue_work: false
656
+ }.freeze
657
+
658
+ private_class_method def self.declared_min_seconds(opts = {})
659
+ declared_contract(request: opts[:request])[:min_seconds].to_i
660
+ end
661
+
662
+ private_class_method def self.declared_contract_unsatisfied?(opts = {})
663
+ contract = declared_contract(request: opts[:request])
664
+ files = Array(contract[:paths]) + Array(contract[:proofs])
665
+ return true if files.any? { |path| deliverable_missing?(path: path) }
666
+ return true if declared_skills_missing?(skills: contract[:skills])
667
+ return true if declared_hosts_missing?(hosts: contract[:hosts], messages: opts[:messages])
668
+ return true if declared_hosts_missing?(hosts: contract[:techniques], messages: opts[:messages])
669
+ return true if contract[:issue_work] && Array(contract[:proofs]).empty?
670
+
671
+ false
629
672
  rescue StandardError
630
673
  false
631
674
  end
632
675
 
676
+ private_class_method def self.declared_skills_missing?(opts = {})
677
+ names = Array(opts[:skills]).map(&:to_s).reject(&:empty?)
678
+ return false if names.empty?
679
+ return true unless defined?(PWN::Skills) && PWN::Skills.is_a?(Hash)
680
+
681
+ have = PWN::Skills.keys.map(&:to_s)
682
+ names.any? { |name| !have.include?(name) }
683
+ end
684
+
685
+ private_class_method def self.declared_hosts_missing?(opts = {})
686
+ hosts = Array(opts[:hosts]).map(&:to_s).reject(&:empty?)
687
+ return false if hosts.empty?
688
+
689
+ blob = Array(opts[:messages]).select { |msg| msg.is_a?(Hash) && msg[:role].to_s == 'tool' }
690
+ .map { |msg| msg[:content].to_s }
691
+ .join("\n")
692
+ .downcase
693
+ hosts.any? { |host| !blob.include?(host.to_s.downcase) }
694
+ end
695
+
696
+ private_class_method def self.declared_deliverables(opts = {})
697
+ Array(declared_contract(request: opts[:request])[:paths])
698
+ end
699
+
700
+ private_class_method def self.declared_contract(opts = {})
701
+ cached = Thread.current[:pwn_loop_deliverables]
702
+ return normalize_contract(raw: cached) if cached.is_a?(Array) || cached.is_a?(Hash)
703
+ return EMPTY_CONTRACT.dup unless Thread.current[:pwn_loop_active]
704
+
705
+ contract = infer_deliverables(request: opts[:request])
706
+ Thread.current[:pwn_loop_deliverables] = contract
707
+ contract
708
+ end
709
+
710
+ private_class_method def self.normalize_contract(opts = {})
711
+ raw = opts[:raw]
712
+ return EMPTY_CONTRACT.merge(paths: raw.map(&:to_s).select { |p| p.start_with?('/') }) if raw.is_a?(Array)
713
+ return EMPTY_CONTRACT.dup unless raw.is_a?(Hash)
714
+
715
+ hours = raw[:hours] || raw['hours']
716
+ secs = (raw[:min_seconds] || raw['min_seconds']).to_i
717
+ secs = hours.to_i * 3600 if secs <= 0 && hours.to_i.positive?
718
+ {
719
+ paths: abs_paths(rows: raw[:paths] || raw['paths']),
720
+ min_seconds: secs,
721
+ skills: Array(raw[:skills] || raw['skills']).map(&:to_s).reject(&:empty?).uniq,
722
+ proofs: abs_paths(rows: raw[:proofs] || raw['proofs']),
723
+ hosts: Array(raw[:hosts] || raw['hosts']).map(&:to_s).reject(&:empty?).uniq,
724
+ techniques: Array(raw[:techniques] || raw['techniques']).map(&:to_s).reject(&:empty?).uniq,
725
+ issue_work: raw[:issue_work] == true || raw['issue_work'] == true
726
+ }
727
+ end
728
+
729
+ private_class_method def self.abs_paths(opts = {})
730
+ Array(opts[:rows]).map(&:to_s).select { |path| path.start_with?('/') }.uniq
731
+ end
732
+
733
+ private_class_method def self.infer_deliverables(opts = {})
734
+ request = opts[:request].to_s
735
+ return EMPTY_CONTRACT.dup if request.strip.empty?
736
+
737
+ reply = call_engine(
738
+ messages: [
739
+ {
740
+ role: 'system',
741
+ content: 'Reply with JSON only. No markdown. No tools.'
742
+ },
743
+ {
744
+ role: 'user',
745
+ content: "Operator request:\n#{request}\n\n" \
746
+ 'When that request is complete, what must be true on this host? ' \
747
+ 'JSON only: {"paths":["/abs/file"],"min_seconds":0,"skills":["name"],' \
748
+ '"proofs":["/abs/poc"],"hosts":["ip-or-hostname"],' \
749
+ '"techniques":["T1059"],"issue_work":false}. ' \
750
+ 'issue_work=true when the ask needs findings, PoCs, or severity — then proofs must be non-empty absolute paths. ' \
751
+ 'Write reports with PWN::Reports::PDF.generate / HTML / Markdown / XML / CSV / JSON (path: or dir_path: + report_name:). ' \
752
+ 'Use [] or 0 when a field is not required. Do not invent work. Paths must be absolute.'
753
+ }
754
+ ],
755
+ tools: nil
756
+ )
757
+ text = reply.is_a?(Hash) ? (reply[:content] || reply['content']).to_s : reply.to_s
758
+ parse_contract(text: text)
759
+ rescue StandardError
760
+ EMPTY_CONTRACT.dup
761
+ end
762
+
763
+ private_class_method def self.parse_contract(opts = {})
764
+ text = opts[:text].to_s
765
+ json = nil
766
+ begin
767
+ json = JSON.parse(text, symbolize_names: true)
768
+ rescue JSON::ParserError
769
+ start = text.index('{')
770
+ stop = text.rindex('}')
771
+ return EMPTY_CONTRACT.dup unless start && stop && stop > start
772
+
773
+ json = JSON.parse(text[start..stop], symbolize_names: true)
774
+ end
775
+ normalize_contract(raw: json)
776
+ rescue StandardError
777
+ EMPTY_CONTRACT.dup
778
+ end
779
+
780
+ private_class_method def self.deliverable_missing?(opts = {})
781
+ path = opts[:path].to_s
782
+ path.empty? || !File.file?(path) || File.size(path) <= 0
783
+ rescue StandardError
784
+ true
785
+ end
786
+
633
787
  private_class_method def self.may_finalize?(opts = {})
634
788
  return false if incomplete_final?(text: opts[:text], last_iter: false)
635
789
  return false if request_unsatisfied?(
@@ -962,12 +1116,17 @@ module PWN
962
1116
  end
963
1117
 
964
1118
  private_class_method def self.no_progress_result(opts = {})
1119
+ checkpoint_result(opts)
1120
+ end
1121
+
1122
+ private_class_method def self.checkpoint_result(opts = {})
965
1123
  name = opts[:name].to_s
966
1124
  sig = payload_sig(opts)
967
1125
  JSON.generate(
968
- success: false,
969
- error: "no_progress: identical #{name} payload repeated (#{sig}). Change the command.",
970
- result: { stdout: '', stderr: "no_progress: #{name}", exit: 2 }
1126
+ success: true,
1127
+ checkpoint: true,
1128
+ error: "checkpoint: identical #{name} payload (#{sig}). World unchanged — vary args, target, or tool.",
1129
+ result: { stdout: "checkpoint #{name} #{sig}", stderr: '', exit: 0 }
971
1130
  )
972
1131
  end
973
1132
 
@@ -1029,15 +1188,10 @@ module PWN
1029
1188
  return nil unless plan_msg && !plan_msg[:content].to_s.strip.empty?
1030
1189
 
1031
1190
  plan = plan_msg[:content].to_s.strip
1032
- messages << { role: 'assistant', content: "PLAN:\n#{plan}" }
1033
- # S4 — adversarial plan review grounded in THIS host's telemetry.
1034
- # P17 — never fork red_team when budget fingerprints dominate: it is
1035
- # another mini agent loop and compounds iteration-budget exhaustion.
1191
+ # TUI-only. Do not put PLAN: or red-team text on the model wire —
1192
+ # original request stays the only user goal.
1036
1193
  rt = nil
1037
- if defined?(Curriculum) && !hot
1038
- rt = Curriculum.red_team_plan(request: opts[:request], plan: plan)
1039
- messages << { role: 'user', content: rt } if rt
1040
- end
1194
+ rt = Curriculum.red_team_plan(request: opts[:request], plan: plan) if defined?(Curriculum) && !hot
1041
1195
  # P2 — unify TaskSummarizer plan object with surviving outline so the
1042
1196
  # task line and adversarial/plan_first plan are one thing. Index-only;
1043
1197
  # credit stays in Reward. Optional: only when ts_state is live.
@@ -1541,11 +1695,7 @@ module PWN
1541
1695
  collapsed << pair
1542
1696
  end
1543
1697
  kept = collapsed.last(keep_pairs).flatten
1544
- kept.each do |m|
1545
- next unless m[:role].to_s == 'tool' && m[:content].to_s.length > max_chars
1546
-
1547
- m[:content] = "#{m[:content].to_s[0, max_chars]}…[compacted]"
1548
- end
1698
+ spill_tool_history!(messages: kept, max_chars: max_chars)
1549
1699
  messages.replace(head + kept)
1550
1700
  repair_tool_history!(messages: messages)
1551
1701
  messages
@@ -1554,6 +1704,43 @@ module PWN
1554
1704
  opts[:messages]
1555
1705
  end
1556
1706
 
1707
+ HISTORY_SPILL_DIR = File.join(Dir.tmpdir, 'pwn-ai-hist')
1708
+ KEEP_FULL_TOOL_TAILS = 2
1709
+
1710
+ private_class_method def self.spill_tool_history!(opts = {})
1711
+ messages = Array(opts[:messages])
1712
+ max_chars = opts[:max_chars].to_i
1713
+ max_chars = 2_000 if max_chars <= 0
1714
+ tools = messages.select { |msg| msg[:role].to_s == 'tool' }
1715
+ spill_n = [tools.length - KEEP_FULL_TOOL_TAILS, 0].max
1716
+ idx = 0
1717
+ messages.each do |msg|
1718
+ next unless msg[:role].to_s == 'tool'
1719
+
1720
+ idx += 1
1721
+ next if idx > spill_n
1722
+
1723
+ body = msg[:content].to_s
1724
+ next if body.length <= max_chars
1725
+ next if body.include?('[compacted path=')
1726
+
1727
+ msg[:content] = spill_tool_body(text: body)
1728
+ end
1729
+ messages
1730
+ end
1731
+
1732
+ private_class_method def self.spill_tool_body(opts = {})
1733
+ text = opts[:text].to_s
1734
+ digest = Digest::SHA256.hexdigest(text)[0, 16]
1735
+ dir = HISTORY_SPILL_DIR
1736
+ FileUtils.mkdir_p(dir)
1737
+ path = File.join(dir, "#{digest}.txt")
1738
+ File.binwrite(path, text) unless File.file?(path)
1739
+ "[compacted path=#{path} sha256=#{digest} bytes=#{text.bytesize}]"
1740
+ rescue StandardError
1741
+ "[compacted bytes=#{opts[:text].to_s.bytesize}]"
1742
+ end
1743
+
1557
1744
  private_class_method def self.repair_tool_history!(opts = {})
1558
1745
  messages = opts[:messages]
1559
1746
  return messages unless messages.is_a?(Array)
@@ -1913,9 +2100,7 @@ module PWN
1913
2100
  session_id: session_id,
1914
2101
  request: request,
1915
2102
  final: txt,
1916
- predicted: 0.85,
1917
- plan: [],
1918
- ts_state: nil
2103
+ predicted: 0.85
1919
2104
  )
1920
2105
  end
1921
2106
  txt
@@ -1984,9 +2169,7 @@ module PWN
1984
2169
  session_id: session_id,
1985
2170
  request: request,
1986
2171
  final: txt,
1987
- predicted: 0.85,
1988
- plan: [],
1989
- ts_state: nil
2172
+ predicted: 0.85
1990
2173
  )
1991
2174
  end
1992
2175
  txt
@@ -2010,9 +2193,7 @@ module PWN
2010
2193
  session_id: session_id,
2011
2194
  request: request,
2012
2195
  final: txt,
2013
- predicted: 0.95,
2014
- plan: ['Acknowledge greeting without tools or weather echo'],
2015
- ts_state: nil
2196
+ predicted: 0.95
2016
2197
  )
2017
2198
  end
2018
2199
  txt
@@ -2094,9 +2275,7 @@ module PWN
2094
2275
  session_id: session_id,
2095
2276
  request: request,
2096
2277
  final: txt,
2097
- predicted: 0.9,
2098
- plan: ['Explain tool usage without live recon'],
2099
- ts_state: nil
2278
+ predicted: 0.9
2100
2279
  )
2101
2280
  end
2102
2281
  txt
@@ -2306,9 +2485,7 @@ module PWN
2306
2485
  session_id: session_id,
2307
2486
  request: request,
2308
2487
  final: txt,
2309
- predicted: 0.95,
2310
- plan: [plan_label || 'Recall prior turn from session transcript'],
2311
- ts_state: nil
2488
+ predicted: 0.95
2312
2489
  )
2313
2490
  end
2314
2491
  return txt
@@ -2379,9 +2556,7 @@ module PWN
2379
2556
  session_id: session_id,
2380
2557
  request: request,
2381
2558
  final: txt,
2382
- predicted: 0.85,
2383
- plan: ['Recall prior turn (empty session fallback)'],
2384
- ts_state: nil
2559
+ predicted: 0.85
2385
2560
  )
2386
2561
  end
2387
2562
  txt
@@ -2434,6 +2609,10 @@ module PWN
2434
2609
  Thread.current[:pwn_extinguished] = {}
2435
2610
  Thread.current[:pwn_same_payload] = Hash.new(0)
2436
2611
  Thread.current[:pwn_loop_t0] = Time.now unless nested
2612
+ unless nested
2613
+ Thread.current[:pwn_loop_active] = true
2614
+ Thread.current[:pwn_loop_deliverables] = nil
2615
+ end
2437
2616
  debug_progress(msg: "intent=#{intent} engine=#{engine}", debug: opts[:debug])
2438
2617
  expose_current_session(session_id: session_id)
2439
2618
  Mistakes.check_user_correction(request: request, session_id: session_id) if defined?(Mistakes)
@@ -2520,6 +2699,10 @@ module PWN
2520
2699
  # CORE_TOOLS is the default action space. Extra schemas are
2521
2700
  # opt-in via enabled_toolsets + core_only: false.
2522
2701
  core_only = opts.fetch(:core_only, true)
2702
+ if nested && needs_host_work?(request: request)
2703
+ opts[:enabled_toolsets] = nil
2704
+ core_only = true
2705
+ end
2523
2706
  tools = Registry.definitions(
2524
2707
  enabled: opts[:enabled_toolsets],
2525
2708
  relevance: request,
@@ -2672,12 +2855,20 @@ module PWN
2672
2855
  # commit that as the answer; drop the empty assistant turn,
2673
2856
  # inject a one-shot nudge, and keep iterating.
2674
2857
  if calls.empty? && text.strip.empty?
2858
+ unsat = request_unsatisfied?(request: request, messages: messages)
2675
2859
  warn "[pwn-ai/loop] empty final from #{engine} on iter=#{i}; nudging" if local
2860
+ empty_nudge = if unsat
2861
+ 'Your previous reply was empty (no tool_calls and no content). ' \
2862
+ 'The original request is not evidenced yet. Emit NATIVE tool_calls NOW. ' \
2863
+ 'Do not write a final answer until that request is done or a tool returned failure evidence.'
2864
+ else
2865
+ 'Your previous reply was empty (no tool_calls and no content). ' \
2866
+ 'Either call a tool now, or write the final answer for the user as plain text. ' \
2867
+ 'Do not reply with an empty message.'
2868
+ end
2676
2869
  messages << {
2677
2870
  role: 'user',
2678
- content: 'Your previous reply was empty (no tool_calls and no content). ' \
2679
- 'Either call a tool now, or write the final answer for the user as plain text. ' \
2680
- 'Do not reply with an empty message.'
2871
+ content: empty_nudge
2681
2872
  }
2682
2873
  turn_fails['empty_final'] += 1
2683
2874
  debug_progress(msg: "bounce empty_final snippet=#{debug_snippet(text: text)}")
@@ -2720,7 +2911,7 @@ module PWN
2720
2911
  debug_final_text!(text: text)
2721
2912
  final_chars = text.to_s.length
2722
2913
  append_session(session_id: session_id, role: 'assistant', content: text)
2723
- Learning.auto_introspect(session_id: session_id, request: request, final: text, predicted: predicted, plan: ts_state && ts_state[:plan], ts_state: ts_state) if defined?(Learning) && !nested && !no_tools && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: i)
2914
+ Learning.auto_introspect(session_id: session_id, request: request, final: text, predicted: predicted, ts_state: ts_state) if defined?(Learning) && !nested && !no_tools && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: i)
2724
2915
  maybe_finish_policy(session_id: session_id, proxy_ok: true, ts_state: ts_state)
2725
2916
  task_summary_flush!(state: ts_state, on_tool: on_tool)
2726
2917
  OpenGoal.clear! if defined?(OpenGoal) && !nested
@@ -2754,11 +2945,7 @@ module PWN
2754
2945
  else
2755
2946
  raw = Dispatch.call(tool_call: tc)
2756
2947
  same_n = note_same_payload!(name: name, args: args)
2757
- if same_n >= 3
2758
- Thread.current[:pwn_extinguished] ||= {}
2759
- Thread.current[:pwn_extinguished][sig] = true
2760
- raw = no_progress_result(name: name, args: args)
2761
- end
2948
+ raw = checkpoint_result(name: name, args: args) if same_n >= 3
2762
2949
  end
2763
2950
  tools_called += 1
2764
2951
  tele = record_metrics(name: name, started: started, raw: raw, args: args, session_id: session_id, engine: engine, ts_state: ts_state)
@@ -2830,6 +3017,10 @@ module PWN
2830
3017
  end
2831
3018
  raise
2832
3019
  ensure
3020
+ unless nested
3021
+ Thread.current[:pwn_loop_active] = nil
3022
+ Thread.current[:pwn_loop_deliverables] = nil
3023
+ end
2833
3024
  Thread.current[:pwn_loop_no_tools] = nil
2834
3025
  finish_debug_request!(
2835
3026
  iter: i,
@@ -752,16 +752,34 @@ module PWN
752
752
  return 0.0 unless ep.is_a?(Hash)
753
753
 
754
754
  bonus = 0.0
755
- new_idx = ts_idx(ts_state: opts[:ts_state])
756
- new_open = ts_open?(ts_state: opts[:ts_state])
757
- bonus += STEP_TASK if !ep[:plan_idx].nil? && new_idx > ep[:plan_idx].to_i
758
- bonus += STEP_CLOSED if ep[:plan_open] && new_open == false
759
- bonus += STEP_GRIND if new_open == false && opts[:ok] && opts[:action].to_s != 'final'
755
+ met = contract_fields_met(request: ep[:request].to_s)
756
+ prev = ep[:contract_met].to_i
757
+ if met > prev
758
+ bonus += STEP_TASK
759
+ ep[:contract_met] = met
760
+ end
761
+ bonus += STEP_GRIND if met.positive? && opts[:ok] && opts[:action].to_s != 'final'
760
762
  bonus
761
763
  rescue StandardError
762
764
  0.0
763
765
  end
764
766
 
767
+ private_class_method def self.contract_fields_met(opts = {})
768
+ raw = Thread.current[:pwn_loop_deliverables]
769
+ return 0 unless raw.is_a?(Hash) || opts.key?(:request)
770
+ return 0 unless raw.is_a?(Hash)
771
+
772
+ n = 0
773
+ n += 1 if raw[:min_seconds].to_i.positive? || raw['min_seconds'].to_i.positive?
774
+ files = Array(raw[:paths] || raw['paths']) + Array(raw[:proofs] || raw['proofs'])
775
+ n += files.count { |path| File.file?(path.to_s) && File.size(path.to_s).positive? }
776
+ n += Array(raw[:hosts] || raw['hosts']).length
777
+ n += Array(raw[:techniques] || raw['techniques']).length
778
+ n
779
+ rescue StandardError
780
+ 0
781
+ end
782
+
765
783
  private_class_method def self.terminal_reward(opts = {})
766
784
  return ((2.0 * opts[:score].to_f) - 1.0).clamp(-1.0, 1.0) unless opts[:score].nil?
767
785
 
@@ -81,15 +81,16 @@ module PWN
81
81
 
82
82
  JUDGE_SYSTEM = <<~SYS
83
83
  You are the pwn-ai Outcome Reward Model. Given a USER REQUEST, the
84
- agent's FINAL ANSWER, a compressed TOOL TRACE, and optional PLAN
85
- COVERAGE, emit ONE line of strict JSON:
84
+ agent's FINAL ANSWER, and a compressed TOOL TRACE, emit ONE line of
85
+ strict JSON:
86
86
  {"score": <0.0-1.0>, "verdict": "solved|partial|wrong|refused",
87
87
  "rationale": "<≤140 chars>", "key_step": <int|-1>}
88
- Grade the HUMAN RESULT, not handler success:
88
+ Grade the HUMAN RESULT against the USER REQUEST only — never a TUI
89
+ plan, stub outline, or competing compass:
89
90
  1.0 = final is usable and complete (every asked point answered with
90
91
  evidence from the trace or a checkable claim).
91
92
  0.7 = mostly complete, one missing detail, still usable.
92
- 0.5 = correct direction but incomplete / truncated / plan open.
93
+ 0.5 = correct direction but incomplete / truncated.
93
94
  0.2 = tools ran but the final does not answer the ask.
94
95
  0.0 = hallucinated, off-goal, empty, polite non-answer, or refused.
95
96
  Ignore {"success":true} as evidence of done. Prefer last tool steps.
@@ -124,8 +125,8 @@ module PWN
124
125
  trace = load_trace(session_id: opts[:session_id]) if trace.empty? && opts[:session_id]
125
126
  commit = opts.key?(:commit) ? opts[:commit] : true
126
127
 
127
- v = llm_judge(request: request, final: final, trace: trace, plan: opts[:plan])
128
- v ||= heuristic_judge(request: request, final: final, trace: trace, plan: opts[:plan])
128
+ v = llm_judge(request: request, final: final, trace: trace)
129
+ v ||= heuristic_judge(request: request, final: final, trace: trace)
129
130
  # Cheap ORM is the intended source. Heuristic overlap is fallback
130
131
  # only — callers (sentinel / Learning.stats / Metrics.effective_rate)
131
132
  # weight :llm_orm samples above :heuristic so the haircut tracks
@@ -1075,17 +1076,7 @@ module PWN
1075
1076
  # human got a usable result. Keep the first step for context.
1076
1077
  shown = compact_trace_tail(steps: steps, keep: CHEAP_ORM_TRACE_N)
1077
1078
  trace = shown.each_with_index.map { |s, i| "#{i + 1}. #{s.to_s.gsub(/\s+/, ' ')[0, 220]}" }.join("\n")
1078
- plan = ''
1079
- if respond_to?(:plan_coverage)
1080
- cov = plan_coverage(
1081
- plan: opts[:plan],
1082
- final: opts[:final],
1083
- request: opts[:request],
1084
- trace: steps
1085
- )
1086
- plan = "\nPLAN COVERAGE: #{cov[:covered]}/#{cov[:total]} (#{cov[:tag]}) missing=#{Array(cov[:missing]).first(3).join(' | ')}" if cov && cov[:total].to_i.positive?
1087
- end
1088
- req = "USER REQUEST:\n#{opts[:request].to_s[0, 700]}\n\nFINAL ANSWER:\n#{opts[:final].to_s[0, 1_600]}\n\nTOOL TRACE (#{steps.length} steps, showing #{shown.length}):\n#{trace}#{plan}"
1079
+ req = "USER REQUEST:\n#{opts[:request].to_s[0, 700]}\n\nFINAL ANSWER:\n#{opts[:final].to_s[0, 1_600]}\n\nTOOL TRACE (#{steps.length} steps, showing #{shown.length}):\n#{trace}"
1089
1080
  resp = cheap_orm_chat(request: req, system_role_content: JUDGE_SYSTEM)
1090
1081
  parsed = parse_llm_judge(resp: resp)
1091
1082
  return nil if parsed.nil?
@@ -1093,7 +1084,7 @@ module PWN
1093
1084
  # Soft-blend a stronger-than-overlap evidence prior so a noisy
1094
1085
  # cheap ORM cannot peg 0.0/1.0 against an obviously incomplete
1095
1086
  # or obviously complete final.
1096
- ev = evidence_prior(request: opts[:request], final: opts[:final], trace: steps, plan: opts[:plan])
1087
+ ev = evidence_prior(request: opts[:request], final: opts[:final], trace: steps)
1097
1088
  if ev && ev[:confidence].to_f >= 0.5
1098
1089
  raw = parsed[:score].to_f
1099
1090
  # Sanity bounds only: do not always blend (would fight a good ORM).
@@ -1317,11 +1308,10 @@ module PWN
1317
1308
  polite = final.match?(/\A\s*(sure|happy to help|of course|i can help|how can i|let me know)\b/i) && final.length < 120
1318
1309
  return { score: 0.1, verdict: :partial, rationale: 'polite non-answer', key_step: -1, source: :heuristic } if polite && trace.empty?
1319
1310
 
1320
- ev = evidence_prior(request: request, final: final, trace: trace, plan: opts[:plan])
1311
+ ev = evidence_prior(request: request, final: final, trace: trace)
1321
1312
  score = ev ? ev[:score].to_f : 0.35
1322
1313
  # Overlap is a small on-topic gate, not the score. The evidence
1323
- # prior (completeness, plan cover, concrete claims, trace echo)
1324
- # is the fallback ORM.
1314
+ # prior (completeness, concrete claims, trace echo) is the fallback ORM.
1325
1315
  req_toks = request.downcase.scan(/[a-z0-9_]{3,}/).uniq
1326
1316
  fin_toks = final.downcase.scan(/[a-z0-9_]{3,}/).uniq
1327
1317
  overlap = req_toks.empty? ? 1.0 : (req_toks & fin_toks).length.to_f / req_toks.length
@@ -1403,8 +1393,6 @@ module PWN
1403
1393
  score += 0.08 if ov >= 0.25
1404
1394
  score -= 0.12 if ov < 0.08 && req_toks.length >= 4
1405
1395
  end
1406
- cov = plan_coverage(plan: opts[:plan], final: final, request: request, trace: trace) if respond_to?(:plan_coverage)
1407
- score = ((score * 0.55) + (cov[:score].to_f * 0.45)) if cov && cov[:total].to_i.positive?
1408
1396
  { score: score.round(3).clamp(0.0, 0.9), confidence: 0.62 }
1409
1397
  rescue StandardError
1410
1398
  nil
@@ -86,7 +86,6 @@ module PWN
86
86
  request: opts[:request],
87
87
  final: opts[:final],
88
88
  predicted: opts[:predicted],
89
- plan: opts[:plan],
90
89
  ts_state: opts[:ts_state],
91
90
  inline: true
92
91
  }
@@ -26,7 +26,7 @@ module PWN
26
26
  report_name = opts[:report_name] ||= File.basename(Dir.pwd)
27
27
  File.write(
28
28
  "#{dir_path}/#{report_name}.json",
29
- JSON.pretty_generate(results_hash)
29
+ ::JSON.pretty_generate(results_hash)
30
30
  )
31
31
 
32
32
  column_names = [
@@ -0,0 +1,38 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'csv'
4
+
5
+ module PWN
6
+ module Reports
7
+ # Generic CSV report writer for pentest / findings payloads.
8
+ module CSV
9
+ public_class_method def self.generate(opts = {})
10
+ out = PWN::Reports.resolve_path(opts.merge(ext: 'csv'))
11
+ payload = PWN::Reports.report_payload(opts)
12
+ rows = payload[:findings]
13
+ headers = %w[id title severity cvss epss description poc impact recommendation]
14
+ extra = rows.flat_map(&:keys).uniq - headers
15
+ cols = (headers + extra).uniq
16
+ ::CSV.open(out, 'w') do |csv|
17
+ csv << cols
18
+ if rows.empty?
19
+ csv << cols.map { |col| col == 'title' ? payload[:title] : nil }
20
+ else
21
+ rows.each do |row|
22
+ csv << cols.map { |col| row[col] }
23
+ end
24
+ end
25
+ end
26
+ out
27
+ end
28
+
29
+ public_class_method def self.authors
30
+ "AUTHOR(S):\n 0day Inc. <support@0dayinc.com>\n"
31
+ end
32
+
33
+ public_class_method def self.help
34
+ puts "USAGE:\n #{self}.generate(\n path: '/tmp/report.csv',\n results_hash: {}\n )\n\n #{self}.authors\n"
35
+ end
36
+ end
37
+ end
38
+ end
@@ -27,7 +27,7 @@ module PWN
27
27
  # JSON object Completion
28
28
  File.open("#{dir_path}/#{report_name}.json", "w:#{char_encoding}") do |f|
29
29
  f.print(
30
- JSON.pretty_generate(results_hash).force_encoding(char_encoding)
30
+ ::JSON.pretty_generate(results_hash).force_encoding(char_encoding)
31
31
  )
32
32
  end
33
33