pwn 0.5.705 → 0.5.707
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/Gemfile +1 -1
- data/documentation/Reporting.md +1 -0
- data/etc/default_skills/pwn/reports/SKILL.md +4 -2
- data/etc/default_skills/pwn/reports/csv/SKILL.md +47 -0
- data/etc/default_skills/pwn/reports/html/SKILL.md +47 -0
- data/etc/default_skills/pwn/reports/json/SKILL.md +47 -0
- data/etc/default_skills/pwn/reports/markdown/SKILL.md +47 -0
- data/etc/default_skills/pwn/reports/pdf/SKILL.md +47 -0
- data/etc/default_skills/pwn/reports/xml/SKILL.md +47 -0
- data/lib/pwn/ai/agent/curriculum.rb +22 -20
- data/lib/pwn/ai/agent/learning.rb +3 -10
- data/lib/pwn/ai/agent/loop.rb +243 -52
- data/lib/pwn/ai/agent/policy.rb +23 -5
- data/lib/pwn/ai/agent/reward.rb +11 -23
- data/lib/pwn/ai/agent/turn_finalizer.rb +0 -1
- data/lib/pwn/reports/ai_red_team.rb +1 -1
- data/lib/pwn/reports/csv.rb +38 -0
- data/lib/pwn/reports/fuzz.rb +1 -1
- data/lib/pwn/reports/html.rb +58 -0
- data/lib/pwn/reports/json.rb +32 -0
- data/lib/pwn/reports/markdown.rb +40 -0
- data/lib/pwn/reports/pdf.rb +93 -0
- data/lib/pwn/reports/phone.rb +1 -1
- data/lib/pwn/reports/sast.rb +1 -1
- data/lib/pwn/reports/uri_buster.rb +1 -1
- data/lib/pwn/reports/xml.rb +44 -0
- data/lib/pwn/reports.rb +54 -6
- data/lib/pwn/version.rb +1 -1
- data/spec/integration/reinforced_feedback_loop_spec.rb +19 -4
- data/spec/lib/pwn/ai/agent/loop_spec.rb +166 -10
- data/spec/lib/pwn/ai/agent/policy_spec.rb +12 -2
- data/spec/lib/pwn/ai/agent/reward_spec.rb +72 -0
- data/spec/lib/pwn/reports/csv_spec.rb +19 -0
- data/spec/lib/pwn/reports/formats_spec.rb +90 -0
- data/spec/lib/pwn/reports/html_spec.rb +19 -0
- data/spec/lib/pwn/reports/json_spec.rb +19 -0
- data/spec/lib/pwn/reports/markdown_spec.rb +19 -0
- data/spec/lib/pwn/reports/pdf_spec.rb +19 -0
- data/spec/lib/pwn/reports/xml_spec.rb +19 -0
- data/third_party/pwn_rdoc.jsonl +46 -1
- metadata +22 -3
data/lib/pwn/ai/agent/loop.rb
CHANGED
|
@@ -3,6 +3,8 @@
|
|
|
3
3
|
require 'json'
|
|
4
4
|
require 'securerandom'
|
|
5
5
|
require 'digest'
|
|
6
|
+
require 'fileutils'
|
|
7
|
+
require 'tmpdir'
|
|
6
8
|
require 'pwn/ai/agent/mistakes'
|
|
7
9
|
|
|
8
10
|
module PWN
|
|
@@ -324,7 +326,12 @@ module PWN
|
|
|
324
326
|
|
|
325
327
|
top = Mistakes.top(limit: 24, unresolved_only: true)
|
|
326
328
|
now = Time.now
|
|
327
|
-
budget = top.select
|
|
329
|
+
budget = top.select do |mistake|
|
|
330
|
+
next false if mistake[:parked]
|
|
331
|
+
next false if %w[agent_loop assistant_answer].include?(mistake[:tool].to_s)
|
|
332
|
+
|
|
333
|
+
budget_hit?(mistake: mistake)
|
|
334
|
+
end
|
|
328
335
|
budget.each do |mistake|
|
|
329
336
|
stamp = mistake_ts(mistake: mistake)
|
|
330
337
|
next if stamp && (now - stamp) <= PARK_COOL_SECS
|
|
@@ -584,6 +591,10 @@ module PWN
|
|
|
584
591
|
live = effects.reject { |fx| %i[recall store].include?(fx) }
|
|
585
592
|
return true if live.empty?
|
|
586
593
|
return true if duration_unsatisfied?(request: request)
|
|
594
|
+
return true if declared_contract_unsatisfied?(
|
|
595
|
+
request: request,
|
|
596
|
+
messages: opts[:messages]
|
|
597
|
+
)
|
|
587
598
|
return true if need == :write && !write_verified?(effects: effects)
|
|
588
599
|
return true if need == :browse && !effects.include?(:browse)
|
|
589
600
|
return true if need == :any && !effects.intersect?(%i[write browse eval])
|
|
@@ -613,23 +624,166 @@ module PWN
|
|
|
613
624
|
}.freeze
|
|
614
625
|
|
|
615
626
|
private_class_method def self.duration_unsatisfied?(opts = {})
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
619
|
-
hours =
|
|
620
|
-
|
|
621
|
-
|
|
627
|
+
secs = declared_min_seconds(request: opts[:request])
|
|
628
|
+
if secs <= 0
|
|
629
|
+
req = opts[:request].to_s
|
|
630
|
+
hours = nil
|
|
631
|
+
if (m = req.match(/\b(\d+)\s*(?:hours?|hrs?)\b/i))
|
|
632
|
+
hours = m[1].to_i
|
|
633
|
+
elsif (m = req.match(/\b(twenty-four|one|two|three|four|five|six|seven|eight|nine|ten|eleven|twelve|thirteen|fourteen|fifteen|sixteen|seventeen|eighteen|nineteen|twenty)\s*(?:hours?|hrs?)\b/i))
|
|
634
|
+
hours = HOUR_WORDS[m[1].downcase]
|
|
635
|
+
end
|
|
636
|
+
secs = hours.to_i * 3600
|
|
622
637
|
end
|
|
623
|
-
return false if
|
|
638
|
+
return false if secs <= 0
|
|
624
639
|
|
|
625
640
|
t0 = Thread.current[:pwn_loop_t0]
|
|
626
641
|
return false unless t0
|
|
627
642
|
|
|
628
|
-
(Time.now - t0) <
|
|
643
|
+
(Time.now - t0) < secs
|
|
644
|
+
rescue StandardError
|
|
645
|
+
false
|
|
646
|
+
end
|
|
647
|
+
|
|
648
|
+
EMPTY_CONTRACT = {
|
|
649
|
+
paths: [],
|
|
650
|
+
min_seconds: 0,
|
|
651
|
+
skills: [],
|
|
652
|
+
proofs: [],
|
|
653
|
+
hosts: [],
|
|
654
|
+
techniques: [],
|
|
655
|
+
issue_work: false
|
|
656
|
+
}.freeze
|
|
657
|
+
|
|
658
|
+
private_class_method def self.declared_min_seconds(opts = {})
|
|
659
|
+
declared_contract(request: opts[:request])[:min_seconds].to_i
|
|
660
|
+
end
|
|
661
|
+
|
|
662
|
+
private_class_method def self.declared_contract_unsatisfied?(opts = {})
|
|
663
|
+
contract = declared_contract(request: opts[:request])
|
|
664
|
+
files = Array(contract[:paths]) + Array(contract[:proofs])
|
|
665
|
+
return true if files.any? { |path| deliverable_missing?(path: path) }
|
|
666
|
+
return true if declared_skills_missing?(skills: contract[:skills])
|
|
667
|
+
return true if declared_hosts_missing?(hosts: contract[:hosts], messages: opts[:messages])
|
|
668
|
+
return true if declared_hosts_missing?(hosts: contract[:techniques], messages: opts[:messages])
|
|
669
|
+
return true if contract[:issue_work] && Array(contract[:proofs]).empty?
|
|
670
|
+
|
|
671
|
+
false
|
|
629
672
|
rescue StandardError
|
|
630
673
|
false
|
|
631
674
|
end
|
|
632
675
|
|
|
676
|
+
private_class_method def self.declared_skills_missing?(opts = {})
|
|
677
|
+
names = Array(opts[:skills]).map(&:to_s).reject(&:empty?)
|
|
678
|
+
return false if names.empty?
|
|
679
|
+
return true unless defined?(PWN::Skills) && PWN::Skills.is_a?(Hash)
|
|
680
|
+
|
|
681
|
+
have = PWN::Skills.keys.map(&:to_s)
|
|
682
|
+
names.any? { |name| !have.include?(name) }
|
|
683
|
+
end
|
|
684
|
+
|
|
685
|
+
private_class_method def self.declared_hosts_missing?(opts = {})
|
|
686
|
+
hosts = Array(opts[:hosts]).map(&:to_s).reject(&:empty?)
|
|
687
|
+
return false if hosts.empty?
|
|
688
|
+
|
|
689
|
+
blob = Array(opts[:messages]).select { |msg| msg.is_a?(Hash) && msg[:role].to_s == 'tool' }
|
|
690
|
+
.map { |msg| msg[:content].to_s }
|
|
691
|
+
.join("\n")
|
|
692
|
+
.downcase
|
|
693
|
+
hosts.any? { |host| !blob.include?(host.to_s.downcase) }
|
|
694
|
+
end
|
|
695
|
+
|
|
696
|
+
private_class_method def self.declared_deliverables(opts = {})
|
|
697
|
+
Array(declared_contract(request: opts[:request])[:paths])
|
|
698
|
+
end
|
|
699
|
+
|
|
700
|
+
private_class_method def self.declared_contract(opts = {})
|
|
701
|
+
cached = Thread.current[:pwn_loop_deliverables]
|
|
702
|
+
return normalize_contract(raw: cached) if cached.is_a?(Array) || cached.is_a?(Hash)
|
|
703
|
+
return EMPTY_CONTRACT.dup unless Thread.current[:pwn_loop_active]
|
|
704
|
+
|
|
705
|
+
contract = infer_deliverables(request: opts[:request])
|
|
706
|
+
Thread.current[:pwn_loop_deliverables] = contract
|
|
707
|
+
contract
|
|
708
|
+
end
|
|
709
|
+
|
|
710
|
+
private_class_method def self.normalize_contract(opts = {})
|
|
711
|
+
raw = opts[:raw]
|
|
712
|
+
return EMPTY_CONTRACT.merge(paths: raw.map(&:to_s).select { |p| p.start_with?('/') }) if raw.is_a?(Array)
|
|
713
|
+
return EMPTY_CONTRACT.dup unless raw.is_a?(Hash)
|
|
714
|
+
|
|
715
|
+
hours = raw[:hours] || raw['hours']
|
|
716
|
+
secs = (raw[:min_seconds] || raw['min_seconds']).to_i
|
|
717
|
+
secs = hours.to_i * 3600 if secs <= 0 && hours.to_i.positive?
|
|
718
|
+
{
|
|
719
|
+
paths: abs_paths(rows: raw[:paths] || raw['paths']),
|
|
720
|
+
min_seconds: secs,
|
|
721
|
+
skills: Array(raw[:skills] || raw['skills']).map(&:to_s).reject(&:empty?).uniq,
|
|
722
|
+
proofs: abs_paths(rows: raw[:proofs] || raw['proofs']),
|
|
723
|
+
hosts: Array(raw[:hosts] || raw['hosts']).map(&:to_s).reject(&:empty?).uniq,
|
|
724
|
+
techniques: Array(raw[:techniques] || raw['techniques']).map(&:to_s).reject(&:empty?).uniq,
|
|
725
|
+
issue_work: raw[:issue_work] == true || raw['issue_work'] == true
|
|
726
|
+
}
|
|
727
|
+
end
|
|
728
|
+
|
|
729
|
+
private_class_method def self.abs_paths(opts = {})
|
|
730
|
+
Array(opts[:rows]).map(&:to_s).select { |path| path.start_with?('/') }.uniq
|
|
731
|
+
end
|
|
732
|
+
|
|
733
|
+
private_class_method def self.infer_deliverables(opts = {})
|
|
734
|
+
request = opts[:request].to_s
|
|
735
|
+
return EMPTY_CONTRACT.dup if request.strip.empty?
|
|
736
|
+
|
|
737
|
+
reply = call_engine(
|
|
738
|
+
messages: [
|
|
739
|
+
{
|
|
740
|
+
role: 'system',
|
|
741
|
+
content: 'Reply with JSON only. No markdown. No tools.'
|
|
742
|
+
},
|
|
743
|
+
{
|
|
744
|
+
role: 'user',
|
|
745
|
+
content: "Operator request:\n#{request}\n\n" \
|
|
746
|
+
'When that request is complete, what must be true on this host? ' \
|
|
747
|
+
'JSON only: {"paths":["/abs/file"],"min_seconds":0,"skills":["name"],' \
|
|
748
|
+
'"proofs":["/abs/poc"],"hosts":["ip-or-hostname"],' \
|
|
749
|
+
'"techniques":["T1059"],"issue_work":false}. ' \
|
|
750
|
+
'issue_work=true when the ask needs findings, PoCs, or severity — then proofs must be non-empty absolute paths. ' \
|
|
751
|
+
'Write reports with PWN::Reports::PDF.generate / HTML / Markdown / XML / CSV / JSON (path: or dir_path: + report_name:). ' \
|
|
752
|
+
'Use [] or 0 when a field is not required. Do not invent work. Paths must be absolute.'
|
|
753
|
+
}
|
|
754
|
+
],
|
|
755
|
+
tools: nil
|
|
756
|
+
)
|
|
757
|
+
text = reply.is_a?(Hash) ? (reply[:content] || reply['content']).to_s : reply.to_s
|
|
758
|
+
parse_contract(text: text)
|
|
759
|
+
rescue StandardError
|
|
760
|
+
EMPTY_CONTRACT.dup
|
|
761
|
+
end
|
|
762
|
+
|
|
763
|
+
private_class_method def self.parse_contract(opts = {})
|
|
764
|
+
text = opts[:text].to_s
|
|
765
|
+
json = nil
|
|
766
|
+
begin
|
|
767
|
+
json = JSON.parse(text, symbolize_names: true)
|
|
768
|
+
rescue JSON::ParserError
|
|
769
|
+
start = text.index('{')
|
|
770
|
+
stop = text.rindex('}')
|
|
771
|
+
return EMPTY_CONTRACT.dup unless start && stop && stop > start
|
|
772
|
+
|
|
773
|
+
json = JSON.parse(text[start..stop], symbolize_names: true)
|
|
774
|
+
end
|
|
775
|
+
normalize_contract(raw: json)
|
|
776
|
+
rescue StandardError
|
|
777
|
+
EMPTY_CONTRACT.dup
|
|
778
|
+
end
|
|
779
|
+
|
|
780
|
+
private_class_method def self.deliverable_missing?(opts = {})
|
|
781
|
+
path = opts[:path].to_s
|
|
782
|
+
path.empty? || !File.file?(path) || File.size(path) <= 0
|
|
783
|
+
rescue StandardError
|
|
784
|
+
true
|
|
785
|
+
end
|
|
786
|
+
|
|
633
787
|
private_class_method def self.may_finalize?(opts = {})
|
|
634
788
|
return false if incomplete_final?(text: opts[:text], last_iter: false)
|
|
635
789
|
return false if request_unsatisfied?(
|
|
@@ -962,12 +1116,17 @@ module PWN
|
|
|
962
1116
|
end
|
|
963
1117
|
|
|
964
1118
|
private_class_method def self.no_progress_result(opts = {})
|
|
1119
|
+
checkpoint_result(opts)
|
|
1120
|
+
end
|
|
1121
|
+
|
|
1122
|
+
private_class_method def self.checkpoint_result(opts = {})
|
|
965
1123
|
name = opts[:name].to_s
|
|
966
1124
|
sig = payload_sig(opts)
|
|
967
1125
|
JSON.generate(
|
|
968
|
-
success:
|
|
969
|
-
|
|
970
|
-
|
|
1126
|
+
success: true,
|
|
1127
|
+
checkpoint: true,
|
|
1128
|
+
error: "checkpoint: identical #{name} payload (#{sig}). World unchanged — vary args, target, or tool.",
|
|
1129
|
+
result: { stdout: "checkpoint #{name} #{sig}", stderr: '', exit: 0 }
|
|
971
1130
|
)
|
|
972
1131
|
end
|
|
973
1132
|
|
|
@@ -1029,15 +1188,10 @@ module PWN
|
|
|
1029
1188
|
return nil unless plan_msg && !plan_msg[:content].to_s.strip.empty?
|
|
1030
1189
|
|
|
1031
1190
|
plan = plan_msg[:content].to_s.strip
|
|
1032
|
-
|
|
1033
|
-
#
|
|
1034
|
-
# P17 — never fork red_team when budget fingerprints dominate: it is
|
|
1035
|
-
# another mini agent loop and compounds iteration-budget exhaustion.
|
|
1191
|
+
# TUI-only. Do not put PLAN: or red-team text on the model wire —
|
|
1192
|
+
# original request stays the only user goal.
|
|
1036
1193
|
rt = nil
|
|
1037
|
-
if defined?(Curriculum) && !hot
|
|
1038
|
-
rt = Curriculum.red_team_plan(request: opts[:request], plan: plan)
|
|
1039
|
-
messages << { role: 'user', content: rt } if rt
|
|
1040
|
-
end
|
|
1194
|
+
rt = Curriculum.red_team_plan(request: opts[:request], plan: plan) if defined?(Curriculum) && !hot
|
|
1041
1195
|
# P2 — unify TaskSummarizer plan object with surviving outline so the
|
|
1042
1196
|
# task line and adversarial/plan_first plan are one thing. Index-only;
|
|
1043
1197
|
# credit stays in Reward. Optional: only when ts_state is live.
|
|
@@ -1541,11 +1695,7 @@ module PWN
|
|
|
1541
1695
|
collapsed << pair
|
|
1542
1696
|
end
|
|
1543
1697
|
kept = collapsed.last(keep_pairs).flatten
|
|
1544
|
-
kept
|
|
1545
|
-
next unless m[:role].to_s == 'tool' && m[:content].to_s.length > max_chars
|
|
1546
|
-
|
|
1547
|
-
m[:content] = "#{m[:content].to_s[0, max_chars]}…[compacted]"
|
|
1548
|
-
end
|
|
1698
|
+
spill_tool_history!(messages: kept, max_chars: max_chars)
|
|
1549
1699
|
messages.replace(head + kept)
|
|
1550
1700
|
repair_tool_history!(messages: messages)
|
|
1551
1701
|
messages
|
|
@@ -1554,6 +1704,43 @@ module PWN
|
|
|
1554
1704
|
opts[:messages]
|
|
1555
1705
|
end
|
|
1556
1706
|
|
|
1707
|
+
HISTORY_SPILL_DIR = File.join(Dir.tmpdir, 'pwn-ai-hist')
|
|
1708
|
+
KEEP_FULL_TOOL_TAILS = 2
|
|
1709
|
+
|
|
1710
|
+
private_class_method def self.spill_tool_history!(opts = {})
|
|
1711
|
+
messages = Array(opts[:messages])
|
|
1712
|
+
max_chars = opts[:max_chars].to_i
|
|
1713
|
+
max_chars = 2_000 if max_chars <= 0
|
|
1714
|
+
tools = messages.select { |msg| msg[:role].to_s == 'tool' }
|
|
1715
|
+
spill_n = [tools.length - KEEP_FULL_TOOL_TAILS, 0].max
|
|
1716
|
+
idx = 0
|
|
1717
|
+
messages.each do |msg|
|
|
1718
|
+
next unless msg[:role].to_s == 'tool'
|
|
1719
|
+
|
|
1720
|
+
idx += 1
|
|
1721
|
+
next if idx > spill_n
|
|
1722
|
+
|
|
1723
|
+
body = msg[:content].to_s
|
|
1724
|
+
next if body.length <= max_chars
|
|
1725
|
+
next if body.include?('[compacted path=')
|
|
1726
|
+
|
|
1727
|
+
msg[:content] = spill_tool_body(text: body)
|
|
1728
|
+
end
|
|
1729
|
+
messages
|
|
1730
|
+
end
|
|
1731
|
+
|
|
1732
|
+
private_class_method def self.spill_tool_body(opts = {})
|
|
1733
|
+
text = opts[:text].to_s
|
|
1734
|
+
digest = Digest::SHA256.hexdigest(text)[0, 16]
|
|
1735
|
+
dir = HISTORY_SPILL_DIR
|
|
1736
|
+
FileUtils.mkdir_p(dir)
|
|
1737
|
+
path = File.join(dir, "#{digest}.txt")
|
|
1738
|
+
File.binwrite(path, text) unless File.file?(path)
|
|
1739
|
+
"[compacted path=#{path} sha256=#{digest} bytes=#{text.bytesize}]"
|
|
1740
|
+
rescue StandardError
|
|
1741
|
+
"[compacted bytes=#{opts[:text].to_s.bytesize}]"
|
|
1742
|
+
end
|
|
1743
|
+
|
|
1557
1744
|
private_class_method def self.repair_tool_history!(opts = {})
|
|
1558
1745
|
messages = opts[:messages]
|
|
1559
1746
|
return messages unless messages.is_a?(Array)
|
|
@@ -1913,9 +2100,7 @@ module PWN
|
|
|
1913
2100
|
session_id: session_id,
|
|
1914
2101
|
request: request,
|
|
1915
2102
|
final: txt,
|
|
1916
|
-
predicted: 0.85
|
|
1917
|
-
plan: [],
|
|
1918
|
-
ts_state: nil
|
|
2103
|
+
predicted: 0.85
|
|
1919
2104
|
)
|
|
1920
2105
|
end
|
|
1921
2106
|
txt
|
|
@@ -1984,9 +2169,7 @@ module PWN
|
|
|
1984
2169
|
session_id: session_id,
|
|
1985
2170
|
request: request,
|
|
1986
2171
|
final: txt,
|
|
1987
|
-
predicted: 0.85
|
|
1988
|
-
plan: [],
|
|
1989
|
-
ts_state: nil
|
|
2172
|
+
predicted: 0.85
|
|
1990
2173
|
)
|
|
1991
2174
|
end
|
|
1992
2175
|
txt
|
|
@@ -2010,9 +2193,7 @@ module PWN
|
|
|
2010
2193
|
session_id: session_id,
|
|
2011
2194
|
request: request,
|
|
2012
2195
|
final: txt,
|
|
2013
|
-
predicted: 0.95
|
|
2014
|
-
plan: ['Acknowledge greeting without tools or weather echo'],
|
|
2015
|
-
ts_state: nil
|
|
2196
|
+
predicted: 0.95
|
|
2016
2197
|
)
|
|
2017
2198
|
end
|
|
2018
2199
|
txt
|
|
@@ -2094,9 +2275,7 @@ module PWN
|
|
|
2094
2275
|
session_id: session_id,
|
|
2095
2276
|
request: request,
|
|
2096
2277
|
final: txt,
|
|
2097
|
-
predicted: 0.9
|
|
2098
|
-
plan: ['Explain tool usage without live recon'],
|
|
2099
|
-
ts_state: nil
|
|
2278
|
+
predicted: 0.9
|
|
2100
2279
|
)
|
|
2101
2280
|
end
|
|
2102
2281
|
txt
|
|
@@ -2306,9 +2485,7 @@ module PWN
|
|
|
2306
2485
|
session_id: session_id,
|
|
2307
2486
|
request: request,
|
|
2308
2487
|
final: txt,
|
|
2309
|
-
predicted: 0.95
|
|
2310
|
-
plan: [plan_label || 'Recall prior turn from session transcript'],
|
|
2311
|
-
ts_state: nil
|
|
2488
|
+
predicted: 0.95
|
|
2312
2489
|
)
|
|
2313
2490
|
end
|
|
2314
2491
|
return txt
|
|
@@ -2379,9 +2556,7 @@ module PWN
|
|
|
2379
2556
|
session_id: session_id,
|
|
2380
2557
|
request: request,
|
|
2381
2558
|
final: txt,
|
|
2382
|
-
predicted: 0.85
|
|
2383
|
-
plan: ['Recall prior turn (empty session fallback)'],
|
|
2384
|
-
ts_state: nil
|
|
2559
|
+
predicted: 0.85
|
|
2385
2560
|
)
|
|
2386
2561
|
end
|
|
2387
2562
|
txt
|
|
@@ -2434,6 +2609,10 @@ module PWN
|
|
|
2434
2609
|
Thread.current[:pwn_extinguished] = {}
|
|
2435
2610
|
Thread.current[:pwn_same_payload] = Hash.new(0)
|
|
2436
2611
|
Thread.current[:pwn_loop_t0] = Time.now unless nested
|
|
2612
|
+
unless nested
|
|
2613
|
+
Thread.current[:pwn_loop_active] = true
|
|
2614
|
+
Thread.current[:pwn_loop_deliverables] = nil
|
|
2615
|
+
end
|
|
2437
2616
|
debug_progress(msg: "intent=#{intent} engine=#{engine}", debug: opts[:debug])
|
|
2438
2617
|
expose_current_session(session_id: session_id)
|
|
2439
2618
|
Mistakes.check_user_correction(request: request, session_id: session_id) if defined?(Mistakes)
|
|
@@ -2520,6 +2699,10 @@ module PWN
|
|
|
2520
2699
|
# CORE_TOOLS is the default action space. Extra schemas are
|
|
2521
2700
|
# opt-in via enabled_toolsets + core_only: false.
|
|
2522
2701
|
core_only = opts.fetch(:core_only, true)
|
|
2702
|
+
if nested && needs_host_work?(request: request)
|
|
2703
|
+
opts[:enabled_toolsets] = nil
|
|
2704
|
+
core_only = true
|
|
2705
|
+
end
|
|
2523
2706
|
tools = Registry.definitions(
|
|
2524
2707
|
enabled: opts[:enabled_toolsets],
|
|
2525
2708
|
relevance: request,
|
|
@@ -2672,12 +2855,20 @@ module PWN
|
|
|
2672
2855
|
# commit that as the answer; drop the empty assistant turn,
|
|
2673
2856
|
# inject a one-shot nudge, and keep iterating.
|
|
2674
2857
|
if calls.empty? && text.strip.empty?
|
|
2858
|
+
unsat = request_unsatisfied?(request: request, messages: messages)
|
|
2675
2859
|
warn "[pwn-ai/loop] empty final from #{engine} on iter=#{i}; nudging" if local
|
|
2860
|
+
empty_nudge = if unsat
|
|
2861
|
+
'Your previous reply was empty (no tool_calls and no content). ' \
|
|
2862
|
+
'The original request is not evidenced yet. Emit NATIVE tool_calls NOW. ' \
|
|
2863
|
+
'Do not write a final answer until that request is done or a tool returned failure evidence.'
|
|
2864
|
+
else
|
|
2865
|
+
'Your previous reply was empty (no tool_calls and no content). ' \
|
|
2866
|
+
'Either call a tool now, or write the final answer for the user as plain text. ' \
|
|
2867
|
+
'Do not reply with an empty message.'
|
|
2868
|
+
end
|
|
2676
2869
|
messages << {
|
|
2677
2870
|
role: 'user',
|
|
2678
|
-
content:
|
|
2679
|
-
'Either call a tool now, or write the final answer for the user as plain text. ' \
|
|
2680
|
-
'Do not reply with an empty message.'
|
|
2871
|
+
content: empty_nudge
|
|
2681
2872
|
}
|
|
2682
2873
|
turn_fails['empty_final'] += 1
|
|
2683
2874
|
debug_progress(msg: "bounce empty_final snippet=#{debug_snippet(text: text)}")
|
|
@@ -2720,7 +2911,7 @@ module PWN
|
|
|
2720
2911
|
debug_final_text!(text: text)
|
|
2721
2912
|
final_chars = text.to_s.length
|
|
2722
2913
|
append_session(session_id: session_id, role: 'assistant', content: text)
|
|
2723
|
-
Learning.auto_introspect(session_id: session_id, request: request, final: text, predicted: predicted,
|
|
2914
|
+
Learning.auto_introspect(session_id: session_id, request: request, final: text, predicted: predicted, ts_state: ts_state) if defined?(Learning) && !nested && !no_tools && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: i)
|
|
2724
2915
|
maybe_finish_policy(session_id: session_id, proxy_ok: true, ts_state: ts_state)
|
|
2725
2916
|
task_summary_flush!(state: ts_state, on_tool: on_tool)
|
|
2726
2917
|
OpenGoal.clear! if defined?(OpenGoal) && !nested
|
|
@@ -2754,11 +2945,7 @@ module PWN
|
|
|
2754
2945
|
else
|
|
2755
2946
|
raw = Dispatch.call(tool_call: tc)
|
|
2756
2947
|
same_n = note_same_payload!(name: name, args: args)
|
|
2757
|
-
if same_n >= 3
|
|
2758
|
-
Thread.current[:pwn_extinguished] ||= {}
|
|
2759
|
-
Thread.current[:pwn_extinguished][sig] = true
|
|
2760
|
-
raw = no_progress_result(name: name, args: args)
|
|
2761
|
-
end
|
|
2948
|
+
raw = checkpoint_result(name: name, args: args) if same_n >= 3
|
|
2762
2949
|
end
|
|
2763
2950
|
tools_called += 1
|
|
2764
2951
|
tele = record_metrics(name: name, started: started, raw: raw, args: args, session_id: session_id, engine: engine, ts_state: ts_state)
|
|
@@ -2830,6 +3017,10 @@ module PWN
|
|
|
2830
3017
|
end
|
|
2831
3018
|
raise
|
|
2832
3019
|
ensure
|
|
3020
|
+
unless nested
|
|
3021
|
+
Thread.current[:pwn_loop_active] = nil
|
|
3022
|
+
Thread.current[:pwn_loop_deliverables] = nil
|
|
3023
|
+
end
|
|
2833
3024
|
Thread.current[:pwn_loop_no_tools] = nil
|
|
2834
3025
|
finish_debug_request!(
|
|
2835
3026
|
iter: i,
|
data/lib/pwn/ai/agent/policy.rb
CHANGED
|
@@ -752,16 +752,34 @@ module PWN
|
|
|
752
752
|
return 0.0 unless ep.is_a?(Hash)
|
|
753
753
|
|
|
754
754
|
bonus = 0.0
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
755
|
+
met = contract_fields_met(request: ep[:request].to_s)
|
|
756
|
+
prev = ep[:contract_met].to_i
|
|
757
|
+
if met > prev
|
|
758
|
+
bonus += STEP_TASK
|
|
759
|
+
ep[:contract_met] = met
|
|
760
|
+
end
|
|
761
|
+
bonus += STEP_GRIND if met.positive? && opts[:ok] && opts[:action].to_s != 'final'
|
|
760
762
|
bonus
|
|
761
763
|
rescue StandardError
|
|
762
764
|
0.0
|
|
763
765
|
end
|
|
764
766
|
|
|
767
|
+
private_class_method def self.contract_fields_met(opts = {})
|
|
768
|
+
raw = Thread.current[:pwn_loop_deliverables]
|
|
769
|
+
return 0 unless raw.is_a?(Hash) || opts.key?(:request)
|
|
770
|
+
return 0 unless raw.is_a?(Hash)
|
|
771
|
+
|
|
772
|
+
n = 0
|
|
773
|
+
n += 1 if raw[:min_seconds].to_i.positive? || raw['min_seconds'].to_i.positive?
|
|
774
|
+
files = Array(raw[:paths] || raw['paths']) + Array(raw[:proofs] || raw['proofs'])
|
|
775
|
+
n += files.count { |path| File.file?(path.to_s) && File.size(path.to_s).positive? }
|
|
776
|
+
n += Array(raw[:hosts] || raw['hosts']).length
|
|
777
|
+
n += Array(raw[:techniques] || raw['techniques']).length
|
|
778
|
+
n
|
|
779
|
+
rescue StandardError
|
|
780
|
+
0
|
|
781
|
+
end
|
|
782
|
+
|
|
765
783
|
private_class_method def self.terminal_reward(opts = {})
|
|
766
784
|
return ((2.0 * opts[:score].to_f) - 1.0).clamp(-1.0, 1.0) unless opts[:score].nil?
|
|
767
785
|
|
data/lib/pwn/ai/agent/reward.rb
CHANGED
|
@@ -81,15 +81,16 @@ module PWN
|
|
|
81
81
|
|
|
82
82
|
JUDGE_SYSTEM = <<~SYS
|
|
83
83
|
You are the pwn-ai Outcome Reward Model. Given a USER REQUEST, the
|
|
84
|
-
agent's FINAL ANSWER, a compressed TOOL TRACE,
|
|
85
|
-
|
|
84
|
+
agent's FINAL ANSWER, and a compressed TOOL TRACE, emit ONE line of
|
|
85
|
+
strict JSON:
|
|
86
86
|
{"score": <0.0-1.0>, "verdict": "solved|partial|wrong|refused",
|
|
87
87
|
"rationale": "<≤140 chars>", "key_step": <int|-1>}
|
|
88
|
-
Grade the HUMAN RESULT
|
|
88
|
+
Grade the HUMAN RESULT against the USER REQUEST only — never a TUI
|
|
89
|
+
plan, stub outline, or competing compass:
|
|
89
90
|
1.0 = final is usable and complete (every asked point answered with
|
|
90
91
|
evidence from the trace or a checkable claim).
|
|
91
92
|
0.7 = mostly complete, one missing detail, still usable.
|
|
92
|
-
0.5 = correct direction but incomplete / truncated
|
|
93
|
+
0.5 = correct direction but incomplete / truncated.
|
|
93
94
|
0.2 = tools ran but the final does not answer the ask.
|
|
94
95
|
0.0 = hallucinated, off-goal, empty, polite non-answer, or refused.
|
|
95
96
|
Ignore {"success":true} as evidence of done. Prefer last tool steps.
|
|
@@ -124,8 +125,8 @@ module PWN
|
|
|
124
125
|
trace = load_trace(session_id: opts[:session_id]) if trace.empty? && opts[:session_id]
|
|
125
126
|
commit = opts.key?(:commit) ? opts[:commit] : true
|
|
126
127
|
|
|
127
|
-
v = llm_judge(request: request, final: final, trace: trace
|
|
128
|
-
v ||= heuristic_judge(request: request, final: final, trace: trace
|
|
128
|
+
v = llm_judge(request: request, final: final, trace: trace)
|
|
129
|
+
v ||= heuristic_judge(request: request, final: final, trace: trace)
|
|
129
130
|
# Cheap ORM is the intended source. Heuristic overlap is fallback
|
|
130
131
|
# only — callers (sentinel / Learning.stats / Metrics.effective_rate)
|
|
131
132
|
# weight :llm_orm samples above :heuristic so the haircut tracks
|
|
@@ -1075,17 +1076,7 @@ module PWN
|
|
|
1075
1076
|
# human got a usable result. Keep the first step for context.
|
|
1076
1077
|
shown = compact_trace_tail(steps: steps, keep: CHEAP_ORM_TRACE_N)
|
|
1077
1078
|
trace = shown.each_with_index.map { |s, i| "#{i + 1}. #{s.to_s.gsub(/\s+/, ' ')[0, 220]}" }.join("\n")
|
|
1078
|
-
|
|
1079
|
-
if respond_to?(:plan_coverage)
|
|
1080
|
-
cov = plan_coverage(
|
|
1081
|
-
plan: opts[:plan],
|
|
1082
|
-
final: opts[:final],
|
|
1083
|
-
request: opts[:request],
|
|
1084
|
-
trace: steps
|
|
1085
|
-
)
|
|
1086
|
-
plan = "\nPLAN COVERAGE: #{cov[:covered]}/#{cov[:total]} (#{cov[:tag]}) missing=#{Array(cov[:missing]).first(3).join(' | ')}" if cov && cov[:total].to_i.positive?
|
|
1087
|
-
end
|
|
1088
|
-
req = "USER REQUEST:\n#{opts[:request].to_s[0, 700]}\n\nFINAL ANSWER:\n#{opts[:final].to_s[0, 1_600]}\n\nTOOL TRACE (#{steps.length} steps, showing #{shown.length}):\n#{trace}#{plan}"
|
|
1079
|
+
req = "USER REQUEST:\n#{opts[:request].to_s[0, 700]}\n\nFINAL ANSWER:\n#{opts[:final].to_s[0, 1_600]}\n\nTOOL TRACE (#{steps.length} steps, showing #{shown.length}):\n#{trace}"
|
|
1089
1080
|
resp = cheap_orm_chat(request: req, system_role_content: JUDGE_SYSTEM)
|
|
1090
1081
|
parsed = parse_llm_judge(resp: resp)
|
|
1091
1082
|
return nil if parsed.nil?
|
|
@@ -1093,7 +1084,7 @@ module PWN
|
|
|
1093
1084
|
# Soft-blend a stronger-than-overlap evidence prior so a noisy
|
|
1094
1085
|
# cheap ORM cannot peg 0.0/1.0 against an obviously incomplete
|
|
1095
1086
|
# or obviously complete final.
|
|
1096
|
-
ev = evidence_prior(request: opts[:request], final: opts[:final], trace: steps
|
|
1087
|
+
ev = evidence_prior(request: opts[:request], final: opts[:final], trace: steps)
|
|
1097
1088
|
if ev && ev[:confidence].to_f >= 0.5
|
|
1098
1089
|
raw = parsed[:score].to_f
|
|
1099
1090
|
# Sanity bounds only: do not always blend (would fight a good ORM).
|
|
@@ -1317,11 +1308,10 @@ module PWN
|
|
|
1317
1308
|
polite = final.match?(/\A\s*(sure|happy to help|of course|i can help|how can i|let me know)\b/i) && final.length < 120
|
|
1318
1309
|
return { score: 0.1, verdict: :partial, rationale: 'polite non-answer', key_step: -1, source: :heuristic } if polite && trace.empty?
|
|
1319
1310
|
|
|
1320
|
-
ev = evidence_prior(request: request, final: final, trace: trace
|
|
1311
|
+
ev = evidence_prior(request: request, final: final, trace: trace)
|
|
1321
1312
|
score = ev ? ev[:score].to_f : 0.35
|
|
1322
1313
|
# Overlap is a small on-topic gate, not the score. The evidence
|
|
1323
|
-
# prior (completeness,
|
|
1324
|
-
# is the fallback ORM.
|
|
1314
|
+
# prior (completeness, concrete claims, trace echo) is the fallback ORM.
|
|
1325
1315
|
req_toks = request.downcase.scan(/[a-z0-9_]{3,}/).uniq
|
|
1326
1316
|
fin_toks = final.downcase.scan(/[a-z0-9_]{3,}/).uniq
|
|
1327
1317
|
overlap = req_toks.empty? ? 1.0 : (req_toks & fin_toks).length.to_f / req_toks.length
|
|
@@ -1403,8 +1393,6 @@ module PWN
|
|
|
1403
1393
|
score += 0.08 if ov >= 0.25
|
|
1404
1394
|
score -= 0.12 if ov < 0.08 && req_toks.length >= 4
|
|
1405
1395
|
end
|
|
1406
|
-
cov = plan_coverage(plan: opts[:plan], final: final, request: request, trace: trace) if respond_to?(:plan_coverage)
|
|
1407
|
-
score = ((score * 0.55) + (cov[:score].to_f * 0.45)) if cov && cov[:total].to_i.positive?
|
|
1408
1396
|
{ score: score.round(3).clamp(0.0, 0.9), confidence: 0.62 }
|
|
1409
1397
|
rescue StandardError
|
|
1410
1398
|
nil
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'csv'
|
|
4
|
+
|
|
5
|
+
module PWN
|
|
6
|
+
module Reports
|
|
7
|
+
# Generic CSV report writer for pentest / findings payloads.
|
|
8
|
+
module CSV
|
|
9
|
+
public_class_method def self.generate(opts = {})
|
|
10
|
+
out = PWN::Reports.resolve_path(opts.merge(ext: 'csv'))
|
|
11
|
+
payload = PWN::Reports.report_payload(opts)
|
|
12
|
+
rows = payload[:findings]
|
|
13
|
+
headers = %w[id title severity cvss epss description poc impact recommendation]
|
|
14
|
+
extra = rows.flat_map(&:keys).uniq - headers
|
|
15
|
+
cols = (headers + extra).uniq
|
|
16
|
+
::CSV.open(out, 'w') do |csv|
|
|
17
|
+
csv << cols
|
|
18
|
+
if rows.empty?
|
|
19
|
+
csv << cols.map { |col| col == 'title' ? payload[:title] : nil }
|
|
20
|
+
else
|
|
21
|
+
rows.each do |row|
|
|
22
|
+
csv << cols.map { |col| row[col] }
|
|
23
|
+
end
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
out
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
public_class_method def self.authors
|
|
30
|
+
"AUTHOR(S):\n 0day Inc. <support@0dayinc.com>\n"
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
public_class_method def self.help
|
|
34
|
+
puts "USAGE:\n #{self}.generate(\n path: '/tmp/report.csv',\n results_hash: {}\n )\n\n #{self}.authors\n"
|
|
35
|
+
end
|
|
36
|
+
end
|
|
37
|
+
end
|
|
38
|
+
end
|
data/lib/pwn/reports/fuzz.rb
CHANGED
|
@@ -27,7 +27,7 @@ module PWN
|
|
|
27
27
|
# JSON object Completion
|
|
28
28
|
File.open("#{dir_path}/#{report_name}.json", "w:#{char_encoding}") do |f|
|
|
29
29
|
f.print(
|
|
30
|
-
JSON.pretty_generate(results_hash).force_encoding(char_encoding)
|
|
30
|
+
::JSON.pretty_generate(results_hash).force_encoding(char_encoding)
|
|
31
31
|
)
|
|
32
32
|
end
|
|
33
33
|
|