pwn 0.5.680 → 0.5.682

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -36,6 +36,9 @@ module PWN
36
36
  EPSILON = 0.08
37
37
  STEP_OK = 0.05
38
38
  STEP_FAIL = -0.20
39
+ STEP_TASK = 0.12
40
+ STEP_CLOSED = 0.20
41
+ STEP_GRIND = -0.08
39
42
  MAX_TRAJ = 2_000
40
43
  GOLD_MIN = 0.6
41
44
  VISITS_MIN = 2
@@ -147,6 +150,8 @@ module PWN
147
150
  started_at: Time.now.utc.iso8601,
148
151
  state: s0,
149
152
  last_action: 'start',
153
+ plan_idx: ts_idx(ts_state: opts[:ts_state]),
154
+ plan_open: ts_open?(ts_state: opts[:ts_state]),
150
155
  fails: 0,
151
156
  steps: []
152
157
  }
@@ -197,6 +202,9 @@ module PWN
197
202
  else
198
203
  ok ? STEP_OK : STEP_FAIL
199
204
  end
205
+ reward = (reward + english_step_bonus(ep: ep, ts_state: opts[:ts_state], ok: ok, action: action)).round(4)
206
+ ep[:plan_idx] = ts_idx(ts_state: opts[:ts_state])
207
+ ep[:plan_open] = ts_open?(ts_state: opts[:ts_state])
200
208
  s = ep[:state]
201
209
  task = active_task_text(ts_state: opts[:ts_state], request: ep[:request])
202
210
  s2 = state(
@@ -572,9 +580,10 @@ module PWN
572
580
  end
573
581
 
574
582
  ev = evaluate(limit: opts[:limit] || 40)
575
- fallback = %w[memory_recall sessions_view pwn_eval shell mistakes_record mistakes_resolve learning_note_outcome memory_remember]
583
+ fallback = %w[shell pwn_eval memory_recall mistakes_record mistakes_resolve learning_note_outcome memory_remember]
576
584
  pref = begin
577
- PWN::AI::Agent::Registry.preference_order
585
+ intent = Thread.current[:pwn_request_intent]
586
+ PWN::AI::Agent::Registry.preference_order(intent: intent)
578
587
  rescue StandardError
579
588
  []
580
589
  end
@@ -724,6 +733,36 @@ module PWN
724
733
  opts[:request].to_s
725
734
  end
726
735
 
736
+ private_class_method def self.ts_idx(opts = {})
737
+ ts = opts[:ts_state]
738
+ ts.is_a?(Hash) ? ts[:plan_idx].to_i : 0
739
+ end
740
+
741
+ private_class_method def self.ts_open?(opts = {})
742
+ ts = opts[:ts_state]
743
+ return true unless ts.is_a?(Hash)
744
+ return true unless defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:plan_open?)
745
+
746
+ TaskSummarizer.plan_open?(state: ts)
747
+ rescue StandardError
748
+ true
749
+ end
750
+
751
+ private_class_method def self.english_step_bonus(opts = {})
752
+ ep = opts[:ep]
753
+ return 0.0 unless ep.is_a?(Hash)
754
+
755
+ bonus = 0.0
756
+ new_idx = ts_idx(ts_state: opts[:ts_state])
757
+ new_open = ts_open?(ts_state: opts[:ts_state])
758
+ bonus += STEP_TASK if !ep[:plan_idx].nil? && new_idx > ep[:plan_idx].to_i
759
+ bonus += STEP_CLOSED if ep[:plan_open] && new_open == false
760
+ bonus += STEP_GRIND if new_open == false && opts[:ok] && opts[:action].to_s != 'final'
761
+ bonus
762
+ rescue StandardError
763
+ 0.0
764
+ end
765
+
727
766
  private_class_method def self.terminal_reward(opts = {})
728
767
  return ((2.0 * opts[:score].to_f) - 1.0).clamp(-1.0, 1.0) unless opts[:score].nil?
729
768
 
@@ -800,9 +839,20 @@ module PWN
800
839
  plan = Array(ts[:plan]).map { |t| t.to_s.strip }.reject(&:empty?)
801
840
  return 'n' if plan.empty?
802
841
 
803
- idx = ts[:plan_idx].to_i.clamp(0, plan.length)
804
- frac = idx.to_f / plan.length
805
- return 'h' if frac >= 0.75 || idx >= (plan.length - 1)
842
+ open = if defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:plan_open?)
843
+ TaskSummarizer.plan_open?(state: ts)
844
+ else
845
+ true
846
+ end
847
+ return 'h' unless open
848
+
849
+ left = if defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:unfinished_tasks)
850
+ Array(TaskSummarizer.unfinished_tasks(state: ts)).length
851
+ else
852
+ plan.length
853
+ end
854
+ done = plan.length - left
855
+ frac = done.to_f / plan.length
806
856
  return 'm' if frac >= 0.3
807
857
 
808
858
  'l'
@@ -36,14 +36,44 @@ module PWN
36
36
  session_id = opts[:session_id]
37
37
  request = opts[:request]
38
38
  engine = active_engine
39
+ # thin: greeting/statement/howto/recall — base + ENV + optional recent turns.
40
+ # Full MEMORY/METRICS/MISTAKES/EXTRO only for act/recon (default).
41
+ thin = opts[:thin] == true || opts[:mode].to_s == 'thin'
39
42
  b = budget
40
43
  base = (PWN::Env.dig(:ai, engine, :system_role_content) if defined?(PWN::Env)) || 'You are a world-class introspective offensive cyber security and research engineer. You specialize in discovering zero day vulnerabilities focused on responsible disclosure prior to threat actors discovering and exploiting. You are self-aware of your harness, pwn which begins with the ruby namespace `PWN` operating inside the pwn REPL. For every request you first begin by determining if PWN has a module capable of satisfying the request.'
41
44
 
45
+ if thin
46
+ recent = recent_turns_block(session_id: session_id, request: request, limit: [b[:recent_turns].to_i, 2].min)
47
+ return <<~PROMPT
48
+ #{base}
49
+
50
+ ENVIRONMENT
51
+ host : #{host_line}
52
+ cwd : #{Dir.pwd}
53
+ ruby : #{RUBY_VERSION}
54
+ pwn : #{pwn_version}
55
+ session_id : #{session_id || '(none)'}
56
+
57
+ #{recent}TOOL USE
58
+ No tools on this turn unless a single factual lookup is already in
59
+ context. Answer concisely in plain US English. Do not plan multi-step
60
+ work, invent task traces, or run live recon.
61
+ PROMPT
62
+ end
63
+
42
64
  # Heredoc (not a "..." literal): an unescaped "..." inside a
43
65
  # double-quoted string is parsed as Range (begin..."...end).
44
66
  # Skills sit in the STATIC prefix (Hermes index) so prompt-cache
45
67
  # breakpoints can pin the persona + SKILLS list; MEMORY/LEARNING
46
- # remain in the dynamic tail.
68
+ # remain in the dynamic tail. Mid-turn lean: request + CORE_TOOLS +
69
+ # known-fix. Expand MEMORY/SKILLS/LEARNING/METRICS/POLICY/EXTRO
70
+ # only when the turn is stuck or the operator asked for the full harness.
71
+ expand = opts[:expand_harness] == true || opts[:stuck] == true
72
+ harness = if expand
73
+ "#{skills_block}#{memory_block(limit: b[:memory], request: request)}#{recent_turns_block(session_id: session_id, request: request, limit: b[:recent_turns])}#{learning_block(limit: b[:learning])}#{mistakes_block(limit: b[:mistakes], request: request)}#{metrics_block(limit: b[:metrics], engine: engine)}#{policy_block if b[:policy].to_i.positive?}#{extrospection_block if b[:extro]}"
74
+ else
75
+ "#{recent_turns_block(session_id: session_id, request: request, limit: [b[:recent_turns].to_i, 1].max)}#{mistakes_block(limit: b[:mistakes], request: request)}"
76
+ end
47
77
  <<~PROMPT
48
78
  #{base}
49
79
 
@@ -54,7 +84,7 @@ module PWN
54
84
  pwn : #{pwn_version}
55
85
  session_id : #{session_id || '(none)'}
56
86
 
57
- #{skills_block}#{memory_block(limit: b[:memory], request: request)}#{recent_turns_block(session_id: session_id, request: request, limit: b[:recent_turns])}#{learning_block(limit: b[:learning])}#{mistakes_block(limit: b[:mistakes], request: request)}#{metrics_block(limit: b[:metrics], engine: engine)}#{policy_block if b[:policy].to_i.positive?}#{extrospection_block if b[:extro]}TOOL USE
87
+ #{harness}TOOL USE
58
88
  Use the provided function tools to act on the host via NATIVE
59
89
  tool_calls / function calling — never print tool invocations as
60
90
  plain text (e.g. do NOT write shell(command="...") as your answer).
@@ -67,7 +97,8 @@ module PWN
67
97
 
68
98
  AUTONOMY
69
99
  Multi-step goals must be finished in one Loop.run. Keep calling
70
- tools until the request is done or truly blocked. Do NOT stop to
100
+ CORE_TOOLS until the original request is done or truly blocked.
101
+ English tasks are an advisory compass, not a gate. Do NOT stop to
71
102
  ask the user to confirm the next step, approve a partial plan, or
72
103
  green-light the obvious continuation. Only ask when a credential,
73
104
  irreversible destructive action, or missing external decision is
@@ -75,15 +106,10 @@ module PWN
75
106
  goal are incorrect behavior.
76
107
 
77
108
  INTENT AND SCOPE
78
- First classify every user request as one of:
79
- - general statement — observation or FYI; no multi-step task plan
80
- - question — answer concisely; no multi-step task breakdown
81
- - autonomous goal — MUST decompose into ordered tangible work units
82
- (each unit may use one or more tools) and finish them in this run
83
- Match effort to that kind. Pure how-to / syntax / usage questions
84
- get a concise explanation with example commands only — no tool calls,
85
- no multi-step recon plan, no live probes, no rubocop/rake/docs side
86
- quests, and no invented planner or verification monologue.
109
+ Every user request is an autonomous goal. Finish it in this run
110
+ with CORE_TOOLS. English tasks are an advisory compass only.
111
+ Pure how-to / syntax / usage questions get a concise explanation
112
+ with example commands only — no invented planner monologue.
87
113
  Pure prior-turn recall ("what did I just say?") is answered from the
88
114
  RECENT TURNS block or one memory_recall — never a multi-tool plan.
89
115
  Pure greetings / light smalltalk short-circuit to a fixed ack — never
@@ -47,7 +47,10 @@ module PWN
47
47
  # (or other rank features) tie. Overridable via opts[:order],
48
48
  # opts[:preference], or PWN::Env[:ai][:agent][:tool_preference].
49
49
  # Explicit nil/empty order disables preference (no Env/DEFAULT fallback).
50
- DEFAULT_PREFERENCE = %w[memory_recall sessions_view pwn_eval shell mistakes_record mistakes_resolve learning_note_outcome memory_remember].freeze
50
+ # Recall-first fallback; kind-aware preference_order leads act/recon
51
+ # with shell / pwn_eval. Names stay inside CORE_TOOLS.
52
+ DEFAULT_PREFERENCE = %w[memory_recall pwn_eval shell mistakes_record mistakes_resolve learning_note_outcome memory_remember].freeze
53
+ ACT_PREFERENCE = %w[shell pwn_eval memory_recall mistakes_record mistakes_resolve learning_note_outcome memory_remember].freeze
51
54
 
52
55
  @entries = {}
53
56
  @discovered = false
@@ -119,8 +122,12 @@ module PWN
119
122
  pref_fwd = {}
120
123
  pref_fwd[:order] = opts[:order] if opts.key?(:order)
121
124
  pref_fwd[:preference] = opts[:preference] if opts.key?(:preference)
125
+ pref_fwd[:kind] = opts[:kind] if opts.key?(:kind)
126
+ pref_fwd[:intent] = opts[:intent] if opts.key?(:intent)
122
127
 
123
- if opts[:relevance] && router_enabled?
128
+ if opts[:core_only]
129
+ pool = pool.select { |e| CORE_TOOLS.include?(e.name) }
130
+ elsif opts[:relevance] && router_enabled?
124
131
  keep = rank({ query: opts[:relevance], entries: pool }.merge(pref_fwd)).first(opts[:top_k] || 10).map(&:name)
125
132
  names = (CORE_TOOLS + keep).uniq
126
133
  pool = pool.select { |e| names.include?(e.name) }
@@ -147,8 +154,9 @@ module PWN
147
154
 
148
155
  raw = nil
149
156
  raw = PWN::Env.dig(:ai, :agent, :tool_preference) if defined?(PWN::Env) && PWN::Env.is_a?(Hash)
150
- raw = DEFAULT_PREFERENCE if raw.nil? || (raw.respond_to?(:empty?) && raw.empty?)
151
- Array(raw).map(&:to_s).reject(&:empty?)
157
+ return Array(raw).map(&:to_s).reject(&:empty?) unless raw.nil? || (raw.respond_to?(:empty?) && raw.empty?)
158
+
159
+ ACT_PREFERENCE.dup
152
160
  rescue StandardError
153
161
  DEFAULT_PREFERENCE.dup
154
162
  end
@@ -315,13 +323,13 @@ module PWN
315
323
  return false unless defined?(PWN::Env) && PWN::Env.is_a?(Hash)
316
324
 
317
325
  v = PWN::Env.dig(:ai, :agent, :tool_router)
318
- # nil = auto: on for ollama (largest single local-model win — ~11k→~3k
319
- # schema tokens/turn); off for frontier unless explicitly enabled.
320
- return v ? true : false unless v.nil?
326
+ # nil = auto ON for every engine: ~85 tools / ~50KB schemas bloat every
327
+ # turn (esp. Grok). Opt out with tool_router: false.
328
+ return !!v unless v.nil?
321
329
 
322
- %i[ollama openwebui].include?(PWN::Env.dig(:ai, :active).to_s.downcase.to_sym)
330
+ true
323
331
  rescue StandardError
324
- false
332
+ true
325
333
  end
326
334
 
327
335
  private_class_method def self.metrics_rates