pwn 0.5.643 → 0.5.650
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/documentation/Agent-Tool-Registry.md +10 -5
- data/documentation/Cron.md +3 -2
- data/documentation/Home.md +2 -2
- data/documentation/How-PWN-Works.md +1 -1
- data/documentation/Mistakes.md +13 -0
- data/documentation/Reinforcement-Learning.md +72 -3
- data/documentation/Skills-Memory-Learning.md +5 -3
- data/documentation/What-is-PWN.md +1 -1
- data/documentation/diagrams/dot/pwn-ai-feedback-learning-loop.dot +3 -3
- data/documentation/diagrams/dot/reinforcement-learning.dot +8 -8
- data/documentation/diagrams/pwn-ai-feedback-learning-loop.svg +318 -319
- data/documentation/diagrams/reinforcement-learning.svg +188 -187
- data/documentation/pwn-ai-Agent.md +6 -5
- data/lib/pwn/ai/agent/curriculum.rb +491 -35
- data/lib/pwn/ai/agent/extrospection.rb +18 -4
- data/lib/pwn/ai/agent/learning.rb +308 -54
- data/lib/pwn/ai/agent/loop.rb +145 -11
- data/lib/pwn/ai/agent/metrics.rb +200 -11
- data/lib/pwn/ai/agent/mistakes.rb +26 -8
- data/lib/pwn/ai/agent/registry.rb +24 -2
- data/lib/pwn/ai/agent/reward.rb +447 -7
- data/lib/pwn/ai/agent/tools/curriculum.rb +23 -0
- data/lib/pwn/ai/agent/tools/reward.rb +86 -0
- data/lib/pwn/ai/agent/tools/ruby_eval.rb +29 -6
- data/lib/pwn/cron.rb +14 -2
- data/lib/pwn/version.rb +1 -1
- data/spec/integration/reinforced_feedback_loop_spec.rb +806 -12
- data/spec/lib/pwn/ai/agent/reward_spec.rb +64 -1
- data/spec/lib/pwn/ai/agent/tools/ruby_eval_spec.rb +39 -1
- data/third_party/pwn_rdoc.jsonl +33 -4
- metadata +1 -1
data/lib/pwn/ai/agent/loop.rb
CHANGED
|
@@ -51,6 +51,10 @@ module PWN
|
|
|
51
51
|
module Loop
|
|
52
52
|
DEFAULT_MAX_ITERS = 777
|
|
53
53
|
ESCALATE_AFTER_FAILS = 4
|
|
54
|
+
# P17 — when empty_final / known thrash shapes dominate, stop before
|
|
55
|
+
# burning the full ollama cap so the corpus is not pure terminal failure.
|
|
56
|
+
BUDGET_HARD_STOP_FAILS = 8
|
|
57
|
+
BUDGET_EMPTY_FINAL_STOP = 3
|
|
54
58
|
|
|
55
59
|
ENGINE_MODS = {
|
|
56
60
|
openai: 'PWN::AI::OpenAI',
|
|
@@ -77,6 +81,25 @@ module PWN
|
|
|
77
81
|
{ role: 'assistant', content: txt, tool_calls: [] }
|
|
78
82
|
end
|
|
79
83
|
|
|
84
|
+
# P17 — true when unresolved agent_loop / assistant_answer budget
|
|
85
|
+
# fingerprints dominate Mistakes.top (the #1 live failure mode).
|
|
86
|
+
private_class_method def self.budget_exhaustion_hot?
|
|
87
|
+
return false unless defined?(Mistakes)
|
|
88
|
+
|
|
89
|
+
top = Mistakes.top(limit: 8, unresolved_only: true)
|
|
90
|
+
return false if top.empty?
|
|
91
|
+
|
|
92
|
+
budgetish = top.count do |m|
|
|
93
|
+
t = m[:tool].to_s
|
|
94
|
+
e = m[:error].to_s.downcase
|
|
95
|
+
t == 'agent_loop' || t == 'assistant_answer' ||
|
|
96
|
+
e.include?('budget exhausted') || e.include?('iteration budget')
|
|
97
|
+
end
|
|
98
|
+
budgetish >= 2 || (budgetish >= 1 && top.first[:tool].to_s == 'agent_loop')
|
|
99
|
+
rescue StandardError
|
|
100
|
+
false
|
|
101
|
+
end
|
|
102
|
+
|
|
80
103
|
private_class_method def self.max_iters
|
|
81
104
|
v = (PWN::Env.dig(:ai, :agent, :max_iters) if defined?(PWN::Env))
|
|
82
105
|
n = v.to_i.positive? ? v.to_i : DEFAULT_MAX_ITERS
|
|
@@ -87,6 +110,18 @@ module PWN
|
|
|
87
110
|
# shrink the tool budget so thrash can't compound on bad plans.
|
|
88
111
|
cal = calibration_state
|
|
89
112
|
n = [n, cal[:max_iters_cap]].min if cal[:overconfident]
|
|
113
|
+
# P17 — tighter default when agent_loop budget_exhaustion dominates
|
|
114
|
+
# open mistakes: finish-under-N is the skill gap, not more thrash.
|
|
115
|
+
# Caps are intentionally harsh (8 local / 12 remote): the open
|
|
116
|
+
# fingerprint is "iteration budget exhausted" ×N; more headroom
|
|
117
|
+
# only produces more empty terminal failures for ORM/PRM/DPO.
|
|
118
|
+
if budget_exhaustion_hot?
|
|
119
|
+
# Remote/overconfident thrash (this host): 12 was still burning full
|
|
120
|
+
# 12-step tool plans. Cap tighter when calibration also says overconfident.
|
|
121
|
+
base = active_engine == :ollama ? 8 : 12
|
|
122
|
+
base = [base, 8].min if cal[:overconfident]
|
|
123
|
+
n = [n, base].min
|
|
124
|
+
end
|
|
90
125
|
n
|
|
91
126
|
rescue StandardError
|
|
92
127
|
DEFAULT_MAX_ITERS
|
|
@@ -233,22 +268,43 @@ module PWN
|
|
|
233
268
|
# without leaking to the user.
|
|
234
269
|
private_class_method def self.plan_first(opts = {})
|
|
235
270
|
messages = opts[:messages]
|
|
271
|
+
# P17 — under budget_exhaustion_hot the plan must be ultra-short:
|
|
272
|
+
# red_team_plan is a nested agent loop and was the #1 amplifier of
|
|
273
|
+
# iteration-budget exhaustion on this host (together with CF).
|
|
274
|
+
hot = begin
|
|
275
|
+
budget_exhaustion_hot?
|
|
276
|
+
rescue StandardError
|
|
277
|
+
false
|
|
278
|
+
end
|
|
279
|
+
plan_prompt = if hot
|
|
280
|
+
'Before acting: write AT MOST 3 numbered tool calls (name + key args) that finish the ask. Prefer fewer. LAST line: "p(success)=<0.0-1.0>". Reply ONLY with the plan + that line — no tools, no prose.'
|
|
281
|
+
else
|
|
282
|
+
'Before acting: (1) list the exact tool calls (name + key args) you will make, in order; (2) on the LAST line write "p(success)=<0.0-1.0>". Reply ONLY with the numbered plan + that line — do NOT call any tool yet.'
|
|
283
|
+
end
|
|
236
284
|
plan_msg = call_engine(
|
|
237
|
-
messages: messages + [{ role: 'user',
|
|
238
|
-
content: 'Before acting: (1) list the exact tool calls (name + key args) you will make, in order; (2) on the LAST line write "p(success)=<0.0-1.0>". Reply ONLY with the numbered plan + that line — do NOT call any tool yet.' }],
|
|
285
|
+
messages: messages + [{ role: 'user', content: plan_prompt }],
|
|
239
286
|
tools: nil
|
|
240
287
|
)
|
|
241
288
|
return nil unless plan_msg && !plan_msg[:content].to_s.strip.empty?
|
|
242
289
|
|
|
243
290
|
plan = plan_msg[:content].to_s.strip
|
|
244
291
|
messages << { role: 'assistant', content: "PLAN:\n#{plan}" }
|
|
245
|
-
# S4 — adversarial plan review grounded in THIS host's telemetry
|
|
246
|
-
|
|
292
|
+
# S4 — adversarial plan review grounded in THIS host's telemetry.
|
|
293
|
+
# P17 — never fork red_team when budget fingerprints dominate: it is
|
|
294
|
+
# another mini agent loop and compounds iteration-budget exhaustion.
|
|
295
|
+
if defined?(Curriculum) && !hot
|
|
247
296
|
rt = Curriculum.red_team_plan(request: opts[:request], plan: plan)
|
|
248
297
|
messages << { role: 'user', content: rt } if rt
|
|
249
298
|
end
|
|
250
|
-
# W3 — extract predicted p(success) for calibration tracking
|
|
251
|
-
|
|
299
|
+
# W3/P22 — extract predicted p(success) for calibration tracking.
|
|
300
|
+
# Accept p(success)=0.7 | p(success) = .7 | confidence=0.7 on last lines.
|
|
301
|
+
predicted = plan[/p\(\s*success\s*\)\s*=\s*([01]?(?:\.\d+)?)/i, 1]&.to_f
|
|
302
|
+
predicted = plan[/\bconfidence\s*=\s*([01]?(?:\.\d+)?)/i, 1]&.to_f if predicted.nil?
|
|
303
|
+
predicted = predicted.clamp(0.0, 1.0) if predicted
|
|
304
|
+
# Stash so auto_introspect / recover can always see it even if the
|
|
305
|
+
# return value is dropped by a caller rescue.
|
|
306
|
+
Thread.current[:pwn_plan_predicted] = predicted
|
|
307
|
+
predicted
|
|
252
308
|
rescue StandardError => e
|
|
253
309
|
warn "[pwn-ai/loop] plan_first swallowed: #{e.class}: #{e.message}"
|
|
254
310
|
nil
|
|
@@ -498,9 +554,20 @@ module PWN
|
|
|
498
554
|
append_session(session_id: session_id, role: 'user', content: request)
|
|
499
555
|
|
|
500
556
|
predicted = nil
|
|
557
|
+
Thread.current[:pwn_plan_predicted] = nil
|
|
501
558
|
cal_state = calibration_state
|
|
502
559
|
force_plan = cal_state[:force_plan]
|
|
503
|
-
|
|
560
|
+
if (force_plan || agent_flag(key: :plan_first, default: local) || budget_exhaustion_hot?) && !Array(tools).empty?
|
|
561
|
+
predicted = plan_first(messages: messages, request: request)
|
|
562
|
+
# P22 — prefer explicit return; fall back to thread stash
|
|
563
|
+
predicted = Thread.current[:pwn_plan_predicted] if predicted.nil?
|
|
564
|
+
end
|
|
565
|
+
if budget_exhaustion_hot?
|
|
566
|
+
messages << {
|
|
567
|
+
role: 'user',
|
|
568
|
+
content: '[pwn-ai/p17] Budget-exhaustion is the top open failure on this host. Prefer the SHORTEST plan that finishes the ask (≤3 tool calls). Emit a final answer as soon as you have evidence — do not explore.'
|
|
569
|
+
}
|
|
570
|
+
end
|
|
504
571
|
if force_plan && cal_state[:cal]
|
|
505
572
|
messages << {
|
|
506
573
|
role: 'user',
|
|
@@ -517,7 +584,25 @@ module PWN
|
|
|
517
584
|
# 3.1 — compact history on local so tool dumps don't fill num_ctx
|
|
518
585
|
compact_history!(messages: messages) if local
|
|
519
586
|
|
|
520
|
-
|
|
587
|
+
# P17 — on the final iteration, strip tools and demand a plain-text
|
|
588
|
+
# answer. Without this the model happily emits one more tool_calls
|
|
589
|
+
# batch, burns the last slot, and lands on budget_exhausted with
|
|
590
|
+
# nothing the user (or ORM) can use.
|
|
591
|
+
# P17 deepen — when budget_hot, force text-only on the LAST TWO
|
|
592
|
+
# iters so a final tool_calls batch cannot burn the terminal slot.
|
|
593
|
+
text_only_iters = budget_exhaustion_hot? ? 2 : 1
|
|
594
|
+
last_iter = (i >= max_iters - text_only_iters)
|
|
595
|
+
if last_iter
|
|
596
|
+
tag = i >= max_iters - 1 ? 'FINAL ITERATION' : 'PENULTIMATE — wrap up'
|
|
597
|
+
messages << {
|
|
598
|
+
role: 'user',
|
|
599
|
+
content: "[pwn-ai/p17] #{tag} — do NOT call any more tools. " \
|
|
600
|
+
'Write the best answer you can from evidence already in this ' \
|
|
601
|
+
'transcript. If blocked, say what failed and the next single step.'
|
|
602
|
+
}
|
|
603
|
+
end
|
|
604
|
+
|
|
605
|
+
msg = call_engine(messages: messages, tools: last_iter ? nil : tools)
|
|
521
606
|
return '[pwn-ai] engine returned no message' if msg.nil?
|
|
522
607
|
|
|
523
608
|
calls = Array(msg[:tool_calls])
|
|
@@ -567,7 +652,10 @@ module PWN
|
|
|
567
652
|
# alt-persona branch, judge both, inject the winner. Real
|
|
568
653
|
# advantage estimation; (loser, winner) → DPO preference.
|
|
569
654
|
thresh = defined?(Mistakes) ? Mistakes::REPEAT_THRESHOLD : 3
|
|
570
|
-
|
|
655
|
+
# P17 — never fork counterfactual when budget fingerprints dominate:
|
|
656
|
+
# CF is another mini agent loop and is the #1 amplifier of
|
|
657
|
+
# iteration-budget exhaustion on this host.
|
|
658
|
+
if count >= thresh && !escalated && defined?(Curriculum) && !budget_exhaustion_hot?
|
|
571
659
|
cf = (turn_fails["cf:#{fkey}"] += 1) == 1 ? Curriculum.counterfactual(request: request, name: name, args: args, error: tele[:err] || raw[0, 200], hint: hint) : nil
|
|
572
660
|
hint = "#{hint}\n[pwn-ai/counterfactual] branch #{cf[:branch]} (score=#{cf[:score].round(2)}): #{cf[:content]}" if cf
|
|
573
661
|
end
|
|
@@ -589,6 +677,31 @@ module PWN
|
|
|
589
677
|
)
|
|
590
678
|
end
|
|
591
679
|
|
|
680
|
+
# P17 — hard stop: empty-final thrash or cumulative fails past cap.
|
|
681
|
+
# Prefer a short apologetic final over another 10 useless tool dumps
|
|
682
|
+
# that poison ORM/PRM/DPO with terminal failures.
|
|
683
|
+
empty_n = turn_fails['empty_final'].to_i
|
|
684
|
+
fail_n = turn_fails.values.sum
|
|
685
|
+
if empty_n >= BUDGET_EMPTY_FINAL_STOP || fail_n >= BUDGET_HARD_STOP_FAILS
|
|
686
|
+
msg = if empty_n >= BUDGET_EMPTY_FINAL_STOP
|
|
687
|
+
'[pwn-ai] stopped: repeated empty finals (budget thrash guard)'
|
|
688
|
+
else
|
|
689
|
+
'[pwn-ai] stopped: too many in-turn failures (budget thrash guard)'
|
|
690
|
+
end
|
|
691
|
+
if defined?(Mistakes)
|
|
692
|
+
Mistakes.record(
|
|
693
|
+
tool: 'agent_loop',
|
|
694
|
+
error: "budget thrash guard fired empty=#{empty_n} fails=#{fail_n} iter=#{i}",
|
|
695
|
+
session_id: session_id,
|
|
696
|
+
source: :loop,
|
|
697
|
+
shape: :budget_thrash
|
|
698
|
+
)
|
|
699
|
+
end
|
|
700
|
+
append_session(session_id: session_id, role: 'assistant', content: msg)
|
|
701
|
+
Learning.auto_introspect(session_id: session_id, request: request, final: msg, predicted: predicted) if defined?(Learning) && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: i)
|
|
702
|
+
return msg
|
|
703
|
+
end
|
|
704
|
+
|
|
592
705
|
next unless local && !escalated && turn_fails.values.sum >= ESCALATE_AFTER_FAILS
|
|
593
706
|
|
|
594
707
|
hint = escalate(request: request, turn_fails: turn_fails, session_id: session_id)
|
|
@@ -599,8 +712,29 @@ module PWN
|
|
|
599
712
|
escalated = true
|
|
600
713
|
end
|
|
601
714
|
|
|
602
|
-
|
|
603
|
-
'
|
|
715
|
+
# P17 — exhaust path must still feed Learning so ORM/PRM/HER see the
|
|
716
|
+
# failure (previously we only Mistakes.record'd and returned a bare
|
|
717
|
+
# string — no session row, no judge, no hindsight).
|
|
718
|
+
final_msg = '[pwn-ai] iteration budget exhausted'
|
|
719
|
+
if defined?(Mistakes)
|
|
720
|
+
Mistakes.record(
|
|
721
|
+
tool: 'agent_loop',
|
|
722
|
+
error: 'iteration budget exhausted without a final answer',
|
|
723
|
+
session_id: session_id,
|
|
724
|
+
source: :loop,
|
|
725
|
+
shape: :budget_exhausted
|
|
726
|
+
)
|
|
727
|
+
end
|
|
728
|
+
append_session(session_id: session_id, role: 'assistant', content: final_msg)
|
|
729
|
+
if defined?(Learning) && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: max_iters)
|
|
730
|
+
Learning.auto_introspect(
|
|
731
|
+
session_id: session_id,
|
|
732
|
+
request: request,
|
|
733
|
+
final: final_msg,
|
|
734
|
+
predicted: predicted
|
|
735
|
+
)
|
|
736
|
+
end
|
|
737
|
+
final_msg
|
|
604
738
|
end
|
|
605
739
|
|
|
606
740
|
# Author(s):: 0day Inc. <support@0dayinc.com>
|
data/lib/pwn/ai/agent/metrics.rb
CHANGED
|
@@ -122,6 +122,8 @@ module PWN
|
|
|
122
122
|
name: name.to_s,
|
|
123
123
|
calls: calls,
|
|
124
124
|
success_rate: rate,
|
|
125
|
+
judge_rate: judge_rate(name: name),
|
|
126
|
+
effective_rate: effective_rate(name: name),
|
|
125
127
|
avg_duration: avg,
|
|
126
128
|
last_error: b[:last_error],
|
|
127
129
|
last_at: b[:last_at]
|
|
@@ -149,15 +151,38 @@ module PWN
|
|
|
149
151
|
scope = engine.to_s.empty? ? 'historical' : "engine=#{engine}"
|
|
150
152
|
scope = "#{scope}, proxy_distrust=#{distrust.round(2)}" if distrust.positive?
|
|
151
153
|
lines = rows.map do |r|
|
|
154
|
+
# P20 — display effective_rate (judge-blended) when available;
|
|
155
|
+
# fall back to distrust haircut on raw proxy.
|
|
152
156
|
rate = r[:success_rate].to_f
|
|
153
|
-
|
|
154
|
-
adj
|
|
157
|
+
eff = r[:effective_rate]
|
|
158
|
+
adj = if eff
|
|
159
|
+
eff.to_f
|
|
160
|
+
else
|
|
161
|
+
rate - ((rate - 0.5) * distrust)
|
|
162
|
+
end
|
|
155
163
|
err = r[:last_error] ? " last_err=#{r[:last_error][0, 60]}" : ''
|
|
156
|
-
|
|
157
|
-
|
|
164
|
+
jtag = r[:judge_rate] ? " judge=#{(r[:judge_rate].to_f * 100).round(1)}%" : ''
|
|
165
|
+
tag = distrust.positive? || r[:judge_rate] ? ' (adj)' : ''
|
|
166
|
+
" - #{r[:name]}: calls=#{r[:calls]} success=#{(adj * 100).round(1)}%#{tag}#{jtag} avg=#{r[:avg_duration]}s#{err}"
|
|
158
167
|
end
|
|
159
168
|
warn_line = distrust.positive? ? "WARNING: reward proxy diverges from judge — success rates haircut by distrust=#{distrust.round(2)}; prefer judge-scored exemplars over raw rates.\n" : ''
|
|
160
|
-
|
|
169
|
+
# P0 ops — surface W1 generator_mix when diet is unhealthy so the
|
|
170
|
+
# online controller (and the model) prefer underfilled sources.
|
|
171
|
+
mix_line = ''
|
|
172
|
+
if defined?(Reward) && Reward.respond_to?(:generator_mix)
|
|
173
|
+
begin
|
|
174
|
+
m = Reward.generator_mix
|
|
175
|
+
unless m[:healthy]
|
|
176
|
+
mix_line = "W1 MIX: n=#{m[:n]} traj=#{m[:trajectory_fraction]} " \
|
|
177
|
+
"urgent=#{Array(m[:urgent]).join(',')} " \
|
|
178
|
+
"suppress=#{Array(m[:suppress]).join(',')} " \
|
|
179
|
+
"rec=#{m[:recommendation]}\n"
|
|
180
|
+
end
|
|
181
|
+
rescue StandardError
|
|
182
|
+
mix_line = ''
|
|
183
|
+
end
|
|
184
|
+
end
|
|
185
|
+
"#{warn_line}#{mix_line}TOOL EFFECTIVENESS (#{scope}, adapt tool choice accordingly)\n#{lines.join("\n")}\n\n"
|
|
161
186
|
end
|
|
162
187
|
|
|
163
188
|
# P4 helper — Registry.rank calls this so β·advantage is scaled down
|
|
@@ -176,7 +201,8 @@ module PWN
|
|
|
176
201
|
t = data[name.to_sym] || blank_bucket
|
|
177
202
|
n = [t[:calls].to_f, 1.0].max
|
|
178
203
|
total = [data.values.sum { |v| v[:calls].to_f }, 1.0].max
|
|
179
|
-
mean
|
|
204
|
+
# P20 — mean from effective_rate (judge-blended when distrust high)
|
|
205
|
+
mean = effective_rate(name: name)
|
|
180
206
|
mean + (c * Math.sqrt(Math.log(total) / n))
|
|
181
207
|
rescue StandardError
|
|
182
208
|
1.0
|
|
@@ -190,7 +216,21 @@ module PWN
|
|
|
190
216
|
|
|
191
217
|
public_class_method def self.thompson(opts = {})
|
|
192
218
|
t = (load[:tools] || {})[opts[:name].to_s.to_sym] || blank_bucket
|
|
193
|
-
|
|
219
|
+
# P20 — when judge samples exist and distrust > 0, tilt Beta toward
|
|
220
|
+
# judge_rate so Thompson explore/exploit tracks ORM not handler-ok.
|
|
221
|
+
ok = t[:ok].to_f
|
|
222
|
+
fail = t[:fail].to_f
|
|
223
|
+
jn = Array(t[:judge_window]).length
|
|
224
|
+
if jn >= 3 && proxy_trust < 0.95
|
|
225
|
+
jr = judge_rate(name: opts[:name]).to_f
|
|
226
|
+
# pseudo-counts from judge window, mixed by distrust
|
|
227
|
+
d = (1.0 - proxy_trust).clamp(0.0, 1.0)
|
|
228
|
+
jok = jr * jn
|
|
229
|
+
jfail = (1.0 - jr) * jn
|
|
230
|
+
ok = (ok * (1.0 - d)) + (jok * d)
|
|
231
|
+
fail = (fail * (1.0 - d)) + (jfail * d)
|
|
232
|
+
end
|
|
233
|
+
beta_sample(alpha: ok + 1.0, beta: fail + 1.0)
|
|
194
234
|
rescue StandardError
|
|
195
235
|
0.5
|
|
196
236
|
end
|
|
@@ -205,14 +245,160 @@ module PWN
|
|
|
205
245
|
t = data[opts[:name].to_s.to_sym]
|
|
206
246
|
return 0.0 unless t
|
|
207
247
|
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
local
|
|
248
|
+
# P20 — local/global from effective_rate so bandit tracks judge
|
|
249
|
+
# when the handler-ok proxy is hacked (proxy_distrust high).
|
|
250
|
+
local = effective_rate(name: opts[:name])
|
|
251
|
+
rates = data.keys.map { |k| effective_rate(name: k) }
|
|
252
|
+
global = rates.empty? ? 0.5 : (rates.sum / rates.length)
|
|
211
253
|
(local - global).round(3)
|
|
212
254
|
rescue StandardError
|
|
213
255
|
0.0
|
|
214
256
|
end
|
|
215
257
|
|
|
258
|
+
# P18/P2 — rolling mean step_reward advantage for a tool.
|
|
259
|
+
# Sample-efficiency gate: < PRM_MIN_N samples → 0 (no rank noise).
|
|
260
|
+
# Shrinkage: adv *= min(1, n/PRM_FULL_N) so sparse signal cannot
|
|
261
|
+
# dominate UCB. Zero-variance windows (all +1 or all -1 from a
|
|
262
|
+
# single session) damp to 0.5×.
|
|
263
|
+
PRM_MIN_N = 5
|
|
264
|
+
PRM_FULL_N = 20
|
|
265
|
+
|
|
266
|
+
public_class_method def self.prm_advantage(opts = {})
|
|
267
|
+
data = load[:tools] || {}
|
|
268
|
+
t = data[opts[:name].to_s.to_sym]
|
|
269
|
+
return 0.0 unless t
|
|
270
|
+
|
|
271
|
+
win = Array(t[:prm_window])
|
|
272
|
+
n = win.length
|
|
273
|
+
return 0.0 if n < PRM_MIN_N
|
|
274
|
+
|
|
275
|
+
mean = win.sum.to_f / n
|
|
276
|
+
globals = data.values.map { |v| Array(v[:prm_window]) }.select { |w| w.length >= PRM_MIN_N }
|
|
277
|
+
gmean = if globals.empty?
|
|
278
|
+
0.0
|
|
279
|
+
else
|
|
280
|
+
all = globals.flatten
|
|
281
|
+
all.sum.to_f / all.length
|
|
282
|
+
end
|
|
283
|
+
adv = mean - gmean
|
|
284
|
+
# shrinkage toward 0 until PRM_FULL_N
|
|
285
|
+
shrink = [n.to_f / PRM_FULL_N, 1.0].min
|
|
286
|
+
# variance damp: if all equal, halve influence
|
|
287
|
+
uniq = win.uniq
|
|
288
|
+
var_damp = uniq.length <= 1 ? 0.5 : 1.0
|
|
289
|
+
(adv * shrink * var_damp).round(3)
|
|
290
|
+
rescue StandardError
|
|
291
|
+
0.0
|
|
292
|
+
end
|
|
293
|
+
|
|
294
|
+
public_class_method def self.prm_n(opts = {})
|
|
295
|
+
t = (load[:tools] || {})[opts[:name].to_s.to_sym]
|
|
296
|
+
return 0 unless t
|
|
297
|
+
|
|
298
|
+
Array(t[:prm_window]).length
|
|
299
|
+
rescue StandardError
|
|
300
|
+
0
|
|
301
|
+
end
|
|
302
|
+
|
|
303
|
+
# P18 — called by Reward.prm after session annotate so live routing
|
|
304
|
+
# can bias toward tools that recently advanced goals (+1 step_reward).
|
|
305
|
+
public_class_method def self.record_step_reward(opts = {})
|
|
306
|
+
name = opts[:name].to_s
|
|
307
|
+
return if name.empty?
|
|
308
|
+
|
|
309
|
+
rew = opts[:reward].to_f.clamp(-1.0, 1.0)
|
|
310
|
+
m = load
|
|
311
|
+
m[:tools] ||= {}
|
|
312
|
+
t = m[:tools][name.to_sym] ||= blank_bucket
|
|
313
|
+
t[:prm_window] = (Array(t[:prm_window]) + [rew]).last(40)
|
|
314
|
+
t[:prm_sum] = t[:prm_window].sum
|
|
315
|
+
t[:prm_n] = t[:prm_window].length
|
|
316
|
+
save(metrics: m)
|
|
317
|
+
t
|
|
318
|
+
rescue StandardError
|
|
319
|
+
nil
|
|
320
|
+
end
|
|
321
|
+
|
|
322
|
+
# P20 — fold ORM judge (0..1) into per-tool telemetry so UCB /
|
|
323
|
+
# Thompson / advantage can prefer judge-grounded rates over the
|
|
324
|
+
# inflated handler-ok proxy when proxy_distrust is high.
|
|
325
|
+
public_class_method def self.record_judge(opts = {})
|
|
326
|
+
name = opts[:name].to_s
|
|
327
|
+
return if name.empty?
|
|
328
|
+
|
|
329
|
+
score = opts[:score].to_f.clamp(0.0, 1.0)
|
|
330
|
+
conf = opts.key?(:confidence) ? opts[:confidence].to_f.clamp(0.0, 1.0) : 0.7
|
|
331
|
+
m = load
|
|
332
|
+
m[:tools] ||= {}
|
|
333
|
+
t = m[:tools][name.to_sym] ||= blank_bucket
|
|
334
|
+
t[:judge_window] = (Array(t[:judge_window]) + [score]).last(40)
|
|
335
|
+
t[:judge_conf_window] = (Array(t[:judge_conf_window]) + [conf]).last(40)
|
|
336
|
+
t[:judge_sum] = t[:judge_window].sum.to_f
|
|
337
|
+
t[:judge_n] = t[:judge_window].length
|
|
338
|
+
save(metrics: m)
|
|
339
|
+
t
|
|
340
|
+
rescue StandardError
|
|
341
|
+
nil
|
|
342
|
+
end
|
|
343
|
+
|
|
344
|
+
# P1 — mean judge confidence for a tool (nil when no samples).
|
|
345
|
+
public_class_method def self.judge_confidence(opts = {})
|
|
346
|
+
t = (load[:tools] || {})[opts[:name].to_s.to_sym]
|
|
347
|
+
return nil unless t
|
|
348
|
+
|
|
349
|
+
win = Array(t[:judge_conf_window])
|
|
350
|
+
return nil if win.empty?
|
|
351
|
+
|
|
352
|
+
(win.sum.to_f / win.length).round(3)
|
|
353
|
+
rescue StandardError
|
|
354
|
+
nil
|
|
355
|
+
end
|
|
356
|
+
|
|
357
|
+
# Mean judge score for a tool (nil when no ORM samples yet).
|
|
358
|
+
public_class_method def self.judge_rate(opts = {})
|
|
359
|
+
t = (load[:tools] || {})[opts[:name].to_s.to_sym]
|
|
360
|
+
return nil unless t
|
|
361
|
+
|
|
362
|
+
win = Array(t[:judge_window])
|
|
363
|
+
return nil if win.empty?
|
|
364
|
+
|
|
365
|
+
(win.sum.to_f / win.length).round(3)
|
|
366
|
+
rescue StandardError
|
|
367
|
+
nil
|
|
368
|
+
end
|
|
369
|
+
|
|
370
|
+
# Blended success rate: when proxy_distrust > 0 and judge samples
|
|
371
|
+
# exist, mix judge_rate into the handler-ok rate. distrust=1 → pure
|
|
372
|
+
# judge (or 0.5 if no judge data). distrust=0 → pure proxy.
|
|
373
|
+
public_class_method def self.effective_rate(opts = {})
|
|
374
|
+
name = opts[:name].to_s
|
|
375
|
+
data = load[:tools] || {}
|
|
376
|
+
t = data[name.to_sym]
|
|
377
|
+
return 0.5 unless t
|
|
378
|
+
|
|
379
|
+
calls = [t[:calls].to_f, 1.0].max
|
|
380
|
+
proxy = t[:ok].to_f / calls
|
|
381
|
+
win = Array(t[:window])
|
|
382
|
+
proxy = win.sum.to_f / win.length if win.length >= 3
|
|
383
|
+
distrust = 1.0 - proxy_trust
|
|
384
|
+
jrate = judge_rate(name: name)
|
|
385
|
+
if distrust > 0.05 && !jrate.nil?
|
|
386
|
+
# P1 — scale judge weight by judge confidence so a thin local
|
|
387
|
+
# heuristic ORM cannot fully replace proxy when distrust is high.
|
|
388
|
+
# effective_distrust = distrust * mean(confidence), floor 0.15 when
|
|
389
|
+
# we do have judge samples so the signal still moves the needle.
|
|
390
|
+
jconf = judge_confidence(name: name)
|
|
391
|
+
jconf = 0.7 if jconf.nil?
|
|
392
|
+
eff_d = (distrust * jconf).clamp(0.0, 1.0)
|
|
393
|
+
eff_d = [eff_d, 0.15].max if jconf >= 0.3
|
|
394
|
+
((proxy * (1.0 - eff_d)) + (jrate * eff_d)).clamp(0.0, 1.0).round(3)
|
|
395
|
+
else
|
|
396
|
+
proxy.round(3)
|
|
397
|
+
end
|
|
398
|
+
rescue StandardError
|
|
399
|
+
0.5
|
|
400
|
+
end
|
|
401
|
+
|
|
216
402
|
# Supported Method Parameters::
|
|
217
403
|
# cps = PWN::AI::Agent::Metrics.changepoints
|
|
218
404
|
#
|
|
@@ -360,7 +546,10 @@ module PWN
|
|
|
360
546
|
PWN::AI::Agent::Metrics.to_context(limit: 8, engine: :ollama) # injected by PromptBuilder
|
|
361
547
|
PWN::AI::Agent::Metrics.ucb(name: 'shell') # C1 exploration bonus
|
|
362
548
|
PWN::AI::Agent::Metrics.thompson(name: 'shell') # C1 Beta(ok+1,fail+1) sample
|
|
363
|
-
PWN::AI::Agent::Metrics.advantage(name: 'shell') # C1 local − global
|
|
549
|
+
PWN::AI::Agent::Metrics.advantage(name: 'shell') # C1 local − global (P20 judge-blended)
|
|
550
|
+
PWN::AI::Agent::Metrics.record_judge(name: 'shell', score: 0.8) # P20 ORM→metrics
|
|
551
|
+
PWN::AI::Agent::Metrics.judge_rate(name: 'shell') # P20 mean ORM
|
|
552
|
+
PWN::AI::Agent::Metrics.effective_rate(name: 'shell') # P20 proxy⋈judge
|
|
364
553
|
PWN::AI::Agent::Metrics.changepoints(within_secs: 3600) # E1 CUSUM regime changes
|
|
365
554
|
PWN::AI::Agent::Metrics.record_calibration(predicted: 0.8, actual: 1.0, brier: 0.04, engine: :ollama)
|
|
366
555
|
PWN::AI::Agent::Metrics.calibration(engine: :ollama) # W3 Brier / overconfidence
|
|
@@ -257,15 +257,33 @@ module PWN
|
|
|
257
257
|
importance: 0.9
|
|
258
258
|
)
|
|
259
259
|
end
|
|
260
|
-
# W1 — every resolve is a
|
|
261
|
-
# (
|
|
260
|
+
# W1/P9 — every resolve is a preference pair. Prefer structured
|
|
261
|
+
# winning_trace (+ strategy/tool) over first-line fix prose so DPO
|
|
262
|
+
# learns tool trajectories, not commentary.
|
|
262
263
|
if defined?(Reward)
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
)
|
|
264
|
+
sf = store[key][:structured_fix] || {}
|
|
265
|
+
trace = sf[:winning_trace].to_s.strip
|
|
266
|
+
strat = [sf[:strategy], sf[:tool], sf[:args_template]].compact.map(&:to_s).reject(&:empty?).join(' | ')
|
|
267
|
+
# P21/P25 — only write W1 pairs when we have a real winning_trace.
|
|
268
|
+
# Prose-only resolve still updates Memory lesson + structured_fix;
|
|
269
|
+
# it must NOT flood DPO with fix commentary (shape: :fix_prose).
|
|
270
|
+
if trace.length >= 40
|
|
271
|
+
parts = []
|
|
272
|
+
parts << "STRATEGY: #{strat}" unless strat.empty?
|
|
273
|
+
parts << "WINNING_TRACE:\n#{trace[0, 3_500]}"
|
|
274
|
+
parts << "FIX: #{fix.strip[0, 400]}"
|
|
275
|
+
chosen = parts.join("\n")
|
|
276
|
+
rejected = store[key][:snippet].to_s
|
|
277
|
+
rejected = "FAILING: tool=#{store[key][:tool]} err=#{store[key][:error]}" if rejected.strip.empty?
|
|
278
|
+
Reward.record_preference(
|
|
279
|
+
prompt: "#{store[key][:tool]}: #{store[key][:error]}",
|
|
280
|
+
rejected: rejected,
|
|
281
|
+
chosen: chosen,
|
|
282
|
+
source: :mistakes_resolve,
|
|
283
|
+
shape: :winning_trace,
|
|
284
|
+
meta: { signature: sig, strategy: sf[:strategy], tool: sf[:tool] }.compact
|
|
285
|
+
)
|
|
286
|
+
end
|
|
269
287
|
end
|
|
270
288
|
store[key]
|
|
271
289
|
end
|
|
@@ -136,7 +136,7 @@ module PWN
|
|
|
136
136
|
|
|
137
137
|
tokens = query.scan(/[a-z0-9_]{3,}/).uniq
|
|
138
138
|
# C1 — advantage-weighted router:
|
|
139
|
-
# score = α·keyword_sim + β·advantage + γ·UCB(tool)
|
|
139
|
+
# score = α·keyword_sim + β·advantage + γ·UCB(tool) + δ·prm_advantage
|
|
140
140
|
# UCB gives untried / low-N tools an exploration bonus so a single
|
|
141
141
|
# early failure (before its dep was installed) does not blacklist
|
|
142
142
|
# it forever; advantage prefers tools that outperform the fleet.
|
|
@@ -145,12 +145,34 @@ module PWN
|
|
|
145
145
|
trust = defined?(Metrics) && Metrics.respond_to?(:proxy_trust) ? Metrics.proxy_trust : 1.0
|
|
146
146
|
beta = 0.3 * trust
|
|
147
147
|
gamma = 0.2
|
|
148
|
+
# P18/P2 — PRM step_reward advantage closes R2 into the controller.
|
|
149
|
+
# Sample-efficiency: if fewer than 3 tools have prm_n≥PRM_MIN_N,
|
|
150
|
+
# drop delta to 0 so sparse PRM cannot inject rank variance.
|
|
151
|
+
# Otherwise scale delta by fleet coverage fraction.
|
|
152
|
+
prm_ready = 0
|
|
153
|
+
if defined?(Metrics) && Metrics.respond_to?(:prm_n)
|
|
154
|
+
prm_ready = entries.count do |e|
|
|
155
|
+
Metrics.prm_n(name: e.name).to_i >= begin
|
|
156
|
+
Metrics::PRM_MIN_N
|
|
157
|
+
rescue StandardError
|
|
158
|
+
5
|
|
159
|
+
end
|
|
160
|
+
end
|
|
161
|
+
end
|
|
162
|
+
fleet = [entries.length, 1].max
|
|
163
|
+
coverage = prm_ready.to_f / fleet
|
|
164
|
+
delta = if prm_ready < 3
|
|
165
|
+
0.0
|
|
166
|
+
else
|
|
167
|
+
0.25 * trust * [coverage / 0.3, 1.0].min
|
|
168
|
+
end
|
|
148
169
|
scored = entries.map do |e|
|
|
149
170
|
hay = "#{e.name} #{e.toolset} #{e.schema[:description]} #{Array(e.schema.dig(:parameters, :properties)&.keys).join(' ')}".downcase
|
|
150
171
|
sim = tokens.count { |t| hay.include?(t) }
|
|
151
172
|
adv = defined?(Metrics) && Metrics.respond_to?(:advantage) ? Metrics.advantage(name: e.name) : 0.0
|
|
152
173
|
ucb = defined?(Metrics) && Metrics.respond_to?(:ucb) ? Metrics.ucb(name: e.name) : 0.5
|
|
153
|
-
|
|
174
|
+
prm = defined?(Metrics) && Metrics.respond_to?(:prm_advantage) ? Metrics.prm_advantage(name: e.name) : 0.0
|
|
175
|
+
[e, sim, (alpha * sim) + (beta * adv) + (gamma * ucb) + (delta * prm)]
|
|
154
176
|
end
|
|
155
177
|
scored.reject { |_, sim, _| sim.zero? }
|
|
156
178
|
.sort_by { |_, _, s| -s }
|