pwn 0.5.643 → 0.5.650

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -51,6 +51,10 @@ module PWN
51
51
  module Loop
52
52
  DEFAULT_MAX_ITERS = 777
53
53
  ESCALATE_AFTER_FAILS = 4
54
+ # P17 — when empty_final / known thrash shapes dominate, stop before
55
+ # burning the full ollama cap so the corpus is not pure terminal failure.
56
+ BUDGET_HARD_STOP_FAILS = 8
57
+ BUDGET_EMPTY_FINAL_STOP = 3
54
58
 
55
59
  ENGINE_MODS = {
56
60
  openai: 'PWN::AI::OpenAI',
@@ -77,6 +81,25 @@ module PWN
77
81
  { role: 'assistant', content: txt, tool_calls: [] }
78
82
  end
79
83
 
84
+ # P17 — true when unresolved agent_loop / assistant_answer budget
85
+ # fingerprints dominate Mistakes.top (the #1 live failure mode).
86
+ private_class_method def self.budget_exhaustion_hot?
87
+ return false unless defined?(Mistakes)
88
+
89
+ top = Mistakes.top(limit: 8, unresolved_only: true)
90
+ return false if top.empty?
91
+
92
+ budgetish = top.count do |m|
93
+ t = m[:tool].to_s
94
+ e = m[:error].to_s.downcase
95
+ t == 'agent_loop' || t == 'assistant_answer' ||
96
+ e.include?('budget exhausted') || e.include?('iteration budget')
97
+ end
98
+ budgetish >= 2 || (budgetish >= 1 && top.first[:tool].to_s == 'agent_loop')
99
+ rescue StandardError
100
+ false
101
+ end
102
+
80
103
  private_class_method def self.max_iters
81
104
  v = (PWN::Env.dig(:ai, :agent, :max_iters) if defined?(PWN::Env))
82
105
  n = v.to_i.positive? ? v.to_i : DEFAULT_MAX_ITERS
@@ -87,6 +110,18 @@ module PWN
87
110
  # shrink the tool budget so thrash can't compound on bad plans.
88
111
  cal = calibration_state
89
112
  n = [n, cal[:max_iters_cap]].min if cal[:overconfident]
113
+ # P17 — tighter default when agent_loop budget_exhaustion dominates
114
+ # open mistakes: finish-under-N is the skill gap, not more thrash.
115
+ # Caps are intentionally harsh (8 local / 12 remote): the open
116
+ # fingerprint is "iteration budget exhausted" ×N; more headroom
117
+ # only produces more empty terminal failures for ORM/PRM/DPO.
118
+ if budget_exhaustion_hot?
119
+ # Remote/overconfident thrash (this host): 12 was still burning full
120
+ # 12-step tool plans. Cap tighter when calibration also says overconfident.
121
+ base = active_engine == :ollama ? 8 : 12
122
+ base = [base, 8].min if cal[:overconfident]
123
+ n = [n, base].min
124
+ end
90
125
  n
91
126
  rescue StandardError
92
127
  DEFAULT_MAX_ITERS
@@ -233,22 +268,43 @@ module PWN
233
268
  # without leaking to the user.
234
269
  private_class_method def self.plan_first(opts = {})
235
270
  messages = opts[:messages]
271
+ # P17 — under budget_exhaustion_hot the plan must be ultra-short:
272
+ # red_team_plan is a nested agent loop and was the #1 amplifier of
273
+ # iteration-budget exhaustion on this host (together with CF).
274
+ hot = begin
275
+ budget_exhaustion_hot?
276
+ rescue StandardError
277
+ false
278
+ end
279
+ plan_prompt = if hot
280
+ 'Before acting: write AT MOST 3 numbered tool calls (name + key args) that finish the ask. Prefer fewer. LAST line: "p(success)=<0.0-1.0>". Reply ONLY with the plan + that line — no tools, no prose.'
281
+ else
282
+ 'Before acting: (1) list the exact tool calls (name + key args) you will make, in order; (2) on the LAST line write "p(success)=<0.0-1.0>". Reply ONLY with the numbered plan + that line — do NOT call any tool yet.'
283
+ end
236
284
  plan_msg = call_engine(
237
- messages: messages + [{ role: 'user',
238
- content: 'Before acting: (1) list the exact tool calls (name + key args) you will make, in order; (2) on the LAST line write "p(success)=<0.0-1.0>". Reply ONLY with the numbered plan + that line — do NOT call any tool yet.' }],
285
+ messages: messages + [{ role: 'user', content: plan_prompt }],
239
286
  tools: nil
240
287
  )
241
288
  return nil unless plan_msg && !plan_msg[:content].to_s.strip.empty?
242
289
 
243
290
  plan = plan_msg[:content].to_s.strip
244
291
  messages << { role: 'assistant', content: "PLAN:\n#{plan}" }
245
- # S4 — adversarial plan review grounded in THIS host's telemetry
246
- if defined?(Curriculum)
292
+ # S4 — adversarial plan review grounded in THIS host's telemetry.
293
+ # P17 — never fork red_team when budget fingerprints dominate: it is
294
+ # another mini agent loop and compounds iteration-budget exhaustion.
295
+ if defined?(Curriculum) && !hot
247
296
  rt = Curriculum.red_team_plan(request: opts[:request], plan: plan)
248
297
  messages << { role: 'user', content: rt } if rt
249
298
  end
250
- # W3 — extract predicted p(success) for calibration tracking
251
- plan[/p\(success\)\s*=\s*([01](?:\.\d+)?)/i, 1]&.to_f
299
+ # W3/P22 — extract predicted p(success) for calibration tracking.
300
+ # Accept p(success)=0.7 | p(success) = .7 | confidence=0.7 on last lines.
301
+ predicted = plan[/p\(\s*success\s*\)\s*=\s*([01]?(?:\.\d+)?)/i, 1]&.to_f
302
+ predicted = plan[/\bconfidence\s*=\s*([01]?(?:\.\d+)?)/i, 1]&.to_f if predicted.nil?
303
+ predicted = predicted.clamp(0.0, 1.0) if predicted
304
+ # Stash so auto_introspect / recover can always see it even if the
305
+ # return value is dropped by a caller rescue.
306
+ Thread.current[:pwn_plan_predicted] = predicted
307
+ predicted
252
308
  rescue StandardError => e
253
309
  warn "[pwn-ai/loop] plan_first swallowed: #{e.class}: #{e.message}"
254
310
  nil
@@ -498,9 +554,20 @@ module PWN
498
554
  append_session(session_id: session_id, role: 'user', content: request)
499
555
 
500
556
  predicted = nil
557
+ Thread.current[:pwn_plan_predicted] = nil
501
558
  cal_state = calibration_state
502
559
  force_plan = cal_state[:force_plan]
503
- predicted = plan_first(messages: messages, request: request) if (force_plan || agent_flag(key: :plan_first, default: local)) && !Array(tools).empty?
560
+ if (force_plan || agent_flag(key: :plan_first, default: local) || budget_exhaustion_hot?) && !Array(tools).empty?
561
+ predicted = plan_first(messages: messages, request: request)
562
+ # P22 — prefer explicit return; fall back to thread stash
563
+ predicted = Thread.current[:pwn_plan_predicted] if predicted.nil?
564
+ end
565
+ if budget_exhaustion_hot?
566
+ messages << {
567
+ role: 'user',
568
+ content: '[pwn-ai/p17] Budget-exhaustion is the top open failure on this host. Prefer the SHORTEST plan that finishes the ask (≤3 tool calls). Emit a final answer as soon as you have evidence — do not explore.'
569
+ }
570
+ end
504
571
  if force_plan && cal_state[:cal]
505
572
  messages << {
506
573
  role: 'user',
@@ -517,7 +584,25 @@ module PWN
517
584
  # 3.1 — compact history on local so tool dumps don't fill num_ctx
518
585
  compact_history!(messages: messages) if local
519
586
 
520
- msg = call_engine(messages: messages, tools: tools)
587
+ # P17 — on the final iteration, strip tools and demand a plain-text
588
+ # answer. Without this the model happily emits one more tool_calls
589
+ # batch, burns the last slot, and lands on budget_exhausted with
590
+ # nothing the user (or ORM) can use.
591
+ # P17 deepen — when budget_hot, force text-only on the LAST TWO
592
+ # iters so a final tool_calls batch cannot burn the terminal slot.
593
+ text_only_iters = budget_exhaustion_hot? ? 2 : 1
594
+ last_iter = (i >= max_iters - text_only_iters)
595
+ if last_iter
596
+ tag = i >= max_iters - 1 ? 'FINAL ITERATION' : 'PENULTIMATE — wrap up'
597
+ messages << {
598
+ role: 'user',
599
+ content: "[pwn-ai/p17] #{tag} — do NOT call any more tools. " \
600
+ 'Write the best answer you can from evidence already in this ' \
601
+ 'transcript. If blocked, say what failed and the next single step.'
602
+ }
603
+ end
604
+
605
+ msg = call_engine(messages: messages, tools: last_iter ? nil : tools)
521
606
  return '[pwn-ai] engine returned no message' if msg.nil?
522
607
 
523
608
  calls = Array(msg[:tool_calls])
@@ -567,7 +652,10 @@ module PWN
567
652
  # alt-persona branch, judge both, inject the winner. Real
568
653
  # advantage estimation; (loser, winner) → DPO preference.
569
654
  thresh = defined?(Mistakes) ? Mistakes::REPEAT_THRESHOLD : 3
570
- if count >= thresh && !escalated && defined?(Curriculum)
655
+ # P17 — never fork counterfactual when budget fingerprints dominate:
656
+ # CF is another mini agent loop and is the #1 amplifier of
657
+ # iteration-budget exhaustion on this host.
658
+ if count >= thresh && !escalated && defined?(Curriculum) && !budget_exhaustion_hot?
571
659
  cf = (turn_fails["cf:#{fkey}"] += 1) == 1 ? Curriculum.counterfactual(request: request, name: name, args: args, error: tele[:err] || raw[0, 200], hint: hint) : nil
572
660
  hint = "#{hint}\n[pwn-ai/counterfactual] branch #{cf[:branch]} (score=#{cf[:score].round(2)}): #{cf[:content]}" if cf
573
661
  end
@@ -589,6 +677,31 @@ module PWN
589
677
  )
590
678
  end
591
679
 
680
+ # P17 — hard stop: empty-final thrash or cumulative fails past cap.
681
+ # Prefer a short apologetic final over another 10 useless tool dumps
682
+ # that poison ORM/PRM/DPO with terminal failures.
683
+ empty_n = turn_fails['empty_final'].to_i
684
+ fail_n = turn_fails.values.sum
685
+ if empty_n >= BUDGET_EMPTY_FINAL_STOP || fail_n >= BUDGET_HARD_STOP_FAILS
686
+ msg = if empty_n >= BUDGET_EMPTY_FINAL_STOP
687
+ '[pwn-ai] stopped: repeated empty finals (budget thrash guard)'
688
+ else
689
+ '[pwn-ai] stopped: too many in-turn failures (budget thrash guard)'
690
+ end
691
+ if defined?(Mistakes)
692
+ Mistakes.record(
693
+ tool: 'agent_loop',
694
+ error: "budget thrash guard fired empty=#{empty_n} fails=#{fail_n} iter=#{i}",
695
+ session_id: session_id,
696
+ source: :loop,
697
+ shape: :budget_thrash
698
+ )
699
+ end
700
+ append_session(session_id: session_id, role: 'assistant', content: msg)
701
+ Learning.auto_introspect(session_id: session_id, request: request, final: msg, predicted: predicted) if defined?(Learning) && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: i)
702
+ return msg
703
+ end
704
+
592
705
  next unless local && !escalated && turn_fails.values.sum >= ESCALATE_AFTER_FAILS
593
706
 
594
707
  hint = escalate(request: request, turn_fails: turn_fails, session_id: session_id)
@@ -599,8 +712,29 @@ module PWN
599
712
  escalated = true
600
713
  end
601
714
 
602
- Mistakes.record(tool: 'agent_loop', error: 'iteration budget exhausted without a final answer', session_id: session_id, source: :loop) if defined?(Mistakes)
603
- '[pwn-ai] iteration budget exhausted'
715
+ # P17 — exhaust path must still feed Learning so ORM/PRM/HER see the
716
+ # failure (previously we only Mistakes.record'd and returned a bare
717
+ # string — no session row, no judge, no hindsight).
718
+ final_msg = '[pwn-ai] iteration budget exhausted'
719
+ if defined?(Mistakes)
720
+ Mistakes.record(
721
+ tool: 'agent_loop',
722
+ error: 'iteration budget exhausted without a final answer',
723
+ session_id: session_id,
724
+ source: :loop,
725
+ shape: :budget_exhausted
726
+ )
727
+ end
728
+ append_session(session_id: session_id, role: 'assistant', content: final_msg)
729
+ if defined?(Learning) && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: max_iters)
730
+ Learning.auto_introspect(
731
+ session_id: session_id,
732
+ request: request,
733
+ final: final_msg,
734
+ predicted: predicted
735
+ )
736
+ end
737
+ final_msg
604
738
  end
605
739
 
606
740
  # Author(s):: 0day Inc. <support@0dayinc.com>
@@ -122,6 +122,8 @@ module PWN
122
122
  name: name.to_s,
123
123
  calls: calls,
124
124
  success_rate: rate,
125
+ judge_rate: judge_rate(name: name),
126
+ effective_rate: effective_rate(name: name),
125
127
  avg_duration: avg,
126
128
  last_error: b[:last_error],
127
129
  last_at: b[:last_at]
@@ -149,15 +151,38 @@ module PWN
149
151
  scope = engine.to_s.empty? ? 'historical' : "engine=#{engine}"
150
152
  scope = "#{scope}, proxy_distrust=#{distrust.round(2)}" if distrust.positive?
151
153
  lines = rows.map do |r|
154
+ # P20 — display effective_rate (judge-blended) when available;
155
+ # fall back to distrust haircut on raw proxy.
152
156
  rate = r[:success_rate].to_f
153
- # blend toward 0.5 (uninformative) proportional to distrust
154
- adj = rate - ((rate - 0.5) * distrust)
157
+ eff = r[:effective_rate]
158
+ adj = if eff
159
+ eff.to_f
160
+ else
161
+ rate - ((rate - 0.5) * distrust)
162
+ end
155
163
  err = r[:last_error] ? " last_err=#{r[:last_error][0, 60]}" : ''
156
- tag = distrust.positive? ? ' (adj)' : ''
157
- " - #{r[:name]}: calls=#{r[:calls]} success=#{(adj * 100).round(1)}%#{tag} avg=#{r[:avg_duration]}s#{err}"
164
+ jtag = r[:judge_rate] ? " judge=#{(r[:judge_rate].to_f * 100).round(1)}%" : ''
165
+ tag = distrust.positive? || r[:judge_rate] ? ' (adj)' : ''
166
+ " - #{r[:name]}: calls=#{r[:calls]} success=#{(adj * 100).round(1)}%#{tag}#{jtag} avg=#{r[:avg_duration]}s#{err}"
158
167
  end
159
168
  warn_line = distrust.positive? ? "WARNING: reward proxy diverges from judge — success rates haircut by distrust=#{distrust.round(2)}; prefer judge-scored exemplars over raw rates.\n" : ''
160
- "#{warn_line}TOOL EFFECTIVENESS (#{scope}, adapt tool choice accordingly)\n#{lines.join("\n")}\n\n"
169
+ # P0 ops — surface W1 generator_mix when diet is unhealthy so the
170
+ # online controller (and the model) prefer underfilled sources.
171
+ mix_line = ''
172
+ if defined?(Reward) && Reward.respond_to?(:generator_mix)
173
+ begin
174
+ m = Reward.generator_mix
175
+ unless m[:healthy]
176
+ mix_line = "W1 MIX: n=#{m[:n]} traj=#{m[:trajectory_fraction]} " \
177
+ "urgent=#{Array(m[:urgent]).join(',')} " \
178
+ "suppress=#{Array(m[:suppress]).join(',')} " \
179
+ "rec=#{m[:recommendation]}\n"
180
+ end
181
+ rescue StandardError
182
+ mix_line = ''
183
+ end
184
+ end
185
+ "#{warn_line}#{mix_line}TOOL EFFECTIVENESS (#{scope}, adapt tool choice accordingly)\n#{lines.join("\n")}\n\n"
161
186
  end
162
187
 
163
188
  # P4 helper — Registry.rank calls this so β·advantage is scaled down
@@ -176,7 +201,8 @@ module PWN
176
201
  t = data[name.to_sym] || blank_bucket
177
202
  n = [t[:calls].to_f, 1.0].max
178
203
  total = [data.values.sum { |v| v[:calls].to_f }, 1.0].max
179
- mean = t[:ok].to_f / n
204
+ # P20 — mean from effective_rate (judge-blended when distrust high)
205
+ mean = effective_rate(name: name)
180
206
  mean + (c * Math.sqrt(Math.log(total) / n))
181
207
  rescue StandardError
182
208
  1.0
@@ -190,7 +216,21 @@ module PWN
190
216
 
191
217
  public_class_method def self.thompson(opts = {})
192
218
  t = (load[:tools] || {})[opts[:name].to_s.to_sym] || blank_bucket
193
- beta_sample(alpha: t[:ok].to_f + 1.0, beta: t[:fail].to_f + 1.0)
219
+ # P20 — when judge samples exist and distrust > 0, tilt Beta toward
220
+ # judge_rate so Thompson explore/exploit tracks ORM not handler-ok.
221
+ ok = t[:ok].to_f
222
+ fail = t[:fail].to_f
223
+ jn = Array(t[:judge_window]).length
224
+ if jn >= 3 && proxy_trust < 0.95
225
+ jr = judge_rate(name: opts[:name]).to_f
226
+ # pseudo-counts from judge window, mixed by distrust
227
+ d = (1.0 - proxy_trust).clamp(0.0, 1.0)
228
+ jok = jr * jn
229
+ jfail = (1.0 - jr) * jn
230
+ ok = (ok * (1.0 - d)) + (jok * d)
231
+ fail = (fail * (1.0 - d)) + (jfail * d)
232
+ end
233
+ beta_sample(alpha: ok + 1.0, beta: fail + 1.0)
194
234
  rescue StandardError
195
235
  0.5
196
236
  end
@@ -205,14 +245,160 @@ module PWN
205
245
  t = data[opts[:name].to_s.to_sym]
206
246
  return 0.0 unless t
207
247
 
208
- global = data.values.sum { |v| v[:ok].to_f } / [data.values.sum { |v| v[:calls].to_f }, 1.0].max
209
- win = Array(t[:window])
210
- local = win.empty? ? (t[:ok].to_f / [t[:calls].to_f, 1.0].max) : (win.sum.to_f / win.length)
248
+ # P20 — local/global from effective_rate so bandit tracks judge
249
+ # when the handler-ok proxy is hacked (proxy_distrust high).
250
+ local = effective_rate(name: opts[:name])
251
+ rates = data.keys.map { |k| effective_rate(name: k) }
252
+ global = rates.empty? ? 0.5 : (rates.sum / rates.length)
211
253
  (local - global).round(3)
212
254
  rescue StandardError
213
255
  0.0
214
256
  end
215
257
 
258
+ # P18/P2 — rolling mean step_reward advantage for a tool.
259
+ # Sample-efficiency gate: < PRM_MIN_N samples → 0 (no rank noise).
260
+ # Shrinkage: adv *= min(1, n/PRM_FULL_N) so sparse signal cannot
261
+ # dominate UCB. Zero-variance windows (all +1 or all -1 from a
262
+ # single session) damp to 0.5×.
263
+ PRM_MIN_N = 5
264
+ PRM_FULL_N = 20
265
+
266
+ public_class_method def self.prm_advantage(opts = {})
267
+ data = load[:tools] || {}
268
+ t = data[opts[:name].to_s.to_sym]
269
+ return 0.0 unless t
270
+
271
+ win = Array(t[:prm_window])
272
+ n = win.length
273
+ return 0.0 if n < PRM_MIN_N
274
+
275
+ mean = win.sum.to_f / n
276
+ globals = data.values.map { |v| Array(v[:prm_window]) }.select { |w| w.length >= PRM_MIN_N }
277
+ gmean = if globals.empty?
278
+ 0.0
279
+ else
280
+ all = globals.flatten
281
+ all.sum.to_f / all.length
282
+ end
283
+ adv = mean - gmean
284
+ # shrinkage toward 0 until PRM_FULL_N
285
+ shrink = [n.to_f / PRM_FULL_N, 1.0].min
286
+ # variance damp: if all equal, halve influence
287
+ uniq = win.uniq
288
+ var_damp = uniq.length <= 1 ? 0.5 : 1.0
289
+ (adv * shrink * var_damp).round(3)
290
+ rescue StandardError
291
+ 0.0
292
+ end
293
+
294
+ public_class_method def self.prm_n(opts = {})
295
+ t = (load[:tools] || {})[opts[:name].to_s.to_sym]
296
+ return 0 unless t
297
+
298
+ Array(t[:prm_window]).length
299
+ rescue StandardError
300
+ 0
301
+ end
302
+
303
+ # P18 — called by Reward.prm after session annotate so live routing
304
+ # can bias toward tools that recently advanced goals (+1 step_reward).
305
+ public_class_method def self.record_step_reward(opts = {})
306
+ name = opts[:name].to_s
307
+ return if name.empty?
308
+
309
+ rew = opts[:reward].to_f.clamp(-1.0, 1.0)
310
+ m = load
311
+ m[:tools] ||= {}
312
+ t = m[:tools][name.to_sym] ||= blank_bucket
313
+ t[:prm_window] = (Array(t[:prm_window]) + [rew]).last(40)
314
+ t[:prm_sum] = t[:prm_window].sum
315
+ t[:prm_n] = t[:prm_window].length
316
+ save(metrics: m)
317
+ t
318
+ rescue StandardError
319
+ nil
320
+ end
321
+
322
+ # P20 — fold ORM judge (0..1) into per-tool telemetry so UCB /
323
+ # Thompson / advantage can prefer judge-grounded rates over the
324
+ # inflated handler-ok proxy when proxy_distrust is high.
325
+ public_class_method def self.record_judge(opts = {})
326
+ name = opts[:name].to_s
327
+ return if name.empty?
328
+
329
+ score = opts[:score].to_f.clamp(0.0, 1.0)
330
+ conf = opts.key?(:confidence) ? opts[:confidence].to_f.clamp(0.0, 1.0) : 0.7
331
+ m = load
332
+ m[:tools] ||= {}
333
+ t = m[:tools][name.to_sym] ||= blank_bucket
334
+ t[:judge_window] = (Array(t[:judge_window]) + [score]).last(40)
335
+ t[:judge_conf_window] = (Array(t[:judge_conf_window]) + [conf]).last(40)
336
+ t[:judge_sum] = t[:judge_window].sum.to_f
337
+ t[:judge_n] = t[:judge_window].length
338
+ save(metrics: m)
339
+ t
340
+ rescue StandardError
341
+ nil
342
+ end
343
+
344
+ # P1 — mean judge confidence for a tool (nil when no samples).
345
+ public_class_method def self.judge_confidence(opts = {})
346
+ t = (load[:tools] || {})[opts[:name].to_s.to_sym]
347
+ return nil unless t
348
+
349
+ win = Array(t[:judge_conf_window])
350
+ return nil if win.empty?
351
+
352
+ (win.sum.to_f / win.length).round(3)
353
+ rescue StandardError
354
+ nil
355
+ end
356
+
357
+ # Mean judge score for a tool (nil when no ORM samples yet).
358
+ public_class_method def self.judge_rate(opts = {})
359
+ t = (load[:tools] || {})[opts[:name].to_s.to_sym]
360
+ return nil unless t
361
+
362
+ win = Array(t[:judge_window])
363
+ return nil if win.empty?
364
+
365
+ (win.sum.to_f / win.length).round(3)
366
+ rescue StandardError
367
+ nil
368
+ end
369
+
370
+ # Blended success rate: when proxy_distrust > 0 and judge samples
371
+ # exist, mix judge_rate into the handler-ok rate. distrust=1 → pure
372
+ # judge (or 0.5 if no judge data). distrust=0 → pure proxy.
373
+ public_class_method def self.effective_rate(opts = {})
374
+ name = opts[:name].to_s
375
+ data = load[:tools] || {}
376
+ t = data[name.to_sym]
377
+ return 0.5 unless t
378
+
379
+ calls = [t[:calls].to_f, 1.0].max
380
+ proxy = t[:ok].to_f / calls
381
+ win = Array(t[:window])
382
+ proxy = win.sum.to_f / win.length if win.length >= 3
383
+ distrust = 1.0 - proxy_trust
384
+ jrate = judge_rate(name: name)
385
+ if distrust > 0.05 && !jrate.nil?
386
+ # P1 — scale judge weight by judge confidence so a thin local
387
+ # heuristic ORM cannot fully replace proxy when distrust is high.
388
+ # effective_distrust = distrust * mean(confidence), floor 0.15 when
389
+ # we do have judge samples so the signal still moves the needle.
390
+ jconf = judge_confidence(name: name)
391
+ jconf = 0.7 if jconf.nil?
392
+ eff_d = (distrust * jconf).clamp(0.0, 1.0)
393
+ eff_d = [eff_d, 0.15].max if jconf >= 0.3
394
+ ((proxy * (1.0 - eff_d)) + (jrate * eff_d)).clamp(0.0, 1.0).round(3)
395
+ else
396
+ proxy.round(3)
397
+ end
398
+ rescue StandardError
399
+ 0.5
400
+ end
401
+
216
402
  # Supported Method Parameters::
217
403
  # cps = PWN::AI::Agent::Metrics.changepoints
218
404
  #
@@ -360,7 +546,10 @@ module PWN
360
546
  PWN::AI::Agent::Metrics.to_context(limit: 8, engine: :ollama) # injected by PromptBuilder
361
547
  PWN::AI::Agent::Metrics.ucb(name: 'shell') # C1 exploration bonus
362
548
  PWN::AI::Agent::Metrics.thompson(name: 'shell') # C1 Beta(ok+1,fail+1) sample
363
- PWN::AI::Agent::Metrics.advantage(name: 'shell') # C1 local − global
549
+ PWN::AI::Agent::Metrics.advantage(name: 'shell') # C1 local − global (P20 judge-blended)
550
+ PWN::AI::Agent::Metrics.record_judge(name: 'shell', score: 0.8) # P20 ORM→metrics
551
+ PWN::AI::Agent::Metrics.judge_rate(name: 'shell') # P20 mean ORM
552
+ PWN::AI::Agent::Metrics.effective_rate(name: 'shell') # P20 proxy⋈judge
364
553
  PWN::AI::Agent::Metrics.changepoints(within_secs: 3600) # E1 CUSUM regime changes
365
554
  PWN::AI::Agent::Metrics.record_calibration(predicted: 0.8, actual: 1.0, brier: 0.04, engine: :ollama)
366
555
  PWN::AI::Agent::Metrics.calibration(engine: :ollama) # W3 Brier / overconfidence
@@ -257,15 +257,33 @@ module PWN
257
257
  importance: 0.9
258
258
  )
259
259
  end
260
- # W1 — every resolve is a naturally-generated preference pair:
261
- # (rejected: the failing action, chosen: the fix).
260
+ # W1/P9 — every resolve is a preference pair. Prefer structured
261
+ # winning_trace (+ strategy/tool) over first-line fix prose so DPO
262
+ # learns tool trajectories, not commentary.
262
263
  if defined?(Reward)
263
- Reward.record_preference(
264
- prompt: "#{store[key][:tool]}: #{store[key][:error]}",
265
- rejected: store[key][:snippet].to_s,
266
- chosen: fix.strip,
267
- source: :mistakes_resolve
268
- )
264
+ sf = store[key][:structured_fix] || {}
265
+ trace = sf[:winning_trace].to_s.strip
266
+ strat = [sf[:strategy], sf[:tool], sf[:args_template]].compact.map(&:to_s).reject(&:empty?).join(' | ')
267
+ # P21/P25 — only write W1 pairs when we have a real winning_trace.
268
+ # Prose-only resolve still updates Memory lesson + structured_fix;
269
+ # it must NOT flood DPO with fix commentary (shape: :fix_prose).
270
+ if trace.length >= 40
271
+ parts = []
272
+ parts << "STRATEGY: #{strat}" unless strat.empty?
273
+ parts << "WINNING_TRACE:\n#{trace[0, 3_500]}"
274
+ parts << "FIX: #{fix.strip[0, 400]}"
275
+ chosen = parts.join("\n")
276
+ rejected = store[key][:snippet].to_s
277
+ rejected = "FAILING: tool=#{store[key][:tool]} err=#{store[key][:error]}" if rejected.strip.empty?
278
+ Reward.record_preference(
279
+ prompt: "#{store[key][:tool]}: #{store[key][:error]}",
280
+ rejected: rejected,
281
+ chosen: chosen,
282
+ source: :mistakes_resolve,
283
+ shape: :winning_trace,
284
+ meta: { signature: sig, strategy: sf[:strategy], tool: sf[:tool] }.compact
285
+ )
286
+ end
269
287
  end
270
288
  store[key]
271
289
  end
@@ -136,7 +136,7 @@ module PWN
136
136
 
137
137
  tokens = query.scan(/[a-z0-9_]{3,}/).uniq
138
138
  # C1 — advantage-weighted router:
139
- # score = α·keyword_sim + β·advantage + γ·UCB(tool)
139
+ # score = α·keyword_sim + β·advantage + γ·UCB(tool) + δ·prm_advantage
140
140
  # UCB gives untried / low-N tools an exploration bonus so a single
141
141
  # early failure (before its dep was installed) does not blacklist
142
142
  # it forever; advantage prefers tools that outperform the fleet.
@@ -145,12 +145,34 @@ module PWN
145
145
  trust = defined?(Metrics) && Metrics.respond_to?(:proxy_trust) ? Metrics.proxy_trust : 1.0
146
146
  beta = 0.3 * trust
147
147
  gamma = 0.2
148
+ # P18/P2 — PRM step_reward advantage closes R2 into the controller.
149
+ # Sample-efficiency: if fewer than 3 tools have prm_n≥PRM_MIN_N,
150
+ # drop delta to 0 so sparse PRM cannot inject rank variance.
151
+ # Otherwise scale delta by fleet coverage fraction.
152
+ prm_ready = 0
153
+ if defined?(Metrics) && Metrics.respond_to?(:prm_n)
154
+ prm_ready = entries.count do |e|
155
+ Metrics.prm_n(name: e.name).to_i >= begin
156
+ Metrics::PRM_MIN_N
157
+ rescue StandardError
158
+ 5
159
+ end
160
+ end
161
+ end
162
+ fleet = [entries.length, 1].max
163
+ coverage = prm_ready.to_f / fleet
164
+ delta = if prm_ready < 3
165
+ 0.0
166
+ else
167
+ 0.25 * trust * [coverage / 0.3, 1.0].min
168
+ end
148
169
  scored = entries.map do |e|
149
170
  hay = "#{e.name} #{e.toolset} #{e.schema[:description]} #{Array(e.schema.dig(:parameters, :properties)&.keys).join(' ')}".downcase
150
171
  sim = tokens.count { |t| hay.include?(t) }
151
172
  adv = defined?(Metrics) && Metrics.respond_to?(:advantage) ? Metrics.advantage(name: e.name) : 0.0
152
173
  ucb = defined?(Metrics) && Metrics.respond_to?(:ucb) ? Metrics.ucb(name: e.name) : 0.5
153
- [e, sim, (alpha * sim) + (beta * adv) + (gamma * ucb)]
174
+ prm = defined?(Metrics) && Metrics.respond_to?(:prm_advantage) ? Metrics.prm_advantage(name: e.name) : 0.0
175
+ [e, sim, (alpha * sim) + (beta * adv) + (gamma * ucb) + (delta * prm)]
154
176
  end
155
177
  scored.reject { |_, sim, _| sim.zero? }
156
178
  .sort_by { |_, _, s| -s }