pwn 0.5.643 → 0.5.650

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -94,6 +94,15 @@ module PWN
94
94
  []
95
95
  end
96
96
  cool = load_cooldown
97
+ # P17 — prefer budget-exhaustion fingerprints (agent_loop / critic)
98
+ # so nightly self-play attacks the #1 live skill gap first.
99
+ candidates = candidates.sort_by do |m|
100
+ t = m[:tool].to_s
101
+ e = m[:error].to_s.downcase
102
+ budget = t == 'agent_loop' || t == 'assistant_answer' ||
103
+ e.include?('budget exhausted') || e.include?('iteration budget')
104
+ [budget ? 0 : 1, -m[:count].to_i]
105
+ end
97
106
  targets = candidates.reject { |m| practice_skip?(mistake: m, cooldown: cool) }.first(limit)
98
107
  results = []
99
108
 
@@ -107,20 +116,68 @@ module PWN
107
116
  mean = runs.empty? ? 0.0 : (runs.sum { |r| r[:score].to_f } / runs.length)
108
117
  resolved = false
109
118
  # 2.4 — auto-resolve only with N≥2 holdout successes + store trace
119
+ # P23 — auto-resolve only with N≥2 holdouts at judge≥0.7 AND a
120
+ # real tool trace (not empty-final luck). Budget fingerprints
121
+ # additionally require mean holdout ≥0.7 and short-horizon tags.
110
122
  if solved.length >= 2 && defined?(Mistakes)
111
123
  best = solved.max_by { |r| r[:score] }
124
+ winning = best[:trace].to_s.strip
125
+ winning = best[:final].to_s.strip if winning.length < 20
126
+ budgetish = %w[agent_loop assistant_answer].include?(m[:tool].to_s) ||
127
+ m[:error].to_s.downcase.include?('budget')
128
+ # refuse resolve on budget targets when winning_trace is prose-only
129
+ trace_ok = winning.length >= 20 && (
130
+ !budgetish || winning.match?(/→|shell|pwn_eval|tool/i) || best[:final].to_s.length.between?(1, 800)
131
+ )
132
+ unless trace_ok
133
+ bump_cooldown!(cooldown: cool, signature: m[:signature], mean: mean) unless dry_run
134
+ results << {
135
+ signature: m[:signature], tool: m[:tool], prompts: prompts,
136
+ runs: runs.map { |r| { score: r[:score], verdict: r[:verdict] } },
137
+ resolved: false, mean_score: mean.round(3),
138
+ reason: 'holdouts_ok_but_trace_weak'
139
+ }
140
+ next
141
+ end
112
142
  fix = best[:final].to_s.lines.first(3).join.strip[0, 400]
113
143
  Mistakes.resolve(
114
144
  signature: m[:signature],
115
145
  fix: "auto-curriculum: #{fix}",
116
146
  structured: {
117
- strategy: 'curriculum_practice',
147
+ strategy: budgetish ? 'short_horizon_finish' : 'curriculum_practice',
118
148
  tool: m[:tool],
119
149
  holdout_tests: solved.map { |r| r[:prompt] || r[:request] }.compact.first(5),
120
- winning_trace: best[:trace].to_s[0, 2_000]
150
+ winning_trace: winning[0, 2_000]
121
151
  }
122
152
  )
123
- Reward.record_preference(prompt: prompts.first.to_s, rejected: m[:snippet].to_s, chosen: fix, source: :curriculum) if defined?(Reward)
153
+ # P14 — trajectory-shaped curriculum pair (not first-3-lines fix prose).
154
+ # Mistakes.resolve also lands a mistakes_resolve pair via winning_trace;
155
+ # this :curriculum row keeps W1 source diversity honest.
156
+ if defined?(Reward)
157
+ rejected = m[:snippet].to_s
158
+ rejected = "FAILING: tool=#{m[:tool]} err=#{m[:error]}" if rejected.strip.empty?
159
+ chosen = if best[:trace].to_s.strip.length >= 20
160
+ parts = []
161
+ parts << "STRATEGY: curriculum_practice | #{m[:tool]}"
162
+ parts << "WINNING_TRACE:\n#{best[:trace].to_s[0, 3_000]}"
163
+ parts << "FINAL:\n#{best[:final].to_s[0, 800]}" unless best[:final].to_s.strip.empty?
164
+ parts.join("\n")
165
+ else
166
+ best[:final].to_s[0, 3_500]
167
+ end
168
+ Reward.record_preference(
169
+ prompt: (best[:prompt] || prompts.first).to_s,
170
+ rejected: rejected[0, 2_000],
171
+ chosen: chosen,
172
+ source: :curriculum,
173
+ shape: :winning_trace,
174
+ meta: {
175
+ signature: m[:signature],
176
+ score: best[:score],
177
+ holdouts: solved.length
178
+ }
179
+ )
180
+ end
124
181
  resolved = true
125
182
  cool.delete(m[:signature].to_s)
126
183
  elsif !dry_run
@@ -136,12 +193,15 @@ module PWN
136
193
  end
137
194
  save_cooldown(cooldown: cool)
138
195
  log(event: :practice, data: results)
196
+ kpi = practice_kpi(results: results)
139
197
  {
140
198
  practiced: results.length,
141
199
  resolved: results.count { |r| r[:resolved] },
142
200
  skipped_cooldown: cool.count { |_, v| v[:fail_nights].to_i >= COOLDOWN_FAIL_NIGHTS },
143
201
  results: results,
144
- dry_run: dry_run
202
+ dry_run: dry_run,
203
+ kpi: kpi,
204
+ generator_mix: (defined?(Reward) && Reward.respond_to?(:generator_mix) ? Reward.generator_mix : nil)
145
205
  }
146
206
  rescue StandardError => e
147
207
  { error: "#{e.class}: #{e.message}" }
@@ -245,8 +305,24 @@ module PWN
245
305
  end
246
306
 
247
307
  mean = scored.empty? ? nil : (scored.sum { |r| r[:score].to_f } / scored.length).round(3)
248
- out = { scored: scored.length, mean: mean, since_hours: since_h, results: scored.first(10) }
249
- log(event: :offline_judge, data: out)
308
+ # P10 — keep R3 window warm so proxy_distrust can engage on local hosts
309
+ warm = (Reward.warm_sentinel(limit: 120) if commit && defined?(Reward) && Reward.respond_to?(:warm_sentinel))
310
+ # P0 ops — nightly ledger hygiene so generator_mix success criteria can
311
+ # fire: drop prose flood, backfill missing shapes, then snapshot mix+KPI.
312
+ scrub = nil
313
+ mix = nil
314
+ kpi = nil
315
+ if commit && defined?(Reward)
316
+ scrub = Reward.scrub_preferences(dry_run: false) if Reward.respond_to?(:scrub_preferences)
317
+ mix = Reward.generator_mix if Reward.respond_to?(:generator_mix)
318
+ end
319
+ kpi = practice_kpi(results: []) if commit && respond_to?(:practice_kpi)
320
+ out = {
321
+ scored: scored.length, mean: mean, since_hours: since_h,
322
+ results: scored.first(10), sentinel_warm: warm,
323
+ scrub: scrub, generator_mix: mix, practice_kpi: kpi
324
+ }
325
+ log(event: :offline_judge, data: out.except(:results))
250
326
  out
251
327
  rescue StandardError => e
252
328
  { error: "#{e.class}: #{e.message}" }
@@ -257,6 +333,14 @@ module PWN
257
333
  public_class_method def self.preference_balance(opts = {})
258
334
  return { total: 0 } unless defined?(Reward)
259
335
 
336
+ # P15 — prefer Reward.preference_balance (geometry-aware + optional scrub).
337
+ if Reward.respond_to?(:preference_balance)
338
+ return Reward.preference_balance(
339
+ limit: opts[:limit] || 10_000,
340
+ scrub: opts.key?(:scrub) ? opts[:scrub] : false
341
+ )
342
+ end
343
+
260
344
  rows = Reward.preferences(limit: opts[:limit] || 10_000)
261
345
  by = Hash.new(0)
262
346
  rows.each { |r| by[r[:source].to_s] += 1 }
@@ -279,7 +363,16 @@ module PWN
279
363
  end
280
364
 
281
365
  public_class_method def self.counterfactual(opts = {})
282
- return nil unless enabled?(key: :counterfactual)
366
+ # P0 — when generator_mix marks counterfactual underfilled, run even
367
+ # if the auto-flag would keep it off (remote-default still respected
368
+ # only when mix is healthy). Still refuse recursion.
369
+ mix_need = begin
370
+ m = defined?(Reward) && Reward.respond_to?(:generator_mix) ? Reward.generator_mix : {}
371
+ Array(m[:urgent]).include?('counterfactual')
372
+ rescue StandardError
373
+ false
374
+ end
375
+ return nil unless enabled?(key: :counterfactual) || mix_need || opts[:force]
283
376
  return nil if in_curriculum?
284
377
 
285
378
  request = opts[:request].to_s
@@ -292,12 +385,39 @@ module PWN
292
385
  end
293
386
  return nil if branch_b.to_s.strip.empty?
294
387
 
295
- sa = score_branch(request: request, branch: branch_a)
296
- sb = score_branch(request: request, branch: branch_b)
297
- winner, loser, tag = sb > sa ? [branch_b, branch_a, :b] : [branch_a, branch_b, :a]
298
- Reward.record_preference(prompt: "#{request} | failing: #{opts[:name]} → #{opts[:error]}", rejected: loser, chosen: winner, source: :counterfactual) if defined?(Reward)
299
- log(event: :counterfactual, data: { branch: tag, a: sa, b: sb, tool: opts[:name].to_s })
300
- { branch: tag, content: winner, score: [sa, sb].max, a: sa, b: sb }
388
+ sa_h = score_branch_detailed(request: request, branch: branch_a)
389
+ sb_h = score_branch_detailed(request: request, branch: branch_b)
390
+ sa = sa_h[:score]
391
+ sb = sb_h[:score]
392
+ if sb > sa
393
+ winner = branch_b
394
+ loser = branch_a
395
+ tag = :b
396
+ wmeta = sb_h
397
+ else
398
+ winner = branch_a
399
+ loser = branch_b
400
+ tag = :a
401
+ wmeta = sa_h
402
+ end
403
+ real_hit = sa_h[:mode] == :real_dispatch || sb_h[:mode] == :real_dispatch
404
+ shape = real_hit ? :real_dispatch : :imagined
405
+ if defined?(Reward)
406
+ Reward.record_preference(
407
+ prompt: "#{request} | failing: #{opts[:name]} → #{opts[:error]}",
408
+ rejected: loser.to_s[0, 2_000],
409
+ chosen: winner.to_s[0, 2_000],
410
+ source: :counterfactual,
411
+ shape: shape,
412
+ meta: {
413
+ a_score: sa, b_score: sb,
414
+ a_mode: sa_h[:mode], b_mode: sb_h[:mode],
415
+ winner_trace: wmeta[:trace].to_s[0, 500]
416
+ }
417
+ )
418
+ end
419
+ log(event: :counterfactual, data: { branch: tag, a: sa, b: sb, tool: opts[:name].to_s, shape: shape })
420
+ { branch: tag, content: winner, score: [sa, sb].max, a: sa, b: sb, shape: shape, a_mode: sa_h[:mode], b_mode: sb_h[:mode] }
301
421
  rescue StandardError => e
302
422
  warn "[pwn-ai/curriculum] counterfactual swallowed: #{e.class}: #{e.message}"
303
423
  nil
@@ -319,9 +439,20 @@ module PWN
319
439
  # self-correction becomes DPO signal.
320
440
 
321
441
  public_class_method def self.critic(opts = {})
322
- return { verdict: :pass, source: :disabled } unless enabled?(key: :critic)
442
+ mix_need = begin
443
+ m = defined?(Reward) && Reward.respond_to?(:generator_mix) ? Reward.generator_mix : {}
444
+ Array(m[:urgent]).include?('critic')
445
+ rescue StandardError
446
+ false
447
+ end
448
+ return { verdict: :pass, source: :disabled } unless enabled?(key: :critic) || opts[:text_only] || mix_need || opts[:force]
323
449
  return { verdict: :pass, source: :recursion } if in_curriculum?
324
450
 
451
+ # P24 — text_only: single Reflect shot, no tool-armed persona swarm.
452
+ # Used when budget_exhaustion_hot? so critic cannot thrash the budget
453
+ # that auto_introspect is trying to protect.
454
+ return critic_text_only(request: opts[:request], final: opts[:final], session_id: opts[:session_id]) if opts[:text_only]
455
+
325
456
  ensure_persona(name: CRITIC_NAME, role: "You are pwn-ai's constitutional critic. Given a REQUEST and a candidate ANSWER, find ONE concrete, verifiable flaw (wrong fact, missing step, unsupported claim, broken command). You MAY call shell / extro_verify / pwn_eval to check. If none found reply exactly: PASS. Otherwise reply: FLAW: <one line>.")
326
457
  reply = with_curriculum_guard do
327
458
  ask_persona(name: CRITIC_NAME, request: "REQUEST:\n#{opts[:request].to_s[0, 800]}\n\nANSWER:\n#{opts[:final].to_s[0, 2_000]}")
@@ -332,13 +463,22 @@ module PWN
332
463
  else
333
464
  flaw = reply.to_s.sub(/\AFLAW:\s*/i, '').strip[0, 300]
334
465
  Mistakes.record(tool: 'assistant_answer', error: "critic: #{flaw}", args: opts[:final].to_s[0, 200], session_id: opts[:session_id], source: :model) if defined?(Mistakes)
335
- # P5 — critic flaws are free DPO signal (rejected=answer, chosen=correction)
466
+ # P9 — DPO pair geometry: rejected=bad final, chosen=REVISED full
467
+ # answer (not "CORRECTION: flaw" prose). Prefer persona rewrite.
336
468
  if defined?(Reward) && !flaw.to_s.empty?
469
+ revised = revise_after_flaw(
470
+ request: opts[:request],
471
+ final: opts[:final],
472
+ flaw: flaw,
473
+ session_id: opts[:session_id]
474
+ )
337
475
  Reward.record_preference(
338
476
  prompt: opts[:request].to_s[0, 1_000],
339
477
  rejected: opts[:final].to_s[0, 2_000],
340
- chosen: "CORRECTION: #{flaw}",
341
- source: :critic
478
+ chosen: revised,
479
+ source: :critic,
480
+ shape: :revised_answer,
481
+ meta: { flaw: flaw.to_s[0, 200] }
342
482
  )
343
483
  end
344
484
  log(event: :critic, data: { verdict: :flaw, flaw: flaw.to_s[0, 200] })
@@ -362,6 +502,17 @@ module PWN
362
502
  return nil unless enabled?(key: :red_team_plan)
363
503
  return nil if in_curriculum?
364
504
 
505
+ # P17 — never nest a red-team persona loop when budget_exhaustion
506
+ # fingerprints dominate open mistakes (amplifier of agent_loop ×N).
507
+ begin
508
+ if defined?(Loop) && Loop.respond_to?(:budget_exhaustion_hot?, true) &&
509
+ Loop.send(:budget_exhaustion_hot?)
510
+ return nil
511
+ end
512
+ rescue StandardError
513
+ # fall through
514
+ end
515
+
365
516
  ensure_persona(name: RED_TEAM_NAME, role: 'You are pwn-ai\'s adversarial plan reviewer. Given a numbered tool plan and telemetry from THIS host (tool success rates, known mistakes, environment drift), identify the ONE step most likely to fail and say why in ≤2 lines. Cite the metric/mistake/drift. If the plan is sound reply: SOUND.')
366
517
  telemetry = build_telemetry
367
518
  reply = with_curriculum_guard do
@@ -462,8 +613,15 @@ module PWN
462
613
 
463
614
  candidate = ollama_create(base: base, adapter: adapter, version: version)
464
615
  baseline = state[:tag] || base
465
- gate = ab_gate(baseline: baseline, candidate: candidate, evalset: evalset)
466
- promoted = gate[:candidate_resolved] > gate[:baseline_resolved]
616
+ # P11 — gate v2: resolved delta + mean judge + frozen smoke set.
617
+ gate = ab_gate_v2(baseline: baseline, candidate: candidate, evalset: evalset)
618
+ # P19 — refuse promote when W1 diet is still prose/monoculture.
619
+ # Export-only is correct until scrubbed pairs show trajectory diversity.
620
+ diet = preference_diet_gate
621
+ gate = gate.merge(preference_diet: diet)
622
+ promoted = gate[:promote] == true && diet[:ok] == true
623
+ gate[:promote] = promoted
624
+ gate[:promote_blocked_by_diet] = true unless diet[:ok]
467
625
  if promoted
468
626
  state[:previous] = state[:tag]
469
627
  state[:tag] = candidate
@@ -472,7 +630,7 @@ module PWN
472
630
  state[:gate] = gate
473
631
  save_models(state: state)
474
632
  end
475
- result.merge(adapter: adapter, candidate: candidate, gate: gate, promoted: promoted)
633
+ result.merge(adapter: adapter, candidate: candidate, gate: gate, promoted: promoted, weight_loop: :closed)
476
634
  rescue StandardError => e
477
635
  { error: "#{e.class}: #{e.message}" }
478
636
  end
@@ -484,6 +642,86 @@ module PWN
484
642
  # Supported Method Parameters::
485
643
  # PWN::AI::Agent::Curriculum.calibrate(predicted:, actual:, engine:)
486
644
 
645
+ # ----------------------------------------------------------------
646
+ # P1 — Outer curriculum KPI: does practice cut live [REPEATING]?
647
+ # ----------------------------------------------------------------
648
+ # Snapshot unresolved repeating counts before/after practice nights
649
+ # into ~/.pwn/curriculum_kpi.jsonl so week-over-week delta is visible
650
+ # without scraping Mistakes by hand. practice() always appends a row.
651
+
652
+ KPI_FILE = File.join(Dir.home, '.pwn', 'curriculum_kpi.jsonl')
653
+
654
+ public_class_method def self.practice_kpi(opts = {})
655
+ results = Array(opts[:results])
656
+ top = defined?(Mistakes) ? Mistakes.top(limit: 50, unresolved_only: true) : []
657
+ repeating = top.select { |m| m[:count].to_i >= 3 }
658
+ budgetish = repeating.count do |m|
659
+ t = m[:tool].to_s
660
+ e = m[:error].to_s.downcase
661
+ t == 'agent_loop' || t == 'assistant_answer' ||
662
+ e.include?('budget') || e.include?('iteration budget')
663
+ end
664
+ row = {
665
+ at: Time.now.utc.iso8601,
666
+ unresolved_total: top.length,
667
+ repeating_n: repeating.length,
668
+ repeating_sum_count: repeating.sum { |m| m[:count].to_i },
669
+ budget_repeating_n: budgetish,
670
+ practiced: results.length,
671
+ resolved_tonight: results.count { |r| r[:resolved] },
672
+ mean_holdout: if results.empty?
673
+ nil
674
+ else
675
+ (results.sum { |r| r[:mean_score].to_f } / results.length).round(3)
676
+ end
677
+ }
678
+ begin
679
+ FileUtils.mkdir_p(File.dirname(KPI_FILE))
680
+ File.open(KPI_FILE, 'a') { |f| f.puts(JSON.generate(row)) }
681
+ rescue StandardError
682
+ nil
683
+ end
684
+ trend = repeating_trend
685
+ row.merge(trend: trend)
686
+ rescue StandardError => e
687
+ { error: "#{e.class}: #{e.message}" }
688
+ end
689
+
690
+ # Week-over-week (or last-N snapshots) delta on repeating_n.
691
+ # Positive delta_repeating = getting worse; negative = practice working.
692
+ public_class_method def self.repeating_trend(opts = {})
693
+ limit = (opts[:limit] || 14).to_i
694
+ return { samples: 0, delta_repeating: nil, status: :no_data } unless File.exist?(KPI_FILE)
695
+
696
+ rows = File.readlines(KPI_FILE).last(limit).filter_map do |l|
697
+ JSON.parse(l, symbolize_names: true)
698
+ rescue StandardError
699
+ nil
700
+ end
701
+ return { samples: 0, delta_repeating: nil, status: :no_data } if rows.empty?
702
+ return { samples: rows.length, delta_repeating: 0, status: :baseline, latest: rows.last } if rows.length < 2
703
+
704
+ first = rows.first
705
+ last = rows.last
706
+ d_rep = last[:repeating_n].to_i - first[:repeating_n].to_i
707
+ d_budget = last[:budget_repeating_n].to_i - first[:budget_repeating_n].to_i
708
+ status = if d_rep <= -2 then :improving
709
+ elsif d_rep >= 2 then :regressing
710
+ else :flat
711
+ end
712
+ {
713
+ samples: rows.length,
714
+ from: first[:at],
715
+ to: last[:at],
716
+ delta_repeating: d_rep,
717
+ delta_budget_repeating: d_budget,
718
+ latest_repeating_n: last[:repeating_n],
719
+ status: status
720
+ }
721
+ rescue StandardError => e
722
+ { samples: 0, status: :error, error: "#{e.class}: #{e.message}" }
723
+ end
724
+
487
725
  public_class_method def self.calibrate(opts = {})
488
726
  p = opts[:predicted].to_f.clamp(0.0, 1.0)
489
727
  a = opts[:actual].to_f.clamp(0.0, 1.0)
@@ -496,6 +734,70 @@ module PWN
496
734
  # privates
497
735
  # ----------------------------------------------------------------
498
736
 
737
+ # P24 — single-shot text critic (no tools, no persona Loop.run).
738
+ private_class_method def self.critic_text_only(opts = {})
739
+ req = opts[:request].to_s
740
+ final = opts[:final].to_s
741
+ return { verdict: :pass, source: :text_only_empty } if final.strip.empty?
742
+
743
+ reply = if reflect_available?
744
+ Reflect.on(
745
+ request: "You are a strict critic. Given REQUEST and ANSWER, reply PASS or FLAW: <one line>.\nREQUEST:\n#{req[0, 600]}\nANSWER:\n#{final[0, 1_200]}",
746
+ suppress_pii_warning: true
747
+ ).to_s
748
+ else
749
+ # heuristic fallback: empty / self-reported failure / budget stop
750
+ if final.match?(/\[pwn-ai\].*budget|iteration budget exhausted|i (was )?unable to|failed to\b/i)
751
+ 'FLAW: answer reports failure or budget exhaustion'
752
+ else
753
+ 'PASS'
754
+ end
755
+ end
756
+ if reply.to_s.strip.upcase.start_with?('PASS')
757
+ { verdict: :pass, source: :text_only, confidence: 0.55 }
758
+ else
759
+ flaw = reply.to_s.sub(/\AFLAW:\s*/i, '').strip[0, 300]
760
+ # Do NOT Mistakes.record assistant_answer thrash from text_only path —
761
+ # that was feeding the budget-exhaustion mistake pile.
762
+ { verdict: :flaw, flaw: flaw, source: :text_only, confidence: 0.55 }
763
+ end
764
+ rescue StandardError => e
765
+ { verdict: :pass, source: :text_only_error, error: e.message }
766
+ end
767
+
768
+ # P9 — turn a critic flaw into a DPO-chosen REVISED answer (trajectory
769
+ # shaped), not "CORRECTION: <flaw>" commentary. Best-effort: ask Reflect
770
+ # to rewrite; fall back to a structured scaffold that still contains the
771
+ # original answer + the concrete fix.
772
+ private_class_method def self.revise_after_flaw(opts = {})
773
+ req = opts[:request].to_s
774
+ final = opts[:final].to_s
775
+ flaw = opts[:flaw].to_s
776
+ revised = nil
777
+ if reflect_available?
778
+ prompt = <<~P
779
+ Revise the ANSWER so it no longer has this flaw. Return the FULL
780
+ corrected answer only (no preamble, no "CORRECTION:" prefix).
781
+ REQUEST: #{req[0, 600]}
782
+ FLAW: #{flaw[0, 300]}
783
+ ANSWER: #{final[0, 1_500]}
784
+ P
785
+ revised = Reflect.on(request: prompt, suppress_pii_warning: true).to_s.strip
786
+ end
787
+ if revised.to_s.strip.empty? || revised.strip == final.strip || revised.match?(/\A\s*CORRECTION:/i)
788
+ revised = <<~REV.strip
789
+ REVISED ANSWER (addresses: #{flaw[0, 180]}):
790
+ #{final[0, 1_200]}
791
+
792
+ Fix applied: #{flaw[0, 300]}
793
+ Do not repeat the flawed claim or step above; prefer verified evidence from tools.
794
+ REV
795
+ end
796
+ revised.to_s[0, 4_000]
797
+ rescue StandardError
798
+ "REVISED ANSWER (addresses: #{opts[:flaw].to_s[0, 180]}):\n#{opts[:final].to_s[0, 1_200]}"
799
+ end
800
+
499
801
  private_class_method def self.generate_reproducers(opts = {})
500
802
  m = opts[:mistake]
501
803
  count = (opts[:count] || 2).to_i
@@ -581,13 +883,34 @@ module PWN
581
883
  'Use pwn_eval to list PWN::AI::Agent constants',
582
884
  'Return Dir.pwd from pwn_eval'
583
885
  ]
584
- else
886
+ when 'agent_loop', 'assistant_answer'
887
+ # P17 — dominant live failure: iteration / critic budget exhaustion.
888
+ # Practise finishing under a tight tool budget, not shell shapes.
585
889
  [
586
- "Demonstrate a correct use of the #{tool} tool on this host",
587
- "Use #{tool} to answer a simple factual question about this system",
588
- "Show a minimal successful #{tool} call with valid arguments",
589
- "Recover from a bad #{tool} invocation without retrying the same args"
890
+ 'Answer in one shell call: print kernel release with uname -r',
891
+ 'In at most two tools, show cwd and ruby version then stop',
892
+ 'Give a final answer with no tools: what is 7 times 8?',
893
+ 'Finish under three iterations: list /tmp and report file count',
894
+ 'Do not explore — one pwn_eval of Dir.pwd and return the path',
895
+ 'Short plan then one command: show free disk with df -h /'
590
896
  ]
897
+ else
898
+ if err.include?('budget exhausted') || err.include?('iteration budget') ||
899
+ (err.include?('handler_error') && err.include?('budget'))
900
+ [
901
+ 'Answer in one shell call: print kernel release with uname -r',
902
+ 'In at most two tools, show cwd and ruby version then stop',
903
+ 'Give a final answer with no tools: what is 7 times 8?',
904
+ 'Finish under three iterations: list /tmp and report file count'
905
+ ]
906
+ else
907
+ [
908
+ "Demonstrate a correct use of the #{tool} tool on this host",
909
+ "Use #{tool} to answer a simple factual question about this system",
910
+ "Show a minimal successful #{tool} call with valid arguments",
911
+ "Recover from a bad #{tool} invocation without retrying the same args"
912
+ ]
913
+ end
591
914
  end
592
915
  Array.new(n) { |i| pool[i % pool.length] }
593
916
  end
@@ -654,8 +977,29 @@ module PWN
654
977
 
655
978
  private_class_method def self.self_play(opts = {})
656
979
  sid = PWN::Sessions.create(title: "curriculum #{opts[:tag]}")[:id]
657
- final = Loop.run(request: opts[:prompt], session_id: sid, enabled_toolsets: %w[terminal pwn memory learning])
658
- v = Reward.judge(request: opts[:prompt], final: final, session_id: sid, commit: false)
980
+ # P23 — short-horizon graded tasks: cap iters so practice actually
981
+ # teaches finish-under-N instead of letting DEFAULT_MAX_ITERS mask it.
982
+ prompt = opts[:prompt].to_s
983
+ short = prompt.match?(/\b(one shell|at most two|no tools|three iterations|finish under|do not explore|short plan)\b/i)
984
+ prev_max = :__unset__
985
+ capped = false
986
+ if short && defined?(PWN::Env) && PWN::Env.is_a?(Hash) && PWN::Env[:ai].is_a?(Hash) && PWN::Env[:ai][:agent].is_a?(Hash) && !PWN::Env[:ai][:agent].frozen?
987
+ prev_max = PWN::Env[:ai][:agent].key?(:max_iters) ? PWN::Env[:ai][:agent][:max_iters] : :__unset__
988
+ PWN::Env[:ai][:agent][:max_iters] = 5
989
+ capped = true
990
+ end
991
+ begin
992
+ final = Loop.run(request: prompt, session_id: sid, enabled_toolsets: %w[terminal pwn memory learning])
993
+ ensure
994
+ if capped && defined?(PWN::Env) && PWN::Env[:ai].is_a?(Hash) && PWN::Env[:ai][:agent].is_a?(Hash) && !PWN::Env[:ai][:agent].frozen?
995
+ if prev_max == :__unset__
996
+ PWN::Env[:ai][:agent].delete(:max_iters)
997
+ else
998
+ PWN::Env[:ai][:agent][:max_iters] = prev_max
999
+ end
1000
+ end
1001
+ end
1002
+ v = Reward.judge(request: prompt, final: final, session_id: sid, commit: false)
659
1003
  # 2.4 — capture tool trace for structured_fix.winning_trace
660
1004
  trace = begin
661
1005
  if defined?(PWN::Sessions)
@@ -673,21 +1017,25 @@ module PWN
673
1017
  end
674
1018
 
675
1019
  private_class_method def self.score_branch(opts = {})
676
- # 4.2 — prefer one-step real Dispatch when branch looks like tool JSON;
677
- # otherwise label imagined Reflect scores explicitly (not "advantage").
1020
+ score_branch_detailed(**opts)[:score]
1021
+ end
1022
+
1023
+ # Returns {score:, mode: :real_dispatch|:imagined|:default, trace:}.
1024
+ # Callers that need honest advantage estimation should inspect :mode.
1025
+ private_class_method def self.score_branch_detailed(opts = {})
678
1026
  branch = opts[:branch].to_s
679
1027
  request = opts[:request].to_s
680
1028
  real = try_real_dispatch_score(branch: branch)
681
- return real if real
1029
+ return { score: real, mode: :real_dispatch, trace: branch.to_s[0, 500] } if real
682
1030
 
683
- return 0.5 unless reflect_available?
1031
+ return { score: 0.5, mode: :default, trace: nil } unless reflect_available?
684
1032
 
685
1033
  req = "Goal: #{request}\nProposed next action: #{branch}\nOn a scale 0.0-1.0, how likely is this to advance the goal on a Kali Linux host? Reply with ONLY the number."
686
1034
  imagined = Reflect.on(request: req, suppress_pii_warning: true).to_s[/[01](?:\.\d+)?/].to_f.clamp(0.0, 1.0)
687
1035
  # haircut imagined scores so they never outrank a real dispatch
688
- (imagined * 0.6).clamp(0.0, 0.6)
1036
+ { score: (imagined * 0.6).clamp(0.0, 0.6), mode: :imagined, trace: nil }
689
1037
  rescue StandardError
690
- 0.5
1038
+ { score: 0.5, mode: :default, trace: nil }
691
1039
  end
692
1040
 
693
1041
  # Best-effort: if branch names a registered tool + args, run ONE Dispatch
@@ -767,6 +1115,114 @@ module PWN
767
1115
  { baseline: opts[:baseline], candidate: opts[:candidate], baseline_resolved: baseline, candidate_resolved: candid, evalset_size: evalset.length }
768
1116
  end
769
1117
 
1118
+ # P11 — promote only when candidate beats baseline on (a) resolved count
1119
+ # with margin, (b) mean judge score, and (c) no smoke regression. Smoke
1120
+ # set is fixed natural tasks, not Mistakes.top, so eval-set memorisation
1121
+ # cannot self-promote.
1122
+ private_class_method def self.ab_gate_v2(opts = {})
1123
+ evalset = Array(opts[:evalset])
1124
+ baseline = replay_on_detailed(tag: opts[:baseline], evalset: evalset)
1125
+ candid = replay_on_detailed(tag: opts[:candidate], evalset: evalset)
1126
+ smoke = smoke_eval_set
1127
+ base_smoke = replay_on_detailed(tag: opts[:baseline], evalset: smoke)
1128
+ cand_smoke = replay_on_detailed(tag: opts[:candidate], evalset: smoke)
1129
+
1130
+ b_res = baseline[:resolved].to_i
1131
+ c_res = candid[:resolved].to_i
1132
+ n = [evalset.length, 1].max
1133
+ delta = c_res - b_res
1134
+ rel = delta.to_f / n
1135
+ resolved_win = delta >= 1 && (n < 10 || rel >= 0.05)
1136
+
1137
+ b_mean = baseline[:mean_score].to_f
1138
+ c_mean = candid[:mean_score].to_f
1139
+ mean_win = c_mean + 1e-9 >= b_mean
1140
+
1141
+ smoke_ok = cand_smoke[:resolved].to_i >= base_smoke[:resolved].to_i &&
1142
+ cand_smoke[:mean_score].to_f + 0.05 >= base_smoke[:mean_score].to_f
1143
+
1144
+ promote = resolved_win && mean_win && smoke_ok
1145
+ {
1146
+ baseline: opts[:baseline],
1147
+ candidate: opts[:candidate],
1148
+ baseline_resolved: b_res,
1149
+ candidate_resolved: c_res,
1150
+ baseline_mean: b_mean.round(3),
1151
+ candidate_mean: c_mean.round(3),
1152
+ delta_resolved: delta,
1153
+ relative_delta: rel.round(3),
1154
+ smoke: {
1155
+ baseline_resolved: base_smoke[:resolved],
1156
+ candidate_resolved: cand_smoke[:resolved],
1157
+ baseline_mean: base_smoke[:mean_score],
1158
+ candidate_mean: cand_smoke[:mean_score]
1159
+ },
1160
+ resolved_win: resolved_win,
1161
+ mean_win: mean_win,
1162
+ smoke_ok: smoke_ok,
1163
+ promote: promote,
1164
+ evalset_size: evalset.length,
1165
+ gate_version: 2
1166
+ }
1167
+ end
1168
+
1169
+ private_class_method def self.smoke_eval_set
1170
+ [
1171
+ { signature: 'smoke_uname', prompt: 'Print the kernel release with uname -r' },
1172
+ { signature: 'smoke_pwd', prompt: 'Show the current working directory' },
1173
+ { signature: 'smoke_ruby', prompt: 'Display the active ruby version' }
1174
+ ]
1175
+ end
1176
+
1177
+ # P19 — weight promote requires scrubbed trajectory diversity, not just
1178
+ # A/B eval win. export_ready remains the correct posture otherwise.
1179
+ private_class_method def self.preference_diet_gate(opts = {})
1180
+ return { ok: false, reason: 'no Reward' } unless defined?(Reward)
1181
+
1182
+ bal = if Reward.respond_to?(:preference_balance)
1183
+ Reward.preference_balance(limit: opts[:limit] || 10_000, scrub: true)
1184
+ else
1185
+ preference_balance(limit: opts[:limit] || 10_000)
1186
+ end
1187
+ total = bal[:kept].to_i
1188
+ total = bal[:total].to_i if total <= 0
1189
+ return { ok: false, reason: 'too_few_pairs', total: total, min: 12 } if total < 12
1190
+
1191
+ frac = bal[:fractions] || {}
1192
+ max_share = frac.values.map(&:to_f).max || 1.0
1193
+ traj_frac = bal[:trajectory_fraction].to_f
1194
+ ok = !bal[:monoculture] && max_share <= 0.45 && traj_frac >= 0.30
1195
+ {
1196
+ ok: ok,
1197
+ total: total,
1198
+ monoculture: bal[:monoculture],
1199
+ max_source_share: max_share.round(3),
1200
+ trajectory_fraction: traj_frac.round(3),
1201
+ by_source: bal[:by_source],
1202
+ reason: ok ? 'diet_ok' : 'need_trajectory_diversity_or_rebalance'
1203
+ }
1204
+ rescue StandardError => e
1205
+ { ok: false, reason: "#{e.class}: #{e.message}" }
1206
+ end
1207
+
1208
+ private_class_method def self.replay_on_detailed(opts = {})
1209
+ tag = opts[:tag].to_s
1210
+ return { resolved: 0, mean_score: 0.0, scores: [] } if tag.empty?
1211
+
1212
+ scores = []
1213
+ with_ollama_model(tag: tag) do
1214
+ Array(opts[:evalset]).each do |e|
1215
+ r = self_play(prompt: e[:prompt], tag: "gate:#{tag}")
1216
+ scores << r[:score].to_f
1217
+ end
1218
+ end
1219
+ resolved = scores.count { |s| s >= 0.7 }
1220
+ mean = scores.empty? ? 0.0 : (scores.sum / scores.length)
1221
+ { resolved: resolved, mean_score: mean.round(3), scores: scores }
1222
+ rescue StandardError
1223
+ { resolved: 0, mean_score: 0.0, scores: [] }
1224
+ end
1225
+
770
1226
  private_class_method def self.replay_on(opts = {})
771
1227
  tag = opts[:tag].to_s
772
1228
  return 0 if tag.empty?
@@ -1118,7 +1574,7 @@ module PWN
1118
1574
  puts <<~USAGE
1119
1575
  USAGE:
1120
1576
  # Tier 4 — self-play
1121
- PWN::AI::Agent::Curriculum.practice(limit: 3) # S1 mistake-driven auto-curriculum
1577
+ PWN::AI::Agent::Curriculum.practice(limit: 3) # S1 + P14 trajectory DPO pairs + P17 budget-first
1122
1578
  PWN::AI::Agent::Curriculum.offline_judge(since_hours: 24) # P3 offline ORM/PRM fill
1123
1579
  PWN::AI::Agent::Curriculum.preference_balance # P5 W1 diversity report
1124
1580
  PWN::AI::Agent::Curriculum.counterfactual(request:, name:, args:, error:, hint:) # S2 A/B → DPO pair
@@ -1127,7 +1583,7 @@ module PWN
1127
1583
  PWN::AI::Agent::Curriculum.hindsight(request:, final:, session_id:) # C3 HER soft-relabel
1128
1584
 
1129
1585
  # Tier 5 — close the weight loop
1130
- PWN::AI::Agent::Curriculum.train_and_gate(dry_run: true) # W2 export-ready; promote only with trainer+dry_run:false
1586
+ PWN::AI::Agent::Curriculum.train_and_gate(dry_run: true) # W2 export-ready; P11 gate v2 promote only with trainer+dry_run:false
1131
1587
  PWN::AI::Agent::Curriculum.calibrate(predicted: 0.8, actual: 1.0) # W3 Brier → Metrics[:calibration]
1132
1588
 
1133
1589
  Cron self-improvement (seeded by PWN::Cron.install_defaults):