pwn 0.5.643 → 0.5.650

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -121,6 +121,40 @@ module PWN
121
121
  v = llm_judge(request: request, final: final, trace: trace)
122
122
  v ||= heuristic_judge(request: request, final: final, trace: trace)
123
123
 
124
+ # P1 — local/heuristic calibration: thin judges must not be treated
125
+ # as ground truth when proxy_distrust is already high. Two levers:
126
+ # (1) low :confidence so Metrics.effective_rate haircuts blend
127
+ # weight (distrust × confidence) instead of replacing proxy;
128
+ # (2) score-path caps only for decisive failure floors and for
129
+ # local no-trace highs (false "solved"). Do NOT pull known
130
+ # wrong (0.0 failure-language) toward 0.5, and do NOT deflate
131
+ # tool-backed heuristic scores that already cleared the bar —
132
+ # confidence handles that in the bandit blend.
133
+ eng = (PWN::Env.dig(:ai, :active) if defined?(PWN::Env)).to_s.downcase
134
+ local = eng == 'ollama' || eng.empty?
135
+ if v[:source].to_s == 'heuristic'
136
+ v[:confidence] = local ? 0.35 : 0.5
137
+ raw = v[:score].to_f
138
+ v[:score_raw] = raw
139
+ # preserve empty / failure-language / polite floors exactly (raw <= 0.15)
140
+ # and pass-through non-capped heuristics via the default arm.
141
+ if local && raw > 0.15 && trace.empty? && raw >= 0.6 && final.length < 400
142
+ v[:score] = [raw, 0.45].min
143
+ v[:rationale] = "#{v[:rationale]} | P1:local_no_trace_cap"
144
+ elsif local && raw > 0.15 && trace.length < 2 && raw >= 0.85
145
+ # thin-evidence local highs: mild shrink toward 0.5
146
+ v[:score] = (0.5 + ((raw - 0.5) * 0.7)).round(3).clamp(0.0, 1.0)
147
+ else
148
+ v[:score] = raw
149
+ end
150
+ v[:verdict] = if v[:score] >= 0.6 then :solved
151
+ elsif v[:score] >= 0.3 then :partial
152
+ else :wrong
153
+ end
154
+ else
155
+ v[:confidence] ||= 0.85
156
+ end
157
+
124
158
  ground = verify_as_reward(final: final)
125
159
  unless ground.nil?
126
160
  # Ground-truth override: a browser-refuted claim caps score at
@@ -129,13 +163,17 @@ module PWN
129
163
  v[:score] = [v[:score], 0.2].min if ground[:verdict] == :refuted
130
164
  v[:score] = [v[:score], 0.6].max if ground[:verdict] == :confirmed
131
165
  v[:grounded] = ground
166
+ v[:confidence] = [v[:confidence].to_f, ground[:confidence].to_f].max if ground[:confidence]
132
167
  end
133
168
 
134
169
  v[:success] = v[:score] >= 0.6
135
- record_sentinel(proxy: opts[:proxy_ok], judge: v[:score]) if commit
170
+ v[:engine] = eng
171
+ # P1 — sentinel stores confidence so distrust math can haircut
172
+ # heuristic-heavy windows differently from LLM ORM windows.
173
+ record_sentinel(proxy: opts[:proxy_ok], judge: v[:score], confidence: v[:confidence]) if commit
136
174
  v
137
175
  rescue StandardError => e
138
- { score: 0.5, verdict: :unknown, rationale: "judge error: #{e.class}", success: !final.strip.empty?, error: e.message }
176
+ { score: 0.5, verdict: :unknown, rationale: "judge error: #{e.class}", success: !final.strip.empty?, error: e.message, confidence: 0.2, source: :error }
139
177
  end
140
178
 
141
179
  # ----------------------------------------------------------------
@@ -297,6 +335,58 @@ module PWN
297
335
  { cleared: true, path: SENTINEL_FILE }
298
336
  end
299
337
 
338
+ # P10 — backfill the R3 ring from Learning outcomes so offline/local
339
+ # hosts reach SENTINEL_WINDOW without waiting for live remote
340
+ # introspect. Only fills empty slots; never flushes a warm window.
341
+ # Called by Curriculum.offline_judge and safe to cron.
342
+ public_class_method def self.warm_sentinel(opts = {})
343
+ s = normalize_sentinel(raw: load_sentinel)
344
+ have = Array(s[:window]).length
345
+ return { added: 0, samples: have, status: :full, proxy_distrust: proxy_distrust } if have >= SENTINEL_WINDOW
346
+ return { added: 0, samples: have, status: :no_learning, proxy_distrust: proxy_distrust } unless defined?(Learning)
347
+
348
+ need = SENTINEL_WINDOW - have
349
+ limit = (opts[:limit] || [need * 4, 200].max).to_i
350
+ # Prefer scored rows; fall back to success-boolean so local hosts still warm.
351
+ rows = Learning.outcomes(limit: limit)
352
+ scored, unscored = rows.partition { |r| !r[:score].nil? }
353
+ ordered = scored.reverse + unscored.reverse
354
+ added = 0
355
+ ordered.each do |r|
356
+ break if added >= need
357
+
358
+ judge = if r[:score]
359
+ r[:score].to_f.clamp(0.0, 1.0)
360
+ else
361
+ case r[:success]
362
+ when true, 'true' then 0.75
363
+ when 'soft' then 0.55
364
+ when false, 'false' then 0.25
365
+ else 0.5
366
+ end
367
+ end
368
+ proxy = case r[:success]
369
+ when true, 'true' then true
370
+ when false, 'false', 'soft' then false
371
+ else judge >= 0.6
372
+ end
373
+ record_sentinel(proxy: proxy, judge: judge)
374
+ added += 1
375
+ end
376
+ final_n = Array(load_sentinel[:window]).length
377
+ # Recompute distrust once window is full so controllers can engage.
378
+ snap = final_n >= SENTINEL_WINDOW ? sentinel : { samples: final_n, status: :insufficient }
379
+ {
380
+ added: added,
381
+ samples: final_n,
382
+ status: (final_n >= SENTINEL_WINDOW ? :warmed_full : :warmed_partial),
383
+ proxy_distrust: proxy_distrust,
384
+ sentinel: snap.is_a?(Hash) ? snap.slice(:samples, :status, :reward_hacked, :proxy_distrust, :proxy, :judge) : nil
385
+ }
386
+ rescue StandardError => e
387
+ { added: 0, error: "#{e.class}: #{e.message}" }
388
+ end
389
+
300
390
  # ----------------------------------------------------------------
301
391
  # R4 — Structured tool-result classifier
302
392
  # ----------------------------------------------------------------
@@ -377,6 +467,14 @@ module PWN
377
467
  claim = final[Learning::CLAIM_RX] if defined?(Learning)
378
468
  return nil if claim.to_s.empty?
379
469
 
470
+ # P26 — drop metric crumbs ("cap 0.2") that match loose patterns
471
+ if defined?(Learning) && Learning.respond_to?(:checkable_claim?, true)
472
+ return nil unless Learning.send(:checkable_claim?, claim: claim)
473
+ elsif claim.match?(/\A(?:cap|share|proxy|judge|success|only|now|gap|score|rate)\b/i) ||
474
+ (claim.match?(/\d+\.\d+/) && !claim.match?(/\d+\.\d+\.\d+|CVE-/i))
475
+ return nil
476
+ end
477
+
380
478
  # 1.5 — sampled E3: always when flag true; never when false;
381
479
  # nil/auto → always on frontier, ~10% on local when CLAIM_RX hits.
382
480
  flag = agent_flag(key: :verify_as_reward, default: nil)
@@ -408,6 +506,29 @@ module PWN
408
506
  # source: 'optional - :user_correction | :mistakes_resolve | :counterfactual | :critic'
409
507
  # )
410
508
 
509
+ # Trajectory-shaped chosen sides that may land DPO without prose flood.
510
+ TRAJECTORY_SHAPES = %w[winning_trace revised_answer real_dispatch].freeze
511
+
512
+ # P9 — write-time source quota (not only export). Prefer diverse online
513
+ # generators over resolve-prose flood. Window is last WRITE_SOURCE_WINDOW
514
+ # pairs; a source already above WRITE_SOURCE_CAP is refused unless
515
+ # force: true (user_correction always forces).
516
+ WRITE_SOURCE_CAP = 0.40
517
+ WRITE_SOURCE_WINDOW = 100
518
+
519
+ # P0 — target online generator mix for W1. Gates alone cannot fill
520
+ # an empty promote; the controller must *prefer underfilled* sources
521
+ # (counterfactual / critic / curriculum / user_correction) when
522
+ # resolve already dominates. Shares are soft targets, not hard caps
523
+ # (hard cap remains WRITE_SOURCE_CAP). Trajectory-only still applies.
524
+ TARGET_SOURCE_MIX = {
525
+ 'mistakes_resolve' => 0.30,
526
+ 'curriculum' => 0.25,
527
+ 'counterfactual' => 0.20,
528
+ 'critic' => 0.15,
529
+ 'user_correction' => 0.10
530
+ }.freeze
531
+
411
532
  public_class_method def self.record_preference(opts = {})
412
533
  prompt = opts[:prompt].to_s
413
534
  rejected = opts[:rejected].to_s
@@ -415,20 +536,299 @@ module PWN
415
536
  return nil if prompt.strip.empty? || chosen.strip.empty? || rejected.strip.empty?
416
537
  return nil if chosen.strip == rejected.strip
417
538
 
539
+ # Reject weak pair geometry: CORRECTION: flaw-prose is not a trajectory.
540
+ return { skipped: :weak_pair_geometry, reason: 'chosen looks like flaw prose, not a revised answer/trace' } if chosen.match?(/\A\s*CORRECTION:\s*/i) && chosen.length < 400 && !opts[:force]
541
+
542
+ source = (opts[:source] || :unknown).to_s
543
+ shape = opts[:shape].to_s
544
+ # P25 — require trajectory shape at write time unless force / user_correction.
545
+ # Stops resolve-prose flood from ever landing in the ledger; export scrub
546
+ # is defense-in-depth, not the primary gate.
547
+ traj = TRAJECTORY_SHAPES.include?(shape)
548
+ # P25 — non-trajectory prose never lands (export scrub is defense-in-depth).
549
+ # user_correction and explicit force: still allowed for human / migration paths.
550
+ unless traj || opts[:force] || source == 'user_correction'
551
+ return {
552
+ skipped: :non_trajectory_shape,
553
+ reason: "shape=#{shape.inspect} not in #{TRAJECTORY_SHAPES.join(',')}; pass force:true or a trajectory shape",
554
+ source: source
555
+ }
556
+ end
557
+ # P9 — write-time source quota still applies to trajectory pairs.
558
+ # P25 made every auto-written row trajectory-shaped; if traj also
559
+ # bypassed the quota, resolve monoculture would return via winning_trace
560
+ # flood. Only user_correction and explicit force:true skip the cap.
561
+ bypass_quota = opts[:force] || source == 'user_correction'
562
+ unless bypass_quota
563
+ quota = write_source_quota(source: source)
564
+ return quota.merge(skipped: :source_quota) if quota[:over_cap]
565
+ end
566
+
418
567
  entry = {
419
568
  id: Digest::SHA256.hexdigest("#{prompt}|#{rejected}|#{chosen}")[0, 12],
420
569
  prompt: prompt[0, 4_000],
421
570
  rejected: rejected[0, 4_000],
422
571
  chosen: chosen[0, 4_000],
423
- source: (opts[:source] || :unknown).to_s,
572
+ source: source,
424
573
  engine: (PWN::Env.dig(:ai, :active) if defined?(PWN::Env)).to_s,
425
574
  timestamp: Time.now.utc.iso8601
426
575
  }
576
+ entry[:meta] = opts[:meta] if opts[:meta].is_a?(Hash)
577
+ entry[:shape] = opts[:shape].to_s if opts[:shape]
427
578
  FileUtils.mkdir_p(File.dirname(PREFERENCES_FILE))
428
579
  File.open(PREFERENCES_FILE, 'a') { |f| f.puts(JSON.generate(entry)) }
429
580
  entry
430
581
  end
431
582
 
583
+ # Share of `source` among the newest WRITE_SOURCE_WINDOW prefs.
584
+ public_class_method def self.write_source_quota(opts = {})
585
+ source = opts[:source].to_s
586
+ recent = preferences(limit: WRITE_SOURCE_WINDOW)
587
+ return { over_cap: false, share: 0.0, n: 0, window: recent.length, underfilled: true } if recent.length < 10
588
+
589
+ n = recent.count { |r| r[:source].to_s == source }
590
+ share = n.to_f / recent.length
591
+ target = TARGET_SOURCE_MIX[source]
592
+ {
593
+ over_cap: share > WRITE_SOURCE_CAP,
594
+ share: share.round(3),
595
+ n: n,
596
+ window: recent.length,
597
+ source: source,
598
+ cap: WRITE_SOURCE_CAP,
599
+ target: target,
600
+ underfilled: target ? share < (target * 0.5) : share < 0.05,
601
+ deficit: target ? (target - share).round(3) : nil
602
+ }
603
+ rescue StandardError
604
+ { over_cap: false, share: 0.0, underfilled: true }
605
+ end
606
+
607
+ # P0 — online generator mix report + urgency flags. Controllers
608
+ # (auto_introspect, practice, counterfactual gate) consult this so
609
+ # underfilled sources get scheduling priority while over-cap
610
+ # resolve stops flooding. Returns {by_source, trajectory_fraction,
611
+ # urgent:[], suppress:[], healthy:}.
612
+ public_class_method def self.generator_mix(opts = {})
613
+ limit = opts[:limit] || WRITE_SOURCE_WINDOW
614
+ rows = preferences(limit: limit)
615
+ usable = rows.select { |r| usable_preference?(row: r) }
616
+ by = Hash.new(0)
617
+ usable.each { |r| by[r[:source].to_s] += 1 }
618
+ n = usable.length
619
+ shares = {}
620
+ TARGET_SOURCE_MIX.each_key { |k| shares[k] = n.zero? ? 0.0 : (by[k].to_f / n).round(3) }
621
+ by.each_key { |k| shares[k] ||= (by[k].to_f / n).round(3) }
622
+
623
+ traj_n = usable.count { |r| TRAJECTORY_SHAPES.include?(r[:shape].to_s) }
624
+ traj_f = n.zero? ? 0.0 : (traj_n.to_f / n).round(3)
625
+
626
+ urgent = []
627
+ suppress = []
628
+ TARGET_SOURCE_MIX.each do |src, target|
629
+ sh = shares[src].to_f
630
+ urgent << src if sh < (target * 0.5) && n >= 5
631
+ suppress << src if sh > WRITE_SOURCE_CAP && n >= 10
632
+ end
633
+ suppress << 'mistakes_resolve' if shares['mistakes_resolve'].to_f > WRITE_SOURCE_CAP && n >= 10 && !suppress.include?('mistakes_resolve')
634
+
635
+ healthy = urgent.empty? && suppress.empty? && traj_f >= 0.5 && n >= 10
636
+ {
637
+ n: n,
638
+ raw_n: rows.length,
639
+ by_source: by,
640
+ shares: shares,
641
+ targets: TARGET_SOURCE_MIX,
642
+ trajectory_fraction: traj_f,
643
+ urgent: urgent.uniq,
644
+ suppress: suppress.uniq,
645
+ healthy: healthy,
646
+ recommendation: if healthy
647
+ 'mix_ok'
648
+ elsif n < 10
649
+ 'need_more_pairs'
650
+ elsif traj_f < 0.5
651
+ 'need_trajectory_shape'
652
+ elsif urgent.any?
653
+ "boost:#{urgent.join(',')}"
654
+ else
655
+ "suppress:#{suppress.join(',')}"
656
+ end
657
+ }
658
+ rescue StandardError => e
659
+ {
660
+ n: 0, healthy: false, error: "#{e.class}: #{e.message}",
661
+ urgent: %w[curriculum counterfactual critic user_correction],
662
+ suppress: []
663
+ }
664
+ end
665
+
666
+ # P0 ops — infer trajectory shape for legacy ledger rows that predate
667
+ # P21/P25 shape tags. Used by scrub_preferences rewrite so generator_mix
668
+ # trajectory_fraction reflects content, not missing keys.
669
+ public_class_method def self.infer_shape(opts = {})
670
+ r = opts.is_a?(Hash) && opts.key?(:row) ? opts[:row] : opts
671
+ r = r.transform_keys(&:to_sym) if r.respond_to?(:transform_keys)
672
+ existing = r[:shape].to_s
673
+ return existing if TRAJECTORY_SHAPES.include?(existing) || existing == 'fix_prose'
674
+
675
+ chosen = r[:chosen].to_s
676
+ source = r[:source].to_s
677
+ # tool-call / trace markers → winning_trace
678
+ if chosen.match?(/\b(shell|pwn_eval|memory_|sessions_|reward_|curriculum_|extro_|mistakes_)\b/i) &&
679
+ (chosen.include?('→') || chosen.include?('tool_call') || chosen.include?('"name"') ||
680
+ chosen.lines.count { |l| l.strip.start_with?('{') || l.include?('arguments') } >= 1)
681
+ return 'winning_trace'
682
+ end
683
+ # long revised answer from critic / user / CF → revised_answer
684
+ return 'revised_answer' if chosen.length >= 200 && %w[critic user_correction counterfactual curriculum].include?(source)
685
+
686
+ # counterfactual real dispatch tag in meta
687
+ meta = r[:meta].is_a?(Hash) ? r[:meta] : {}
688
+ return 'real_dispatch' if meta[:mode].to_s == 'real_dispatch' || meta['mode'].to_s == 'real_dispatch'
689
+
690
+ existing.empty? ? nil : existing
691
+ rescue StandardError
692
+ nil
693
+ end
694
+
695
+ # P15 — keep only usable preference pairs for balance/export/promote.
696
+ # Drops CORRECTION-only chosen, resolve rows without trajectory shape,
697
+ # and chosen≪rejected unless shape is a known trajectory form.
698
+ public_class_method def self.usable_preference?(opts = {})
699
+ r = opts.is_a?(Hash) && opts.key?(:row) ? opts[:row] : opts
700
+ r = r.transform_keys(&:to_sym) if r.respond_to?(:transform_keys)
701
+ chosen = r[:chosen].to_s
702
+ rejected = r[:rejected].to_s
703
+ shape = r[:shape].to_s
704
+ source = r[:source].to_s
705
+ return false if chosen.strip.empty? || rejected.strip.empty?
706
+ return false if chosen.strip == rejected.strip
707
+ return false if chosen.match?(/\A\s*CORRECTION:\s*/i) && chosen.length < 400
708
+ return false if shape == 'fix_prose'
709
+ # P25 — resolve rows must be trajectory-shaped to count as usable
710
+ return false if source == 'mistakes_resolve' && !TRAJECTORY_SHAPES.include?(shape)
711
+
712
+ # chosen ≪ rejected without trajectory shape → commentary, not policy
713
+ unless TRAJECTORY_SHAPES.include?(shape)
714
+ return false if rejected.length >= 200 && chosen.length < (rejected.length * 0.25) && chosen.length < 200
715
+ return false if rejected.length >= 400 && chosen.length < 120
716
+ end
717
+ true
718
+ rescue StandardError
719
+ false
720
+ end
721
+
722
+ # P15 — one-shot ledger hygiene. Filters in place (rewrite jsonl) or
723
+ # report-only. Returns {before:, after:, dropped:, by_reason:, path:}.
724
+ public_class_method def self.scrub_preferences(opts = {})
725
+ dry = opts.key?(:dry_run) ? opts[:dry_run] : false
726
+ path = PREFERENCES_FILE
727
+ return { before: 0, after: 0, dropped: 0, dry_run: dry, path: path } unless File.exist?(path)
728
+
729
+ raw = File.readlines(path)
730
+ kept = []
731
+ reasons = Hash.new(0)
732
+ raw.each do |line|
733
+ begin
734
+ r = JSON.parse(line, symbolize_names: true)
735
+ rescue StandardError
736
+ reasons[:parse_error] += 1
737
+ next
738
+ end
739
+ if usable_preference?(row: r)
740
+ # P0 ops — backfill shape so trajectory_fraction is meaningful
741
+ if r[:shape].to_s.empty?
742
+ inferred = infer_shape(row: r)
743
+ r = r.merge(shape: inferred) if inferred
744
+ end
745
+ kept << r
746
+ else
747
+ why = if r[:chosen].to_s.match?(/\A\s*CORRECTION:\s*/i)
748
+ :correction_prose
749
+ elsif r[:shape].to_s == 'fix_prose'
750
+ :fix_prose
751
+ elsif r[:chosen].to_s.length < (r[:rejected].to_s.length * 0.25)
752
+ :chosen_too_short
753
+ else
754
+ :weak_geometry
755
+ end
756
+ reasons[why] += 1
757
+ end
758
+ end
759
+ unless dry
760
+ bak = "#{path}.bak-p15-#{Time.now.utc.strftime('%Y%m%d%H%M%S')}"
761
+ FileUtils.cp(path, bak)
762
+ File.open(path, 'w') { |f| kept.each { |r| f.puts(JSON.generate(r)) } }
763
+ end
764
+ {
765
+ before: raw.length,
766
+ after: kept.length,
767
+ dropped: raw.length - kept.length,
768
+ by_reason: reasons,
769
+ dry_run: dry,
770
+ path: path,
771
+ backup: dry ? nil : bak
772
+ }
773
+ rescue StandardError => e
774
+ { error: "#{e.class}: #{e.message}" }
775
+ end
776
+
777
+ # P15/P5 — geometry-aware source mix. scrub:true uses usable_preference?
778
+ # so operators see the post-hygiene diet (what export_dpo will train on).
779
+ public_class_method def self.preference_balance(opts = {})
780
+ limit = opts[:limit] || 10_000
781
+ scrub = opts.key?(:scrub) ? opts[:scrub] : false
782
+ rows = preferences(limit: limit)
783
+ before = rows.length
784
+ rows = rows.select { |r| usable_preference?(row: r) } if scrub
785
+ by = Hash.new(0)
786
+ by_shape = Hash.new(0)
787
+ rows.each do |r|
788
+ by[r[:source].to_s] += 1
789
+ sh = r[:shape].to_s
790
+ sh = 'unspecified' if sh.empty?
791
+ by_shape[sh] += 1
792
+ end
793
+ total = rows.length
794
+ frac = by.transform_values { |n| total.zero? ? 0.0 : (n.to_f / total).round(3) }
795
+ shape_frac = by_shape.transform_values { |n| total.zero? ? 0.0 : (n.to_f / total).round(3) }
796
+ traj_n = rows.count { |r| TRAJECTORY_SHAPES.include?(r[:shape].to_s) }
797
+ traj_frac = total.zero? ? 0.0 : (traj_n.to_f / total).round(3)
798
+ monoculture = total.positive? && (by.values.max.to_f / total) > 0.7
799
+ mix = begin
800
+ generator_mix(limit: limit)
801
+ rescue StandardError
802
+ nil
803
+ end
804
+ {
805
+ total: before,
806
+ kept: total,
807
+ scrubbed: scrub,
808
+ dropped: before - total,
809
+ by_source: by,
810
+ fractions: frac,
811
+ by_shape: by_shape,
812
+ by_shape_fraction: shape_frac,
813
+ trajectory_fraction: traj_frac,
814
+ monoculture: monoculture,
815
+ generator_mix: mix,
816
+ advice: if total < 12
817
+ 'W1 thin: need more trajectory-shaped pairs before LoRA promote.'
818
+ elsif monoculture
819
+ 'W1 monoculture: run Reward.scrub_preferences; enable :counterfactual/:critic; stop resolve-prose flood.'
820
+ elsif traj_frac < 0.30
821
+ 'W1 geometry weak: <30% trajectory-shaped chosen sides — DPO would teach commentary.'
822
+ elsif mix && !mix[:healthy]
823
+ "W1 generator mix: #{mix[:recommendation]}"
824
+ else
825
+ 'W1 source mix OK for gated export'
826
+ end
827
+ }
828
+ rescue StandardError => e
829
+ { error: "#{e.class}: #{e.message}" }
830
+ end
831
+
432
832
  # Supported Method Parameters::
433
833
  # rows = PWN::AI::Agent::Reward.preferences(limit: 500, source: nil)
434
834
 
@@ -463,6 +863,15 @@ module PWN
463
863
  FileUtils.mkdir_p(DPO_DIR)
464
864
  out = opts[:out] || File.join(DPO_DIR, "pwn-dpo-#{Time.now.utc.strftime('%Y%m%d')}.jsonl")
465
865
  rows = preferences(limit: 100_000)
866
+ # P15 — drop weak geometry before source-cap so resolve prose cannot
867
+ # dominate the kept set after balance. opt-out with scrub: false.
868
+ scrub = opts.key?(:scrub) ? opts[:scrub] : true
869
+ geometry_dropped = 0
870
+ if scrub
871
+ usable = rows.select { |r| usable_preference?(row: r) }
872
+ geometry_dropped = rows.length - usable.length
873
+ rows = usable
874
+ end
466
875
  # P5 — downsample so no single source exceeds DPO_SOURCE_CAP of the export.
467
876
  # opt-out with balance: false (raw dump for diagnostics).
468
877
  balance = opts.key?(:balance) ? opts[:balance] : true
@@ -484,8 +893,14 @@ module PWN
484
893
  by_src = selected.group_by { |r| r[:source].to_s }.transform_values(&:length)
485
894
  {
486
895
  path: out, format: fmt, pairs: selected.length, bytes: File.size(out),
487
- balanced: balance, dropped: dropped, by_source: by_src,
488
- source_cap: balance ? (opts[:source_cap] || DPO_SOURCE_CAP).to_f : nil
896
+ balanced: balance, dropped: dropped, geometry_dropped: geometry_dropped,
897
+ scrubbed: scrub, by_source: by_src,
898
+ source_cap: balance ? (opts[:source_cap] || DPO_SOURCE_CAP).to_f : nil,
899
+ preference_balance: begin
900
+ preference_balance(limit: 10_000, scrub: true)
901
+ rescue StandardError
902
+ nil
903
+ end
489
904
  }
490
905
  end
491
906
 
@@ -615,7 +1030,11 @@ module PWN
615
1030
  j = JSON.parse(l, symbolize_names: true)
616
1031
  if j[:role].to_s == 'tool'
617
1032
  ti += 1
618
- j[:step_reward] = rewards[ti] if rewards[ti]
1033
+ if rewards[ti]
1034
+ j[:step_reward] = rewards[ti]
1035
+ # P18 — fold step_reward into Metrics so Registry.rank can bias
1036
+ fold_step_reward_to_metrics(content: j[:content], reward: rewards[ti])
1037
+ end
619
1038
  end
620
1039
  "#{JSON.generate(j)}\n"
621
1040
  rescue StandardError
@@ -626,12 +1045,29 @@ module PWN
626
1045
  nil
627
1046
  end
628
1047
 
1048
+ # Extract tool name from "name → …" session content and record PRM.
1049
+ private_class_method def self.fold_step_reward_to_metrics(opts = {})
1050
+ return unless defined?(Metrics) && Metrics.respond_to?(:record_step_reward)
1051
+
1052
+ content = opts[:content].to_s
1053
+ name = content[/\A([a-z_][a-z0-9_]*)\s*→/i, 1] ||
1054
+ content[/\A([a-z_][a-z0-9_]*)\s*->/i, 1] ||
1055
+ content[/\A([a-z_][a-z0-9_]*)/, 1]
1056
+ return if name.to_s.empty?
1057
+
1058
+ Metrics.record_step_reward(name: name, reward: opts[:reward])
1059
+ rescue StandardError
1060
+ nil
1061
+ end
1062
+
629
1063
  private_class_method def self.record_sentinel(opts = {})
630
1064
  s = normalize_sentinel(raw: load_sentinel)
631
1065
  # Clamp judge to [0,1] — LLM/heuristic should already, but a bad
632
1066
  # write must not poison rolling means forever.
633
1067
  judge = opts[:judge].to_f.clamp(0.0, 1.0)
634
1068
  entry = { judge: judge, at: Time.now.utc.iso8601 }
1069
+ # P1 — optional per-sample confidence (heuristic < LLM ORM)
1070
+ entry[:confidence] = opts[:confidence].to_f.clamp(0.0, 1.0) unless opts[:confidence].nil?
635
1071
  # 1.3 — only roll proxy into the window when the caller actually
636
1072
  # supplied a R4-aligned proxy_ok. Pre-ORM boolean noise no longer
637
1073
  # dilutes gap_proxy_judge. Proxy is ALWAYS 0.0 or 1.0 when present.
@@ -872,13 +1308,17 @@ module PWN
872
1308
  PWN::AI::Agent::Reward.prm(request: req, session_id: sid) # R2 PRM → per-step credit
873
1309
  PWN::AI::Agent::Reward.sentinel # R3 reward-hacking detector
874
1310
  PWN::AI::Agent::Reward.reset_sentinel # wipe corrupt window + distrust
1311
+ PWN::AI::Agent::Reward.warm_sentinel # P10 fill R3 window from Learning outcomes
875
1312
  PWN::AI::Agent::Reward.semantic_ok(name: 'shell', raw: json, args: args) # R4 kills phantom exit≠0 mistakes
876
1313
 
877
1314
  # Tier 5 — preference pairs → DPO
878
1315
  PWN::AI::Agent::Reward.record_preference(prompt: p, rejected: r, chosen: c, source: :user_correction)
879
1316
  PWN::AI::Agent::Reward.preferences(limit: 100)
880
- PWN::AI::Agent::Reward.export_dpo(format: :dpo) # W1 → ~/.pwn/finetune/pwn-dpo-*.jsonl (≤40%/source)
1317
+ PWN::AI::Agent::Reward.export_dpo(format: :dpo) # W1 → ~/.pwn/finetune/pwn-dpo-*.jsonl (≤40%/source, scrubbed)
881
1318
  PWN::AI::Agent::Reward.export_dpo(format: :dpo, balance: false) # raw dump (diagnostics)
1319
+ PWN::AI::Agent::Reward.scrub_preferences(dry_run: true) # P15 ledger hygiene report
1320
+ PWN::AI::Agent::Reward.scrub_preferences # P15 rewrite jsonl (backup first)
1321
+ PWN::AI::Agent::Reward.preference_balance(scrub: true) # P15 geometry-aware mix
882
1322
 
883
1323
  # Tier 6 — grounded reward
884
1324
  PWN::AI::Agent::Reward.verify_as_reward(final: text) # E3 browser-verified reward
@@ -144,3 +144,26 @@ PWN::AI::Agent::Registry.register(
144
144
  check: -> { defined?(PWN::AI::Agent::Curriculum) && PWN::AI::Agent::Curriculum.respond_to?(:preference_balance) },
145
145
  handler: ->(args) { PWN::AI::Agent::Curriculum.preference_balance(limit: args[:limit]) }
146
146
  )
147
+
148
+ PWN::AI::Agent::Registry.register(
149
+ name: 'curriculum_practice_kpi',
150
+ toolset: 'curriculum',
151
+ schema: {
152
+ name: 'curriculum_practice_kpi',
153
+ description: 'P1 — Outer curriculum KPI: unresolved [REPEATING] count + week-over-week trend (improving|flat|regressing). Snapshots land in ~/.pwn/curriculum_kpi.jsonl from practice().',
154
+ parameters: {
155
+ type: 'object',
156
+ properties: {
157
+ limit: { type: 'integer', description: 'Trend window snapshots (default 14)' }
158
+ },
159
+ required: []
160
+ }
161
+ },
162
+ check: -> { defined?(PWN::AI::Agent::Curriculum) && PWN::AI::Agent::Curriculum.respond_to?(:repeating_trend) },
163
+ handler: lambda { |args|
164
+ {
165
+ trend: PWN::AI::Agent::Curriculum.repeating_trend(limit: args[:limit]),
166
+ snapshot: PWN::AI::Agent::Curriculum.practice_kpi(results: [])
167
+ }
168
+ }
169
+ )
@@ -126,3 +126,89 @@ PWN::AI::Agent::Registry.register(
126
126
  )
127
127
  }
128
128
  )
129
+
130
+ PWN::AI::Agent::Registry.register(
131
+ name: 'reward_warm_sentinel',
132
+ toolset: 'learning',
133
+ schema: {
134
+ name: 'reward_warm_sentinel',
135
+ description: 'P10 — Backfill the R3 sentinel ring buffer from Learning ' \
136
+ 'outcomes so proxy_distrust can engage on local/offline hosts ' \
137
+ 'without waiting for live remote introspect. Only fills empty ' \
138
+ 'slots; never flushes a warm window.',
139
+ parameters: {
140
+ type: 'object',
141
+ properties: {
142
+ limit: { type: 'integer', default: 120, description: 'Max outcomes to scan.' }
143
+ },
144
+ required: []
145
+ }
146
+ },
147
+ check: -> { defined?(PWN::AI::Agent::Reward) && PWN::AI::Agent::Reward.respond_to?(:warm_sentinel) },
148
+ handler: ->(args) { PWN::AI::Agent::Reward.warm_sentinel(limit: args[:limit] || 120) }
149
+ )
150
+
151
+ PWN::AI::Agent::Registry.register(
152
+ name: 'reward_scrub_preferences',
153
+ toolset: 'learning',
154
+ schema: {
155
+ name: 'reward_scrub_preferences',
156
+ description: 'P15 — One-shot W1 ledger hygiene. Drops CORRECTION:-only chosen, resolve rows without trajectory shape, and chosen≪rejected pairs. dry_run:true reports; false rewrites ~/.pwn/preferences.jsonl (backup first).',
157
+ parameters: {
158
+ type: 'object',
159
+ properties: {
160
+ dry_run: { type: 'boolean', default: true, description: 'Report only (default true).' }
161
+ },
162
+ required: []
163
+ }
164
+ },
165
+ check: -> { defined?(PWN::AI::Agent::Reward) && PWN::AI::Agent::Reward.respond_to?(:scrub_preferences) },
166
+ handler: lambda { |args|
167
+ dry = args.key?(:dry_run) ? args[:dry_run] : true
168
+ PWN::AI::Agent::Reward.scrub_preferences(dry_run: dry)
169
+ }
170
+ )
171
+
172
+ PWN::AI::Agent::Registry.register(
173
+ name: 'reward_preference_balance',
174
+ toolset: 'learning',
175
+ schema: {
176
+ name: 'reward_preference_balance',
177
+ description: 'P15/P5 — Geometry-aware W1 diversity report. scrub:true scores the post-hygiene diet (trajectory_fraction, by_shape) so promote/export gates see usable pairs only.',
178
+ parameters: {
179
+ type: 'object',
180
+ properties: {
181
+ limit: { type: 'integer', default: 10_000 },
182
+ scrub: { type: 'boolean', default: true }
183
+ },
184
+ required: []
185
+ }
186
+ },
187
+ check: -> { defined?(PWN::AI::Agent::Reward) && PWN::AI::Agent::Reward.respond_to?(:preference_balance) },
188
+ handler: lambda { |args|
189
+ PWN::AI::Agent::Reward.preference_balance(
190
+ limit: args[:limit] || 10_000,
191
+ scrub: args.key?(:scrub) ? args[:scrub] : true
192
+ )
193
+ }
194
+ )
195
+
196
+ PWN::AI::Agent::Registry.register(
197
+ name: 'reward_generator_mix',
198
+ toolset: 'reward',
199
+ schema: {
200
+ name: 'reward_generator_mix',
201
+ description: 'P0 — Online W1 generator mix vs TARGET_SOURCE_MIX. Returns urgent/suppress sources, trajectory_fraction, recommendation (boost:… / suppress:…). Controllers use this to schedule counterfactual/critic/curriculum over resolve flood.',
202
+ parameters: {
203
+ type: 'object',
204
+ properties: {
205
+ limit: { type: 'integer', description: 'Window size (default WRITE_SOURCE_WINDOW)' }
206
+ },
207
+ required: []
208
+ }
209
+ },
210
+ check: -> { defined?(PWN::AI::Agent::Reward) && PWN::AI::Agent::Reward.respond_to?(:generator_mix) },
211
+ handler: lambda { |args|
212
+ PWN::AI::Agent::Reward.generator_mix(limit: args[:limit])
213
+ }
214
+ )