pwn 0.5.643 → 0.5.650
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/documentation/Agent-Tool-Registry.md +10 -5
- data/documentation/Cron.md +3 -2
- data/documentation/Home.md +2 -2
- data/documentation/How-PWN-Works.md +1 -1
- data/documentation/Mistakes.md +13 -0
- data/documentation/Reinforcement-Learning.md +72 -3
- data/documentation/Skills-Memory-Learning.md +5 -3
- data/documentation/What-is-PWN.md +1 -1
- data/documentation/diagrams/dot/pwn-ai-feedback-learning-loop.dot +3 -3
- data/documentation/diagrams/dot/reinforcement-learning.dot +8 -8
- data/documentation/diagrams/pwn-ai-feedback-learning-loop.svg +318 -319
- data/documentation/diagrams/reinforcement-learning.svg +188 -187
- data/documentation/pwn-ai-Agent.md +6 -5
- data/lib/pwn/ai/agent/curriculum.rb +491 -35
- data/lib/pwn/ai/agent/extrospection.rb +18 -4
- data/lib/pwn/ai/agent/learning.rb +308 -54
- data/lib/pwn/ai/agent/loop.rb +145 -11
- data/lib/pwn/ai/agent/metrics.rb +200 -11
- data/lib/pwn/ai/agent/mistakes.rb +26 -8
- data/lib/pwn/ai/agent/registry.rb +24 -2
- data/lib/pwn/ai/agent/reward.rb +447 -7
- data/lib/pwn/ai/agent/tools/curriculum.rb +23 -0
- data/lib/pwn/ai/agent/tools/reward.rb +86 -0
- data/lib/pwn/ai/agent/tools/ruby_eval.rb +29 -6
- data/lib/pwn/cron.rb +14 -2
- data/lib/pwn/version.rb +1 -1
- data/spec/integration/reinforced_feedback_loop_spec.rb +806 -12
- data/spec/lib/pwn/ai/agent/reward_spec.rb +64 -1
- data/spec/lib/pwn/ai/agent/tools/ruby_eval_spec.rb +39 -1
- data/third_party/pwn_rdoc.jsonl +33 -4
- metadata +1 -1
data/lib/pwn/ai/agent/reward.rb
CHANGED
|
@@ -121,6 +121,40 @@ module PWN
|
|
|
121
121
|
v = llm_judge(request: request, final: final, trace: trace)
|
|
122
122
|
v ||= heuristic_judge(request: request, final: final, trace: trace)
|
|
123
123
|
|
|
124
|
+
# P1 — local/heuristic calibration: thin judges must not be treated
|
|
125
|
+
# as ground truth when proxy_distrust is already high. Two levers:
|
|
126
|
+
# (1) low :confidence so Metrics.effective_rate haircuts blend
|
|
127
|
+
# weight (distrust × confidence) instead of replacing proxy;
|
|
128
|
+
# (2) score-path caps only for decisive failure floors and for
|
|
129
|
+
# local no-trace highs (false "solved"). Do NOT pull known
|
|
130
|
+
# wrong (0.0 failure-language) toward 0.5, and do NOT deflate
|
|
131
|
+
# tool-backed heuristic scores that already cleared the bar —
|
|
132
|
+
# confidence handles that in the bandit blend.
|
|
133
|
+
eng = (PWN::Env.dig(:ai, :active) if defined?(PWN::Env)).to_s.downcase
|
|
134
|
+
local = eng == 'ollama' || eng.empty?
|
|
135
|
+
if v[:source].to_s == 'heuristic'
|
|
136
|
+
v[:confidence] = local ? 0.35 : 0.5
|
|
137
|
+
raw = v[:score].to_f
|
|
138
|
+
v[:score_raw] = raw
|
|
139
|
+
# preserve empty / failure-language / polite floors exactly (raw <= 0.15)
|
|
140
|
+
# and pass-through non-capped heuristics via the default arm.
|
|
141
|
+
if local && raw > 0.15 && trace.empty? && raw >= 0.6 && final.length < 400
|
|
142
|
+
v[:score] = [raw, 0.45].min
|
|
143
|
+
v[:rationale] = "#{v[:rationale]} | P1:local_no_trace_cap"
|
|
144
|
+
elsif local && raw > 0.15 && trace.length < 2 && raw >= 0.85
|
|
145
|
+
# thin-evidence local highs: mild shrink toward 0.5
|
|
146
|
+
v[:score] = (0.5 + ((raw - 0.5) * 0.7)).round(3).clamp(0.0, 1.0)
|
|
147
|
+
else
|
|
148
|
+
v[:score] = raw
|
|
149
|
+
end
|
|
150
|
+
v[:verdict] = if v[:score] >= 0.6 then :solved
|
|
151
|
+
elsif v[:score] >= 0.3 then :partial
|
|
152
|
+
else :wrong
|
|
153
|
+
end
|
|
154
|
+
else
|
|
155
|
+
v[:confidence] ||= 0.85
|
|
156
|
+
end
|
|
157
|
+
|
|
124
158
|
ground = verify_as_reward(final: final)
|
|
125
159
|
unless ground.nil?
|
|
126
160
|
# Ground-truth override: a browser-refuted claim caps score at
|
|
@@ -129,13 +163,17 @@ module PWN
|
|
|
129
163
|
v[:score] = [v[:score], 0.2].min if ground[:verdict] == :refuted
|
|
130
164
|
v[:score] = [v[:score], 0.6].max if ground[:verdict] == :confirmed
|
|
131
165
|
v[:grounded] = ground
|
|
166
|
+
v[:confidence] = [v[:confidence].to_f, ground[:confidence].to_f].max if ground[:confidence]
|
|
132
167
|
end
|
|
133
168
|
|
|
134
169
|
v[:success] = v[:score] >= 0.6
|
|
135
|
-
|
|
170
|
+
v[:engine] = eng
|
|
171
|
+
# P1 — sentinel stores confidence so distrust math can haircut
|
|
172
|
+
# heuristic-heavy windows differently from LLM ORM windows.
|
|
173
|
+
record_sentinel(proxy: opts[:proxy_ok], judge: v[:score], confidence: v[:confidence]) if commit
|
|
136
174
|
v
|
|
137
175
|
rescue StandardError => e
|
|
138
|
-
{ score: 0.5, verdict: :unknown, rationale: "judge error: #{e.class}", success: !final.strip.empty?, error: e.message }
|
|
176
|
+
{ score: 0.5, verdict: :unknown, rationale: "judge error: #{e.class}", success: !final.strip.empty?, error: e.message, confidence: 0.2, source: :error }
|
|
139
177
|
end
|
|
140
178
|
|
|
141
179
|
# ----------------------------------------------------------------
|
|
@@ -297,6 +335,58 @@ module PWN
|
|
|
297
335
|
{ cleared: true, path: SENTINEL_FILE }
|
|
298
336
|
end
|
|
299
337
|
|
|
338
|
+
# P10 — backfill the R3 ring from Learning outcomes so offline/local
|
|
339
|
+
# hosts reach SENTINEL_WINDOW without waiting for live remote
|
|
340
|
+
# introspect. Only fills empty slots; never flushes a warm window.
|
|
341
|
+
# Called by Curriculum.offline_judge and safe to cron.
|
|
342
|
+
public_class_method def self.warm_sentinel(opts = {})
|
|
343
|
+
s = normalize_sentinel(raw: load_sentinel)
|
|
344
|
+
have = Array(s[:window]).length
|
|
345
|
+
return { added: 0, samples: have, status: :full, proxy_distrust: proxy_distrust } if have >= SENTINEL_WINDOW
|
|
346
|
+
return { added: 0, samples: have, status: :no_learning, proxy_distrust: proxy_distrust } unless defined?(Learning)
|
|
347
|
+
|
|
348
|
+
need = SENTINEL_WINDOW - have
|
|
349
|
+
limit = (opts[:limit] || [need * 4, 200].max).to_i
|
|
350
|
+
# Prefer scored rows; fall back to success-boolean so local hosts still warm.
|
|
351
|
+
rows = Learning.outcomes(limit: limit)
|
|
352
|
+
scored, unscored = rows.partition { |r| !r[:score].nil? }
|
|
353
|
+
ordered = scored.reverse + unscored.reverse
|
|
354
|
+
added = 0
|
|
355
|
+
ordered.each do |r|
|
|
356
|
+
break if added >= need
|
|
357
|
+
|
|
358
|
+
judge = if r[:score]
|
|
359
|
+
r[:score].to_f.clamp(0.0, 1.0)
|
|
360
|
+
else
|
|
361
|
+
case r[:success]
|
|
362
|
+
when true, 'true' then 0.75
|
|
363
|
+
when 'soft' then 0.55
|
|
364
|
+
when false, 'false' then 0.25
|
|
365
|
+
else 0.5
|
|
366
|
+
end
|
|
367
|
+
end
|
|
368
|
+
proxy = case r[:success]
|
|
369
|
+
when true, 'true' then true
|
|
370
|
+
when false, 'false', 'soft' then false
|
|
371
|
+
else judge >= 0.6
|
|
372
|
+
end
|
|
373
|
+
record_sentinel(proxy: proxy, judge: judge)
|
|
374
|
+
added += 1
|
|
375
|
+
end
|
|
376
|
+
final_n = Array(load_sentinel[:window]).length
|
|
377
|
+
# Recompute distrust once window is full so controllers can engage.
|
|
378
|
+
snap = final_n >= SENTINEL_WINDOW ? sentinel : { samples: final_n, status: :insufficient }
|
|
379
|
+
{
|
|
380
|
+
added: added,
|
|
381
|
+
samples: final_n,
|
|
382
|
+
status: (final_n >= SENTINEL_WINDOW ? :warmed_full : :warmed_partial),
|
|
383
|
+
proxy_distrust: proxy_distrust,
|
|
384
|
+
sentinel: snap.is_a?(Hash) ? snap.slice(:samples, :status, :reward_hacked, :proxy_distrust, :proxy, :judge) : nil
|
|
385
|
+
}
|
|
386
|
+
rescue StandardError => e
|
|
387
|
+
{ added: 0, error: "#{e.class}: #{e.message}" }
|
|
388
|
+
end
|
|
389
|
+
|
|
300
390
|
# ----------------------------------------------------------------
|
|
301
391
|
# R4 — Structured tool-result classifier
|
|
302
392
|
# ----------------------------------------------------------------
|
|
@@ -377,6 +467,14 @@ module PWN
|
|
|
377
467
|
claim = final[Learning::CLAIM_RX] if defined?(Learning)
|
|
378
468
|
return nil if claim.to_s.empty?
|
|
379
469
|
|
|
470
|
+
# P26 — drop metric crumbs ("cap 0.2") that match loose patterns
|
|
471
|
+
if defined?(Learning) && Learning.respond_to?(:checkable_claim?, true)
|
|
472
|
+
return nil unless Learning.send(:checkable_claim?, claim: claim)
|
|
473
|
+
elsif claim.match?(/\A(?:cap|share|proxy|judge|success|only|now|gap|score|rate)\b/i) ||
|
|
474
|
+
(claim.match?(/\d+\.\d+/) && !claim.match?(/\d+\.\d+\.\d+|CVE-/i))
|
|
475
|
+
return nil
|
|
476
|
+
end
|
|
477
|
+
|
|
380
478
|
# 1.5 — sampled E3: always when flag true; never when false;
|
|
381
479
|
# nil/auto → always on frontier, ~10% on local when CLAIM_RX hits.
|
|
382
480
|
flag = agent_flag(key: :verify_as_reward, default: nil)
|
|
@@ -408,6 +506,29 @@ module PWN
|
|
|
408
506
|
# source: 'optional - :user_correction | :mistakes_resolve | :counterfactual | :critic'
|
|
409
507
|
# )
|
|
410
508
|
|
|
509
|
+
# Trajectory-shaped chosen sides that may land DPO without prose flood.
|
|
510
|
+
TRAJECTORY_SHAPES = %w[winning_trace revised_answer real_dispatch].freeze
|
|
511
|
+
|
|
512
|
+
# P9 — write-time source quota (not only export). Prefer diverse online
|
|
513
|
+
# generators over resolve-prose flood. Window is last WRITE_SOURCE_WINDOW
|
|
514
|
+
# pairs; a source already above WRITE_SOURCE_CAP is refused unless
|
|
515
|
+
# force: true (user_correction always forces).
|
|
516
|
+
WRITE_SOURCE_CAP = 0.40
|
|
517
|
+
WRITE_SOURCE_WINDOW = 100
|
|
518
|
+
|
|
519
|
+
# P0 — target online generator mix for W1. Gates alone cannot fill
|
|
520
|
+
# an empty promote; the controller must *prefer underfilled* sources
|
|
521
|
+
# (counterfactual / critic / curriculum / user_correction) when
|
|
522
|
+
# resolve already dominates. Shares are soft targets, not hard caps
|
|
523
|
+
# (hard cap remains WRITE_SOURCE_CAP). Trajectory-only still applies.
|
|
524
|
+
TARGET_SOURCE_MIX = {
|
|
525
|
+
'mistakes_resolve' => 0.30,
|
|
526
|
+
'curriculum' => 0.25,
|
|
527
|
+
'counterfactual' => 0.20,
|
|
528
|
+
'critic' => 0.15,
|
|
529
|
+
'user_correction' => 0.10
|
|
530
|
+
}.freeze
|
|
531
|
+
|
|
411
532
|
public_class_method def self.record_preference(opts = {})
|
|
412
533
|
prompt = opts[:prompt].to_s
|
|
413
534
|
rejected = opts[:rejected].to_s
|
|
@@ -415,20 +536,299 @@ module PWN
|
|
|
415
536
|
return nil if prompt.strip.empty? || chosen.strip.empty? || rejected.strip.empty?
|
|
416
537
|
return nil if chosen.strip == rejected.strip
|
|
417
538
|
|
|
539
|
+
# Reject weak pair geometry: CORRECTION: flaw-prose is not a trajectory.
|
|
540
|
+
return { skipped: :weak_pair_geometry, reason: 'chosen looks like flaw prose, not a revised answer/trace' } if chosen.match?(/\A\s*CORRECTION:\s*/i) && chosen.length < 400 && !opts[:force]
|
|
541
|
+
|
|
542
|
+
source = (opts[:source] || :unknown).to_s
|
|
543
|
+
shape = opts[:shape].to_s
|
|
544
|
+
# P25 — require trajectory shape at write time unless force / user_correction.
|
|
545
|
+
# Stops resolve-prose flood from ever landing in the ledger; export scrub
|
|
546
|
+
# is defense-in-depth, not the primary gate.
|
|
547
|
+
traj = TRAJECTORY_SHAPES.include?(shape)
|
|
548
|
+
# P25 — non-trajectory prose never lands (export scrub is defense-in-depth).
|
|
549
|
+
# user_correction and explicit force: still allowed for human / migration paths.
|
|
550
|
+
unless traj || opts[:force] || source == 'user_correction'
|
|
551
|
+
return {
|
|
552
|
+
skipped: :non_trajectory_shape,
|
|
553
|
+
reason: "shape=#{shape.inspect} not in #{TRAJECTORY_SHAPES.join(',')}; pass force:true or a trajectory shape",
|
|
554
|
+
source: source
|
|
555
|
+
}
|
|
556
|
+
end
|
|
557
|
+
# P9 — write-time source quota still applies to trajectory pairs.
|
|
558
|
+
# P25 made every auto-written row trajectory-shaped; if traj also
|
|
559
|
+
# bypassed the quota, resolve monoculture would return via winning_trace
|
|
560
|
+
# flood. Only user_correction and explicit force:true skip the cap.
|
|
561
|
+
bypass_quota = opts[:force] || source == 'user_correction'
|
|
562
|
+
unless bypass_quota
|
|
563
|
+
quota = write_source_quota(source: source)
|
|
564
|
+
return quota.merge(skipped: :source_quota) if quota[:over_cap]
|
|
565
|
+
end
|
|
566
|
+
|
|
418
567
|
entry = {
|
|
419
568
|
id: Digest::SHA256.hexdigest("#{prompt}|#{rejected}|#{chosen}")[0, 12],
|
|
420
569
|
prompt: prompt[0, 4_000],
|
|
421
570
|
rejected: rejected[0, 4_000],
|
|
422
571
|
chosen: chosen[0, 4_000],
|
|
423
|
-
source:
|
|
572
|
+
source: source,
|
|
424
573
|
engine: (PWN::Env.dig(:ai, :active) if defined?(PWN::Env)).to_s,
|
|
425
574
|
timestamp: Time.now.utc.iso8601
|
|
426
575
|
}
|
|
576
|
+
entry[:meta] = opts[:meta] if opts[:meta].is_a?(Hash)
|
|
577
|
+
entry[:shape] = opts[:shape].to_s if opts[:shape]
|
|
427
578
|
FileUtils.mkdir_p(File.dirname(PREFERENCES_FILE))
|
|
428
579
|
File.open(PREFERENCES_FILE, 'a') { |f| f.puts(JSON.generate(entry)) }
|
|
429
580
|
entry
|
|
430
581
|
end
|
|
431
582
|
|
|
583
|
+
# Share of `source` among the newest WRITE_SOURCE_WINDOW prefs.
|
|
584
|
+
public_class_method def self.write_source_quota(opts = {})
|
|
585
|
+
source = opts[:source].to_s
|
|
586
|
+
recent = preferences(limit: WRITE_SOURCE_WINDOW)
|
|
587
|
+
return { over_cap: false, share: 0.0, n: 0, window: recent.length, underfilled: true } if recent.length < 10
|
|
588
|
+
|
|
589
|
+
n = recent.count { |r| r[:source].to_s == source }
|
|
590
|
+
share = n.to_f / recent.length
|
|
591
|
+
target = TARGET_SOURCE_MIX[source]
|
|
592
|
+
{
|
|
593
|
+
over_cap: share > WRITE_SOURCE_CAP,
|
|
594
|
+
share: share.round(3),
|
|
595
|
+
n: n,
|
|
596
|
+
window: recent.length,
|
|
597
|
+
source: source,
|
|
598
|
+
cap: WRITE_SOURCE_CAP,
|
|
599
|
+
target: target,
|
|
600
|
+
underfilled: target ? share < (target * 0.5) : share < 0.05,
|
|
601
|
+
deficit: target ? (target - share).round(3) : nil
|
|
602
|
+
}
|
|
603
|
+
rescue StandardError
|
|
604
|
+
{ over_cap: false, share: 0.0, underfilled: true }
|
|
605
|
+
end
|
|
606
|
+
|
|
607
|
+
# P0 — online generator mix report + urgency flags. Controllers
|
|
608
|
+
# (auto_introspect, practice, counterfactual gate) consult this so
|
|
609
|
+
# underfilled sources get scheduling priority while over-cap
|
|
610
|
+
# resolve stops flooding. Returns {by_source, trajectory_fraction,
|
|
611
|
+
# urgent:[], suppress:[], healthy:}.
|
|
612
|
+
public_class_method def self.generator_mix(opts = {})
|
|
613
|
+
limit = opts[:limit] || WRITE_SOURCE_WINDOW
|
|
614
|
+
rows = preferences(limit: limit)
|
|
615
|
+
usable = rows.select { |r| usable_preference?(row: r) }
|
|
616
|
+
by = Hash.new(0)
|
|
617
|
+
usable.each { |r| by[r[:source].to_s] += 1 }
|
|
618
|
+
n = usable.length
|
|
619
|
+
shares = {}
|
|
620
|
+
TARGET_SOURCE_MIX.each_key { |k| shares[k] = n.zero? ? 0.0 : (by[k].to_f / n).round(3) }
|
|
621
|
+
by.each_key { |k| shares[k] ||= (by[k].to_f / n).round(3) }
|
|
622
|
+
|
|
623
|
+
traj_n = usable.count { |r| TRAJECTORY_SHAPES.include?(r[:shape].to_s) }
|
|
624
|
+
traj_f = n.zero? ? 0.0 : (traj_n.to_f / n).round(3)
|
|
625
|
+
|
|
626
|
+
urgent = []
|
|
627
|
+
suppress = []
|
|
628
|
+
TARGET_SOURCE_MIX.each do |src, target|
|
|
629
|
+
sh = shares[src].to_f
|
|
630
|
+
urgent << src if sh < (target * 0.5) && n >= 5
|
|
631
|
+
suppress << src if sh > WRITE_SOURCE_CAP && n >= 10
|
|
632
|
+
end
|
|
633
|
+
suppress << 'mistakes_resolve' if shares['mistakes_resolve'].to_f > WRITE_SOURCE_CAP && n >= 10 && !suppress.include?('mistakes_resolve')
|
|
634
|
+
|
|
635
|
+
healthy = urgent.empty? && suppress.empty? && traj_f >= 0.5 && n >= 10
|
|
636
|
+
{
|
|
637
|
+
n: n,
|
|
638
|
+
raw_n: rows.length,
|
|
639
|
+
by_source: by,
|
|
640
|
+
shares: shares,
|
|
641
|
+
targets: TARGET_SOURCE_MIX,
|
|
642
|
+
trajectory_fraction: traj_f,
|
|
643
|
+
urgent: urgent.uniq,
|
|
644
|
+
suppress: suppress.uniq,
|
|
645
|
+
healthy: healthy,
|
|
646
|
+
recommendation: if healthy
|
|
647
|
+
'mix_ok'
|
|
648
|
+
elsif n < 10
|
|
649
|
+
'need_more_pairs'
|
|
650
|
+
elsif traj_f < 0.5
|
|
651
|
+
'need_trajectory_shape'
|
|
652
|
+
elsif urgent.any?
|
|
653
|
+
"boost:#{urgent.join(',')}"
|
|
654
|
+
else
|
|
655
|
+
"suppress:#{suppress.join(',')}"
|
|
656
|
+
end
|
|
657
|
+
}
|
|
658
|
+
rescue StandardError => e
|
|
659
|
+
{
|
|
660
|
+
n: 0, healthy: false, error: "#{e.class}: #{e.message}",
|
|
661
|
+
urgent: %w[curriculum counterfactual critic user_correction],
|
|
662
|
+
suppress: []
|
|
663
|
+
}
|
|
664
|
+
end
|
|
665
|
+
|
|
666
|
+
# P0 ops — infer trajectory shape for legacy ledger rows that predate
|
|
667
|
+
# P21/P25 shape tags. Used by scrub_preferences rewrite so generator_mix
|
|
668
|
+
# trajectory_fraction reflects content, not missing keys.
|
|
669
|
+
public_class_method def self.infer_shape(opts = {})
|
|
670
|
+
r = opts.is_a?(Hash) && opts.key?(:row) ? opts[:row] : opts
|
|
671
|
+
r = r.transform_keys(&:to_sym) if r.respond_to?(:transform_keys)
|
|
672
|
+
existing = r[:shape].to_s
|
|
673
|
+
return existing if TRAJECTORY_SHAPES.include?(existing) || existing == 'fix_prose'
|
|
674
|
+
|
|
675
|
+
chosen = r[:chosen].to_s
|
|
676
|
+
source = r[:source].to_s
|
|
677
|
+
# tool-call / trace markers → winning_trace
|
|
678
|
+
if chosen.match?(/\b(shell|pwn_eval|memory_|sessions_|reward_|curriculum_|extro_|mistakes_)\b/i) &&
|
|
679
|
+
(chosen.include?('→') || chosen.include?('tool_call') || chosen.include?('"name"') ||
|
|
680
|
+
chosen.lines.count { |l| l.strip.start_with?('{') || l.include?('arguments') } >= 1)
|
|
681
|
+
return 'winning_trace'
|
|
682
|
+
end
|
|
683
|
+
# long revised answer from critic / user / CF → revised_answer
|
|
684
|
+
return 'revised_answer' if chosen.length >= 200 && %w[critic user_correction counterfactual curriculum].include?(source)
|
|
685
|
+
|
|
686
|
+
# counterfactual real dispatch tag in meta
|
|
687
|
+
meta = r[:meta].is_a?(Hash) ? r[:meta] : {}
|
|
688
|
+
return 'real_dispatch' if meta[:mode].to_s == 'real_dispatch' || meta['mode'].to_s == 'real_dispatch'
|
|
689
|
+
|
|
690
|
+
existing.empty? ? nil : existing
|
|
691
|
+
rescue StandardError
|
|
692
|
+
nil
|
|
693
|
+
end
|
|
694
|
+
|
|
695
|
+
# P15 — keep only usable preference pairs for balance/export/promote.
|
|
696
|
+
# Drops CORRECTION-only chosen, resolve rows without trajectory shape,
|
|
697
|
+
# and chosen≪rejected unless shape is a known trajectory form.
|
|
698
|
+
public_class_method def self.usable_preference?(opts = {})
|
|
699
|
+
r = opts.is_a?(Hash) && opts.key?(:row) ? opts[:row] : opts
|
|
700
|
+
r = r.transform_keys(&:to_sym) if r.respond_to?(:transform_keys)
|
|
701
|
+
chosen = r[:chosen].to_s
|
|
702
|
+
rejected = r[:rejected].to_s
|
|
703
|
+
shape = r[:shape].to_s
|
|
704
|
+
source = r[:source].to_s
|
|
705
|
+
return false if chosen.strip.empty? || rejected.strip.empty?
|
|
706
|
+
return false if chosen.strip == rejected.strip
|
|
707
|
+
return false if chosen.match?(/\A\s*CORRECTION:\s*/i) && chosen.length < 400
|
|
708
|
+
return false if shape == 'fix_prose'
|
|
709
|
+
# P25 — resolve rows must be trajectory-shaped to count as usable
|
|
710
|
+
return false if source == 'mistakes_resolve' && !TRAJECTORY_SHAPES.include?(shape)
|
|
711
|
+
|
|
712
|
+
# chosen ≪ rejected without trajectory shape → commentary, not policy
|
|
713
|
+
unless TRAJECTORY_SHAPES.include?(shape)
|
|
714
|
+
return false if rejected.length >= 200 && chosen.length < (rejected.length * 0.25) && chosen.length < 200
|
|
715
|
+
return false if rejected.length >= 400 && chosen.length < 120
|
|
716
|
+
end
|
|
717
|
+
true
|
|
718
|
+
rescue StandardError
|
|
719
|
+
false
|
|
720
|
+
end
|
|
721
|
+
|
|
722
|
+
# P15 — one-shot ledger hygiene. Filters in place (rewrite jsonl) or
|
|
723
|
+
# report-only. Returns {before:, after:, dropped:, by_reason:, path:}.
|
|
724
|
+
public_class_method def self.scrub_preferences(opts = {})
|
|
725
|
+
dry = opts.key?(:dry_run) ? opts[:dry_run] : false
|
|
726
|
+
path = PREFERENCES_FILE
|
|
727
|
+
return { before: 0, after: 0, dropped: 0, dry_run: dry, path: path } unless File.exist?(path)
|
|
728
|
+
|
|
729
|
+
raw = File.readlines(path)
|
|
730
|
+
kept = []
|
|
731
|
+
reasons = Hash.new(0)
|
|
732
|
+
raw.each do |line|
|
|
733
|
+
begin
|
|
734
|
+
r = JSON.parse(line, symbolize_names: true)
|
|
735
|
+
rescue StandardError
|
|
736
|
+
reasons[:parse_error] += 1
|
|
737
|
+
next
|
|
738
|
+
end
|
|
739
|
+
if usable_preference?(row: r)
|
|
740
|
+
# P0 ops — backfill shape so trajectory_fraction is meaningful
|
|
741
|
+
if r[:shape].to_s.empty?
|
|
742
|
+
inferred = infer_shape(row: r)
|
|
743
|
+
r = r.merge(shape: inferred) if inferred
|
|
744
|
+
end
|
|
745
|
+
kept << r
|
|
746
|
+
else
|
|
747
|
+
why = if r[:chosen].to_s.match?(/\A\s*CORRECTION:\s*/i)
|
|
748
|
+
:correction_prose
|
|
749
|
+
elsif r[:shape].to_s == 'fix_prose'
|
|
750
|
+
:fix_prose
|
|
751
|
+
elsif r[:chosen].to_s.length < (r[:rejected].to_s.length * 0.25)
|
|
752
|
+
:chosen_too_short
|
|
753
|
+
else
|
|
754
|
+
:weak_geometry
|
|
755
|
+
end
|
|
756
|
+
reasons[why] += 1
|
|
757
|
+
end
|
|
758
|
+
end
|
|
759
|
+
unless dry
|
|
760
|
+
bak = "#{path}.bak-p15-#{Time.now.utc.strftime('%Y%m%d%H%M%S')}"
|
|
761
|
+
FileUtils.cp(path, bak)
|
|
762
|
+
File.open(path, 'w') { |f| kept.each { |r| f.puts(JSON.generate(r)) } }
|
|
763
|
+
end
|
|
764
|
+
{
|
|
765
|
+
before: raw.length,
|
|
766
|
+
after: kept.length,
|
|
767
|
+
dropped: raw.length - kept.length,
|
|
768
|
+
by_reason: reasons,
|
|
769
|
+
dry_run: dry,
|
|
770
|
+
path: path,
|
|
771
|
+
backup: dry ? nil : bak
|
|
772
|
+
}
|
|
773
|
+
rescue StandardError => e
|
|
774
|
+
{ error: "#{e.class}: #{e.message}" }
|
|
775
|
+
end
|
|
776
|
+
|
|
777
|
+
# P15/P5 — geometry-aware source mix. scrub:true uses usable_preference?
|
|
778
|
+
# so operators see the post-hygiene diet (what export_dpo will train on).
|
|
779
|
+
public_class_method def self.preference_balance(opts = {})
|
|
780
|
+
limit = opts[:limit] || 10_000
|
|
781
|
+
scrub = opts.key?(:scrub) ? opts[:scrub] : false
|
|
782
|
+
rows = preferences(limit: limit)
|
|
783
|
+
before = rows.length
|
|
784
|
+
rows = rows.select { |r| usable_preference?(row: r) } if scrub
|
|
785
|
+
by = Hash.new(0)
|
|
786
|
+
by_shape = Hash.new(0)
|
|
787
|
+
rows.each do |r|
|
|
788
|
+
by[r[:source].to_s] += 1
|
|
789
|
+
sh = r[:shape].to_s
|
|
790
|
+
sh = 'unspecified' if sh.empty?
|
|
791
|
+
by_shape[sh] += 1
|
|
792
|
+
end
|
|
793
|
+
total = rows.length
|
|
794
|
+
frac = by.transform_values { |n| total.zero? ? 0.0 : (n.to_f / total).round(3) }
|
|
795
|
+
shape_frac = by_shape.transform_values { |n| total.zero? ? 0.0 : (n.to_f / total).round(3) }
|
|
796
|
+
traj_n = rows.count { |r| TRAJECTORY_SHAPES.include?(r[:shape].to_s) }
|
|
797
|
+
traj_frac = total.zero? ? 0.0 : (traj_n.to_f / total).round(3)
|
|
798
|
+
monoculture = total.positive? && (by.values.max.to_f / total) > 0.7
|
|
799
|
+
mix = begin
|
|
800
|
+
generator_mix(limit: limit)
|
|
801
|
+
rescue StandardError
|
|
802
|
+
nil
|
|
803
|
+
end
|
|
804
|
+
{
|
|
805
|
+
total: before,
|
|
806
|
+
kept: total,
|
|
807
|
+
scrubbed: scrub,
|
|
808
|
+
dropped: before - total,
|
|
809
|
+
by_source: by,
|
|
810
|
+
fractions: frac,
|
|
811
|
+
by_shape: by_shape,
|
|
812
|
+
by_shape_fraction: shape_frac,
|
|
813
|
+
trajectory_fraction: traj_frac,
|
|
814
|
+
monoculture: monoculture,
|
|
815
|
+
generator_mix: mix,
|
|
816
|
+
advice: if total < 12
|
|
817
|
+
'W1 thin: need more trajectory-shaped pairs before LoRA promote.'
|
|
818
|
+
elsif monoculture
|
|
819
|
+
'W1 monoculture: run Reward.scrub_preferences; enable :counterfactual/:critic; stop resolve-prose flood.'
|
|
820
|
+
elsif traj_frac < 0.30
|
|
821
|
+
'W1 geometry weak: <30% trajectory-shaped chosen sides — DPO would teach commentary.'
|
|
822
|
+
elsif mix && !mix[:healthy]
|
|
823
|
+
"W1 generator mix: #{mix[:recommendation]}"
|
|
824
|
+
else
|
|
825
|
+
'W1 source mix OK for gated export'
|
|
826
|
+
end
|
|
827
|
+
}
|
|
828
|
+
rescue StandardError => e
|
|
829
|
+
{ error: "#{e.class}: #{e.message}" }
|
|
830
|
+
end
|
|
831
|
+
|
|
432
832
|
# Supported Method Parameters::
|
|
433
833
|
# rows = PWN::AI::Agent::Reward.preferences(limit: 500, source: nil)
|
|
434
834
|
|
|
@@ -463,6 +863,15 @@ module PWN
|
|
|
463
863
|
FileUtils.mkdir_p(DPO_DIR)
|
|
464
864
|
out = opts[:out] || File.join(DPO_DIR, "pwn-dpo-#{Time.now.utc.strftime('%Y%m%d')}.jsonl")
|
|
465
865
|
rows = preferences(limit: 100_000)
|
|
866
|
+
# P15 — drop weak geometry before source-cap so resolve prose cannot
|
|
867
|
+
# dominate the kept set after balance. opt-out with scrub: false.
|
|
868
|
+
scrub = opts.key?(:scrub) ? opts[:scrub] : true
|
|
869
|
+
geometry_dropped = 0
|
|
870
|
+
if scrub
|
|
871
|
+
usable = rows.select { |r| usable_preference?(row: r) }
|
|
872
|
+
geometry_dropped = rows.length - usable.length
|
|
873
|
+
rows = usable
|
|
874
|
+
end
|
|
466
875
|
# P5 — downsample so no single source exceeds DPO_SOURCE_CAP of the export.
|
|
467
876
|
# opt-out with balance: false (raw dump for diagnostics).
|
|
468
877
|
balance = opts.key?(:balance) ? opts[:balance] : true
|
|
@@ -484,8 +893,14 @@ module PWN
|
|
|
484
893
|
by_src = selected.group_by { |r| r[:source].to_s }.transform_values(&:length)
|
|
485
894
|
{
|
|
486
895
|
path: out, format: fmt, pairs: selected.length, bytes: File.size(out),
|
|
487
|
-
balanced: balance, dropped: dropped,
|
|
488
|
-
|
|
896
|
+
balanced: balance, dropped: dropped, geometry_dropped: geometry_dropped,
|
|
897
|
+
scrubbed: scrub, by_source: by_src,
|
|
898
|
+
source_cap: balance ? (opts[:source_cap] || DPO_SOURCE_CAP).to_f : nil,
|
|
899
|
+
preference_balance: begin
|
|
900
|
+
preference_balance(limit: 10_000, scrub: true)
|
|
901
|
+
rescue StandardError
|
|
902
|
+
nil
|
|
903
|
+
end
|
|
489
904
|
}
|
|
490
905
|
end
|
|
491
906
|
|
|
@@ -615,7 +1030,11 @@ module PWN
|
|
|
615
1030
|
j = JSON.parse(l, symbolize_names: true)
|
|
616
1031
|
if j[:role].to_s == 'tool'
|
|
617
1032
|
ti += 1
|
|
618
|
-
|
|
1033
|
+
if rewards[ti]
|
|
1034
|
+
j[:step_reward] = rewards[ti]
|
|
1035
|
+
# P18 — fold step_reward into Metrics so Registry.rank can bias
|
|
1036
|
+
fold_step_reward_to_metrics(content: j[:content], reward: rewards[ti])
|
|
1037
|
+
end
|
|
619
1038
|
end
|
|
620
1039
|
"#{JSON.generate(j)}\n"
|
|
621
1040
|
rescue StandardError
|
|
@@ -626,12 +1045,29 @@ module PWN
|
|
|
626
1045
|
nil
|
|
627
1046
|
end
|
|
628
1047
|
|
|
1048
|
+
# Extract tool name from "name → …" session content and record PRM.
|
|
1049
|
+
private_class_method def self.fold_step_reward_to_metrics(opts = {})
|
|
1050
|
+
return unless defined?(Metrics) && Metrics.respond_to?(:record_step_reward)
|
|
1051
|
+
|
|
1052
|
+
content = opts[:content].to_s
|
|
1053
|
+
name = content[/\A([a-z_][a-z0-9_]*)\s*→/i, 1] ||
|
|
1054
|
+
content[/\A([a-z_][a-z0-9_]*)\s*->/i, 1] ||
|
|
1055
|
+
content[/\A([a-z_][a-z0-9_]*)/, 1]
|
|
1056
|
+
return if name.to_s.empty?
|
|
1057
|
+
|
|
1058
|
+
Metrics.record_step_reward(name: name, reward: opts[:reward])
|
|
1059
|
+
rescue StandardError
|
|
1060
|
+
nil
|
|
1061
|
+
end
|
|
1062
|
+
|
|
629
1063
|
private_class_method def self.record_sentinel(opts = {})
|
|
630
1064
|
s = normalize_sentinel(raw: load_sentinel)
|
|
631
1065
|
# Clamp judge to [0,1] — LLM/heuristic should already, but a bad
|
|
632
1066
|
# write must not poison rolling means forever.
|
|
633
1067
|
judge = opts[:judge].to_f.clamp(0.0, 1.0)
|
|
634
1068
|
entry = { judge: judge, at: Time.now.utc.iso8601 }
|
|
1069
|
+
# P1 — optional per-sample confidence (heuristic < LLM ORM)
|
|
1070
|
+
entry[:confidence] = opts[:confidence].to_f.clamp(0.0, 1.0) unless opts[:confidence].nil?
|
|
635
1071
|
# 1.3 — only roll proxy into the window when the caller actually
|
|
636
1072
|
# supplied a R4-aligned proxy_ok. Pre-ORM boolean noise no longer
|
|
637
1073
|
# dilutes gap_proxy_judge. Proxy is ALWAYS 0.0 or 1.0 when present.
|
|
@@ -872,13 +1308,17 @@ module PWN
|
|
|
872
1308
|
PWN::AI::Agent::Reward.prm(request: req, session_id: sid) # R2 PRM → per-step credit
|
|
873
1309
|
PWN::AI::Agent::Reward.sentinel # R3 reward-hacking detector
|
|
874
1310
|
PWN::AI::Agent::Reward.reset_sentinel # wipe corrupt window + distrust
|
|
1311
|
+
PWN::AI::Agent::Reward.warm_sentinel # P10 fill R3 window from Learning outcomes
|
|
875
1312
|
PWN::AI::Agent::Reward.semantic_ok(name: 'shell', raw: json, args: args) # R4 kills phantom exit≠0 mistakes
|
|
876
1313
|
|
|
877
1314
|
# Tier 5 — preference pairs → DPO
|
|
878
1315
|
PWN::AI::Agent::Reward.record_preference(prompt: p, rejected: r, chosen: c, source: :user_correction)
|
|
879
1316
|
PWN::AI::Agent::Reward.preferences(limit: 100)
|
|
880
|
-
PWN::AI::Agent::Reward.export_dpo(format: :dpo) # W1 → ~/.pwn/finetune/pwn-dpo-*.jsonl (≤40%/source)
|
|
1317
|
+
PWN::AI::Agent::Reward.export_dpo(format: :dpo) # W1 → ~/.pwn/finetune/pwn-dpo-*.jsonl (≤40%/source, scrubbed)
|
|
881
1318
|
PWN::AI::Agent::Reward.export_dpo(format: :dpo, balance: false) # raw dump (diagnostics)
|
|
1319
|
+
PWN::AI::Agent::Reward.scrub_preferences(dry_run: true) # P15 ledger hygiene report
|
|
1320
|
+
PWN::AI::Agent::Reward.scrub_preferences # P15 rewrite jsonl (backup first)
|
|
1321
|
+
PWN::AI::Agent::Reward.preference_balance(scrub: true) # P15 geometry-aware mix
|
|
882
1322
|
|
|
883
1323
|
# Tier 6 — grounded reward
|
|
884
1324
|
PWN::AI::Agent::Reward.verify_as_reward(final: text) # E3 browser-verified reward
|
|
@@ -144,3 +144,26 @@ PWN::AI::Agent::Registry.register(
|
|
|
144
144
|
check: -> { defined?(PWN::AI::Agent::Curriculum) && PWN::AI::Agent::Curriculum.respond_to?(:preference_balance) },
|
|
145
145
|
handler: ->(args) { PWN::AI::Agent::Curriculum.preference_balance(limit: args[:limit]) }
|
|
146
146
|
)
|
|
147
|
+
|
|
148
|
+
PWN::AI::Agent::Registry.register(
|
|
149
|
+
name: 'curriculum_practice_kpi',
|
|
150
|
+
toolset: 'curriculum',
|
|
151
|
+
schema: {
|
|
152
|
+
name: 'curriculum_practice_kpi',
|
|
153
|
+
description: 'P1 — Outer curriculum KPI: unresolved [REPEATING] count + week-over-week trend (improving|flat|regressing). Snapshots land in ~/.pwn/curriculum_kpi.jsonl from practice().',
|
|
154
|
+
parameters: {
|
|
155
|
+
type: 'object',
|
|
156
|
+
properties: {
|
|
157
|
+
limit: { type: 'integer', description: 'Trend window snapshots (default 14)' }
|
|
158
|
+
},
|
|
159
|
+
required: []
|
|
160
|
+
}
|
|
161
|
+
},
|
|
162
|
+
check: -> { defined?(PWN::AI::Agent::Curriculum) && PWN::AI::Agent::Curriculum.respond_to?(:repeating_trend) },
|
|
163
|
+
handler: lambda { |args|
|
|
164
|
+
{
|
|
165
|
+
trend: PWN::AI::Agent::Curriculum.repeating_trend(limit: args[:limit]),
|
|
166
|
+
snapshot: PWN::AI::Agent::Curriculum.practice_kpi(results: [])
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
)
|
|
@@ -126,3 +126,89 @@ PWN::AI::Agent::Registry.register(
|
|
|
126
126
|
)
|
|
127
127
|
}
|
|
128
128
|
)
|
|
129
|
+
|
|
130
|
+
PWN::AI::Agent::Registry.register(
|
|
131
|
+
name: 'reward_warm_sentinel',
|
|
132
|
+
toolset: 'learning',
|
|
133
|
+
schema: {
|
|
134
|
+
name: 'reward_warm_sentinel',
|
|
135
|
+
description: 'P10 — Backfill the R3 sentinel ring buffer from Learning ' \
|
|
136
|
+
'outcomes so proxy_distrust can engage on local/offline hosts ' \
|
|
137
|
+
'without waiting for live remote introspect. Only fills empty ' \
|
|
138
|
+
'slots; never flushes a warm window.',
|
|
139
|
+
parameters: {
|
|
140
|
+
type: 'object',
|
|
141
|
+
properties: {
|
|
142
|
+
limit: { type: 'integer', default: 120, description: 'Max outcomes to scan.' }
|
|
143
|
+
},
|
|
144
|
+
required: []
|
|
145
|
+
}
|
|
146
|
+
},
|
|
147
|
+
check: -> { defined?(PWN::AI::Agent::Reward) && PWN::AI::Agent::Reward.respond_to?(:warm_sentinel) },
|
|
148
|
+
handler: ->(args) { PWN::AI::Agent::Reward.warm_sentinel(limit: args[:limit] || 120) }
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
PWN::AI::Agent::Registry.register(
|
|
152
|
+
name: 'reward_scrub_preferences',
|
|
153
|
+
toolset: 'learning',
|
|
154
|
+
schema: {
|
|
155
|
+
name: 'reward_scrub_preferences',
|
|
156
|
+
description: 'P15 — One-shot W1 ledger hygiene. Drops CORRECTION:-only chosen, resolve rows without trajectory shape, and chosen≪rejected pairs. dry_run:true reports; false rewrites ~/.pwn/preferences.jsonl (backup first).',
|
|
157
|
+
parameters: {
|
|
158
|
+
type: 'object',
|
|
159
|
+
properties: {
|
|
160
|
+
dry_run: { type: 'boolean', default: true, description: 'Report only (default true).' }
|
|
161
|
+
},
|
|
162
|
+
required: []
|
|
163
|
+
}
|
|
164
|
+
},
|
|
165
|
+
check: -> { defined?(PWN::AI::Agent::Reward) && PWN::AI::Agent::Reward.respond_to?(:scrub_preferences) },
|
|
166
|
+
handler: lambda { |args|
|
|
167
|
+
dry = args.key?(:dry_run) ? args[:dry_run] : true
|
|
168
|
+
PWN::AI::Agent::Reward.scrub_preferences(dry_run: dry)
|
|
169
|
+
}
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
PWN::AI::Agent::Registry.register(
|
|
173
|
+
name: 'reward_preference_balance',
|
|
174
|
+
toolset: 'learning',
|
|
175
|
+
schema: {
|
|
176
|
+
name: 'reward_preference_balance',
|
|
177
|
+
description: 'P15/P5 — Geometry-aware W1 diversity report. scrub:true scores the post-hygiene diet (trajectory_fraction, by_shape) so promote/export gates see usable pairs only.',
|
|
178
|
+
parameters: {
|
|
179
|
+
type: 'object',
|
|
180
|
+
properties: {
|
|
181
|
+
limit: { type: 'integer', default: 10_000 },
|
|
182
|
+
scrub: { type: 'boolean', default: true }
|
|
183
|
+
},
|
|
184
|
+
required: []
|
|
185
|
+
}
|
|
186
|
+
},
|
|
187
|
+
check: -> { defined?(PWN::AI::Agent::Reward) && PWN::AI::Agent::Reward.respond_to?(:preference_balance) },
|
|
188
|
+
handler: lambda { |args|
|
|
189
|
+
PWN::AI::Agent::Reward.preference_balance(
|
|
190
|
+
limit: args[:limit] || 10_000,
|
|
191
|
+
scrub: args.key?(:scrub) ? args[:scrub] : true
|
|
192
|
+
)
|
|
193
|
+
}
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
PWN::AI::Agent::Registry.register(
|
|
197
|
+
name: 'reward_generator_mix',
|
|
198
|
+
toolset: 'reward',
|
|
199
|
+
schema: {
|
|
200
|
+
name: 'reward_generator_mix',
|
|
201
|
+
description: 'P0 — Online W1 generator mix vs TARGET_SOURCE_MIX. Returns urgent/suppress sources, trajectory_fraction, recommendation (boost:… / suppress:…). Controllers use this to schedule counterfactual/critic/curriculum over resolve flood.',
|
|
202
|
+
parameters: {
|
|
203
|
+
type: 'object',
|
|
204
|
+
properties: {
|
|
205
|
+
limit: { type: 'integer', description: 'Window size (default WRITE_SOURCE_WINDOW)' }
|
|
206
|
+
},
|
|
207
|
+
required: []
|
|
208
|
+
}
|
|
209
|
+
},
|
|
210
|
+
check: -> { defined?(PWN::AI::Agent::Reward) && PWN::AI::Agent::Reward.respond_to?(:generator_mix) },
|
|
211
|
+
handler: lambda { |args|
|
|
212
|
+
PWN::AI::Agent::Reward.generator_mix(limit: args[:limit])
|
|
213
|
+
}
|
|
214
|
+
)
|