pwn 0.5.643 → 0.5.650
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/documentation/Agent-Tool-Registry.md +10 -5
- data/documentation/Cron.md +3 -2
- data/documentation/Home.md +2 -2
- data/documentation/How-PWN-Works.md +1 -1
- data/documentation/Mistakes.md +13 -0
- data/documentation/Reinforcement-Learning.md +72 -3
- data/documentation/Skills-Memory-Learning.md +5 -3
- data/documentation/What-is-PWN.md +1 -1
- data/documentation/diagrams/dot/pwn-ai-feedback-learning-loop.dot +3 -3
- data/documentation/diagrams/dot/reinforcement-learning.dot +8 -8
- data/documentation/diagrams/pwn-ai-feedback-learning-loop.svg +318 -319
- data/documentation/diagrams/reinforcement-learning.svg +188 -187
- data/documentation/pwn-ai-Agent.md +6 -5
- data/lib/pwn/ai/agent/curriculum.rb +491 -35
- data/lib/pwn/ai/agent/extrospection.rb +18 -4
- data/lib/pwn/ai/agent/learning.rb +308 -54
- data/lib/pwn/ai/agent/loop.rb +145 -11
- data/lib/pwn/ai/agent/metrics.rb +200 -11
- data/lib/pwn/ai/agent/mistakes.rb +26 -8
- data/lib/pwn/ai/agent/registry.rb +24 -2
- data/lib/pwn/ai/agent/reward.rb +447 -7
- data/lib/pwn/ai/agent/tools/curriculum.rb +23 -0
- data/lib/pwn/ai/agent/tools/reward.rb +86 -0
- data/lib/pwn/ai/agent/tools/ruby_eval.rb +29 -6
- data/lib/pwn/cron.rb +14 -2
- data/lib/pwn/version.rb +1 -1
- data/spec/integration/reinforced_feedback_loop_spec.rb +806 -12
- data/spec/lib/pwn/ai/agent/reward_spec.rb +64 -1
- data/spec/lib/pwn/ai/agent/tools/ruby_eval_spec.rb +39 -1
- data/third_party/pwn_rdoc.jsonl +33 -4
- metadata +1 -1
|
@@ -94,6 +94,15 @@ module PWN
|
|
|
94
94
|
[]
|
|
95
95
|
end
|
|
96
96
|
cool = load_cooldown
|
|
97
|
+
# P17 — prefer budget-exhaustion fingerprints (agent_loop / critic)
|
|
98
|
+
# so nightly self-play attacks the #1 live skill gap first.
|
|
99
|
+
candidates = candidates.sort_by do |m|
|
|
100
|
+
t = m[:tool].to_s
|
|
101
|
+
e = m[:error].to_s.downcase
|
|
102
|
+
budget = t == 'agent_loop' || t == 'assistant_answer' ||
|
|
103
|
+
e.include?('budget exhausted') || e.include?('iteration budget')
|
|
104
|
+
[budget ? 0 : 1, -m[:count].to_i]
|
|
105
|
+
end
|
|
97
106
|
targets = candidates.reject { |m| practice_skip?(mistake: m, cooldown: cool) }.first(limit)
|
|
98
107
|
results = []
|
|
99
108
|
|
|
@@ -107,20 +116,68 @@ module PWN
|
|
|
107
116
|
mean = runs.empty? ? 0.0 : (runs.sum { |r| r[:score].to_f } / runs.length)
|
|
108
117
|
resolved = false
|
|
109
118
|
# 2.4 — auto-resolve only with N≥2 holdout successes + store trace
|
|
119
|
+
# P23 — auto-resolve only with N≥2 holdouts at judge≥0.7 AND a
|
|
120
|
+
# real tool trace (not empty-final luck). Budget fingerprints
|
|
121
|
+
# additionally require mean holdout ≥0.7 and short-horizon tags.
|
|
110
122
|
if solved.length >= 2 && defined?(Mistakes)
|
|
111
123
|
best = solved.max_by { |r| r[:score] }
|
|
124
|
+
winning = best[:trace].to_s.strip
|
|
125
|
+
winning = best[:final].to_s.strip if winning.length < 20
|
|
126
|
+
budgetish = %w[agent_loop assistant_answer].include?(m[:tool].to_s) ||
|
|
127
|
+
m[:error].to_s.downcase.include?('budget')
|
|
128
|
+
# refuse resolve on budget targets when winning_trace is prose-only
|
|
129
|
+
trace_ok = winning.length >= 20 && (
|
|
130
|
+
!budgetish || winning.match?(/→|shell|pwn_eval|tool/i) || best[:final].to_s.length.between?(1, 800)
|
|
131
|
+
)
|
|
132
|
+
unless trace_ok
|
|
133
|
+
bump_cooldown!(cooldown: cool, signature: m[:signature], mean: mean) unless dry_run
|
|
134
|
+
results << {
|
|
135
|
+
signature: m[:signature], tool: m[:tool], prompts: prompts,
|
|
136
|
+
runs: runs.map { |r| { score: r[:score], verdict: r[:verdict] } },
|
|
137
|
+
resolved: false, mean_score: mean.round(3),
|
|
138
|
+
reason: 'holdouts_ok_but_trace_weak'
|
|
139
|
+
}
|
|
140
|
+
next
|
|
141
|
+
end
|
|
112
142
|
fix = best[:final].to_s.lines.first(3).join.strip[0, 400]
|
|
113
143
|
Mistakes.resolve(
|
|
114
144
|
signature: m[:signature],
|
|
115
145
|
fix: "auto-curriculum: #{fix}",
|
|
116
146
|
structured: {
|
|
117
|
-
strategy: 'curriculum_practice',
|
|
147
|
+
strategy: budgetish ? 'short_horizon_finish' : 'curriculum_practice',
|
|
118
148
|
tool: m[:tool],
|
|
119
149
|
holdout_tests: solved.map { |r| r[:prompt] || r[:request] }.compact.first(5),
|
|
120
|
-
winning_trace:
|
|
150
|
+
winning_trace: winning[0, 2_000]
|
|
121
151
|
}
|
|
122
152
|
)
|
|
123
|
-
|
|
153
|
+
# P14 — trajectory-shaped curriculum pair (not first-3-lines fix prose).
|
|
154
|
+
# Mistakes.resolve also lands a mistakes_resolve pair via winning_trace;
|
|
155
|
+
# this :curriculum row keeps W1 source diversity honest.
|
|
156
|
+
if defined?(Reward)
|
|
157
|
+
rejected = m[:snippet].to_s
|
|
158
|
+
rejected = "FAILING: tool=#{m[:tool]} err=#{m[:error]}" if rejected.strip.empty?
|
|
159
|
+
chosen = if best[:trace].to_s.strip.length >= 20
|
|
160
|
+
parts = []
|
|
161
|
+
parts << "STRATEGY: curriculum_practice | #{m[:tool]}"
|
|
162
|
+
parts << "WINNING_TRACE:\n#{best[:trace].to_s[0, 3_000]}"
|
|
163
|
+
parts << "FINAL:\n#{best[:final].to_s[0, 800]}" unless best[:final].to_s.strip.empty?
|
|
164
|
+
parts.join("\n")
|
|
165
|
+
else
|
|
166
|
+
best[:final].to_s[0, 3_500]
|
|
167
|
+
end
|
|
168
|
+
Reward.record_preference(
|
|
169
|
+
prompt: (best[:prompt] || prompts.first).to_s,
|
|
170
|
+
rejected: rejected[0, 2_000],
|
|
171
|
+
chosen: chosen,
|
|
172
|
+
source: :curriculum,
|
|
173
|
+
shape: :winning_trace,
|
|
174
|
+
meta: {
|
|
175
|
+
signature: m[:signature],
|
|
176
|
+
score: best[:score],
|
|
177
|
+
holdouts: solved.length
|
|
178
|
+
}
|
|
179
|
+
)
|
|
180
|
+
end
|
|
124
181
|
resolved = true
|
|
125
182
|
cool.delete(m[:signature].to_s)
|
|
126
183
|
elsif !dry_run
|
|
@@ -136,12 +193,15 @@ module PWN
|
|
|
136
193
|
end
|
|
137
194
|
save_cooldown(cooldown: cool)
|
|
138
195
|
log(event: :practice, data: results)
|
|
196
|
+
kpi = practice_kpi(results: results)
|
|
139
197
|
{
|
|
140
198
|
practiced: results.length,
|
|
141
199
|
resolved: results.count { |r| r[:resolved] },
|
|
142
200
|
skipped_cooldown: cool.count { |_, v| v[:fail_nights].to_i >= COOLDOWN_FAIL_NIGHTS },
|
|
143
201
|
results: results,
|
|
144
|
-
dry_run: dry_run
|
|
202
|
+
dry_run: dry_run,
|
|
203
|
+
kpi: kpi,
|
|
204
|
+
generator_mix: (defined?(Reward) && Reward.respond_to?(:generator_mix) ? Reward.generator_mix : nil)
|
|
145
205
|
}
|
|
146
206
|
rescue StandardError => e
|
|
147
207
|
{ error: "#{e.class}: #{e.message}" }
|
|
@@ -245,8 +305,24 @@ module PWN
|
|
|
245
305
|
end
|
|
246
306
|
|
|
247
307
|
mean = scored.empty? ? nil : (scored.sum { |r| r[:score].to_f } / scored.length).round(3)
|
|
248
|
-
|
|
249
|
-
|
|
308
|
+
# P10 — keep R3 window warm so proxy_distrust can engage on local hosts
|
|
309
|
+
warm = (Reward.warm_sentinel(limit: 120) if commit && defined?(Reward) && Reward.respond_to?(:warm_sentinel))
|
|
310
|
+
# P0 ops — nightly ledger hygiene so generator_mix success criteria can
|
|
311
|
+
# fire: drop prose flood, backfill missing shapes, then snapshot mix+KPI.
|
|
312
|
+
scrub = nil
|
|
313
|
+
mix = nil
|
|
314
|
+
kpi = nil
|
|
315
|
+
if commit && defined?(Reward)
|
|
316
|
+
scrub = Reward.scrub_preferences(dry_run: false) if Reward.respond_to?(:scrub_preferences)
|
|
317
|
+
mix = Reward.generator_mix if Reward.respond_to?(:generator_mix)
|
|
318
|
+
end
|
|
319
|
+
kpi = practice_kpi(results: []) if commit && respond_to?(:practice_kpi)
|
|
320
|
+
out = {
|
|
321
|
+
scored: scored.length, mean: mean, since_hours: since_h,
|
|
322
|
+
results: scored.first(10), sentinel_warm: warm,
|
|
323
|
+
scrub: scrub, generator_mix: mix, practice_kpi: kpi
|
|
324
|
+
}
|
|
325
|
+
log(event: :offline_judge, data: out.except(:results))
|
|
250
326
|
out
|
|
251
327
|
rescue StandardError => e
|
|
252
328
|
{ error: "#{e.class}: #{e.message}" }
|
|
@@ -257,6 +333,14 @@ module PWN
|
|
|
257
333
|
public_class_method def self.preference_balance(opts = {})
|
|
258
334
|
return { total: 0 } unless defined?(Reward)
|
|
259
335
|
|
|
336
|
+
# P15 — prefer Reward.preference_balance (geometry-aware + optional scrub).
|
|
337
|
+
if Reward.respond_to?(:preference_balance)
|
|
338
|
+
return Reward.preference_balance(
|
|
339
|
+
limit: opts[:limit] || 10_000,
|
|
340
|
+
scrub: opts.key?(:scrub) ? opts[:scrub] : false
|
|
341
|
+
)
|
|
342
|
+
end
|
|
343
|
+
|
|
260
344
|
rows = Reward.preferences(limit: opts[:limit] || 10_000)
|
|
261
345
|
by = Hash.new(0)
|
|
262
346
|
rows.each { |r| by[r[:source].to_s] += 1 }
|
|
@@ -279,7 +363,16 @@ module PWN
|
|
|
279
363
|
end
|
|
280
364
|
|
|
281
365
|
public_class_method def self.counterfactual(opts = {})
|
|
282
|
-
|
|
366
|
+
# P0 — when generator_mix marks counterfactual underfilled, run even
|
|
367
|
+
# if the auto-flag would keep it off (remote-default still respected
|
|
368
|
+
# only when mix is healthy). Still refuse recursion.
|
|
369
|
+
mix_need = begin
|
|
370
|
+
m = defined?(Reward) && Reward.respond_to?(:generator_mix) ? Reward.generator_mix : {}
|
|
371
|
+
Array(m[:urgent]).include?('counterfactual')
|
|
372
|
+
rescue StandardError
|
|
373
|
+
false
|
|
374
|
+
end
|
|
375
|
+
return nil unless enabled?(key: :counterfactual) || mix_need || opts[:force]
|
|
283
376
|
return nil if in_curriculum?
|
|
284
377
|
|
|
285
378
|
request = opts[:request].to_s
|
|
@@ -292,12 +385,39 @@ module PWN
|
|
|
292
385
|
end
|
|
293
386
|
return nil if branch_b.to_s.strip.empty?
|
|
294
387
|
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
388
|
+
sa_h = score_branch_detailed(request: request, branch: branch_a)
|
|
389
|
+
sb_h = score_branch_detailed(request: request, branch: branch_b)
|
|
390
|
+
sa = sa_h[:score]
|
|
391
|
+
sb = sb_h[:score]
|
|
392
|
+
if sb > sa
|
|
393
|
+
winner = branch_b
|
|
394
|
+
loser = branch_a
|
|
395
|
+
tag = :b
|
|
396
|
+
wmeta = sb_h
|
|
397
|
+
else
|
|
398
|
+
winner = branch_a
|
|
399
|
+
loser = branch_b
|
|
400
|
+
tag = :a
|
|
401
|
+
wmeta = sa_h
|
|
402
|
+
end
|
|
403
|
+
real_hit = sa_h[:mode] == :real_dispatch || sb_h[:mode] == :real_dispatch
|
|
404
|
+
shape = real_hit ? :real_dispatch : :imagined
|
|
405
|
+
if defined?(Reward)
|
|
406
|
+
Reward.record_preference(
|
|
407
|
+
prompt: "#{request} | failing: #{opts[:name]} → #{opts[:error]}",
|
|
408
|
+
rejected: loser.to_s[0, 2_000],
|
|
409
|
+
chosen: winner.to_s[0, 2_000],
|
|
410
|
+
source: :counterfactual,
|
|
411
|
+
shape: shape,
|
|
412
|
+
meta: {
|
|
413
|
+
a_score: sa, b_score: sb,
|
|
414
|
+
a_mode: sa_h[:mode], b_mode: sb_h[:mode],
|
|
415
|
+
winner_trace: wmeta[:trace].to_s[0, 500]
|
|
416
|
+
}
|
|
417
|
+
)
|
|
418
|
+
end
|
|
419
|
+
log(event: :counterfactual, data: { branch: tag, a: sa, b: sb, tool: opts[:name].to_s, shape: shape })
|
|
420
|
+
{ branch: tag, content: winner, score: [sa, sb].max, a: sa, b: sb, shape: shape, a_mode: sa_h[:mode], b_mode: sb_h[:mode] }
|
|
301
421
|
rescue StandardError => e
|
|
302
422
|
warn "[pwn-ai/curriculum] counterfactual swallowed: #{e.class}: #{e.message}"
|
|
303
423
|
nil
|
|
@@ -319,9 +439,20 @@ module PWN
|
|
|
319
439
|
# self-correction becomes DPO signal.
|
|
320
440
|
|
|
321
441
|
public_class_method def self.critic(opts = {})
|
|
322
|
-
|
|
442
|
+
mix_need = begin
|
|
443
|
+
m = defined?(Reward) && Reward.respond_to?(:generator_mix) ? Reward.generator_mix : {}
|
|
444
|
+
Array(m[:urgent]).include?('critic')
|
|
445
|
+
rescue StandardError
|
|
446
|
+
false
|
|
447
|
+
end
|
|
448
|
+
return { verdict: :pass, source: :disabled } unless enabled?(key: :critic) || opts[:text_only] || mix_need || opts[:force]
|
|
323
449
|
return { verdict: :pass, source: :recursion } if in_curriculum?
|
|
324
450
|
|
|
451
|
+
# P24 — text_only: single Reflect shot, no tool-armed persona swarm.
|
|
452
|
+
# Used when budget_exhaustion_hot? so critic cannot thrash the budget
|
|
453
|
+
# that auto_introspect is trying to protect.
|
|
454
|
+
return critic_text_only(request: opts[:request], final: opts[:final], session_id: opts[:session_id]) if opts[:text_only]
|
|
455
|
+
|
|
325
456
|
ensure_persona(name: CRITIC_NAME, role: "You are pwn-ai's constitutional critic. Given a REQUEST and a candidate ANSWER, find ONE concrete, verifiable flaw (wrong fact, missing step, unsupported claim, broken command). You MAY call shell / extro_verify / pwn_eval to check. If none found reply exactly: PASS. Otherwise reply: FLAW: <one line>.")
|
|
326
457
|
reply = with_curriculum_guard do
|
|
327
458
|
ask_persona(name: CRITIC_NAME, request: "REQUEST:\n#{opts[:request].to_s[0, 800]}\n\nANSWER:\n#{opts[:final].to_s[0, 2_000]}")
|
|
@@ -332,13 +463,22 @@ module PWN
|
|
|
332
463
|
else
|
|
333
464
|
flaw = reply.to_s.sub(/\AFLAW:\s*/i, '').strip[0, 300]
|
|
334
465
|
Mistakes.record(tool: 'assistant_answer', error: "critic: #{flaw}", args: opts[:final].to_s[0, 200], session_id: opts[:session_id], source: :model) if defined?(Mistakes)
|
|
335
|
-
#
|
|
466
|
+
# P9 — DPO pair geometry: rejected=bad final, chosen=REVISED full
|
|
467
|
+
# answer (not "CORRECTION: flaw" prose). Prefer persona rewrite.
|
|
336
468
|
if defined?(Reward) && !flaw.to_s.empty?
|
|
469
|
+
revised = revise_after_flaw(
|
|
470
|
+
request: opts[:request],
|
|
471
|
+
final: opts[:final],
|
|
472
|
+
flaw: flaw,
|
|
473
|
+
session_id: opts[:session_id]
|
|
474
|
+
)
|
|
337
475
|
Reward.record_preference(
|
|
338
476
|
prompt: opts[:request].to_s[0, 1_000],
|
|
339
477
|
rejected: opts[:final].to_s[0, 2_000],
|
|
340
|
-
chosen:
|
|
341
|
-
source: :critic
|
|
478
|
+
chosen: revised,
|
|
479
|
+
source: :critic,
|
|
480
|
+
shape: :revised_answer,
|
|
481
|
+
meta: { flaw: flaw.to_s[0, 200] }
|
|
342
482
|
)
|
|
343
483
|
end
|
|
344
484
|
log(event: :critic, data: { verdict: :flaw, flaw: flaw.to_s[0, 200] })
|
|
@@ -362,6 +502,17 @@ module PWN
|
|
|
362
502
|
return nil unless enabled?(key: :red_team_plan)
|
|
363
503
|
return nil if in_curriculum?
|
|
364
504
|
|
|
505
|
+
# P17 — never nest a red-team persona loop when budget_exhaustion
|
|
506
|
+
# fingerprints dominate open mistakes (amplifier of agent_loop ×N).
|
|
507
|
+
begin
|
|
508
|
+
if defined?(Loop) && Loop.respond_to?(:budget_exhaustion_hot?, true) &&
|
|
509
|
+
Loop.send(:budget_exhaustion_hot?)
|
|
510
|
+
return nil
|
|
511
|
+
end
|
|
512
|
+
rescue StandardError
|
|
513
|
+
# fall through
|
|
514
|
+
end
|
|
515
|
+
|
|
365
516
|
ensure_persona(name: RED_TEAM_NAME, role: 'You are pwn-ai\'s adversarial plan reviewer. Given a numbered tool plan and telemetry from THIS host (tool success rates, known mistakes, environment drift), identify the ONE step most likely to fail and say why in ≤2 lines. Cite the metric/mistake/drift. If the plan is sound reply: SOUND.')
|
|
366
517
|
telemetry = build_telemetry
|
|
367
518
|
reply = with_curriculum_guard do
|
|
@@ -462,8 +613,15 @@ module PWN
|
|
|
462
613
|
|
|
463
614
|
candidate = ollama_create(base: base, adapter: adapter, version: version)
|
|
464
615
|
baseline = state[:tag] || base
|
|
465
|
-
|
|
466
|
-
|
|
616
|
+
# P11 — gate v2: resolved delta + mean judge + frozen smoke set.
|
|
617
|
+
gate = ab_gate_v2(baseline: baseline, candidate: candidate, evalset: evalset)
|
|
618
|
+
# P19 — refuse promote when W1 diet is still prose/monoculture.
|
|
619
|
+
# Export-only is correct until scrubbed pairs show trajectory diversity.
|
|
620
|
+
diet = preference_diet_gate
|
|
621
|
+
gate = gate.merge(preference_diet: diet)
|
|
622
|
+
promoted = gate[:promote] == true && diet[:ok] == true
|
|
623
|
+
gate[:promote] = promoted
|
|
624
|
+
gate[:promote_blocked_by_diet] = true unless diet[:ok]
|
|
467
625
|
if promoted
|
|
468
626
|
state[:previous] = state[:tag]
|
|
469
627
|
state[:tag] = candidate
|
|
@@ -472,7 +630,7 @@ module PWN
|
|
|
472
630
|
state[:gate] = gate
|
|
473
631
|
save_models(state: state)
|
|
474
632
|
end
|
|
475
|
-
result.merge(adapter: adapter, candidate: candidate, gate: gate, promoted: promoted)
|
|
633
|
+
result.merge(adapter: adapter, candidate: candidate, gate: gate, promoted: promoted, weight_loop: :closed)
|
|
476
634
|
rescue StandardError => e
|
|
477
635
|
{ error: "#{e.class}: #{e.message}" }
|
|
478
636
|
end
|
|
@@ -484,6 +642,86 @@ module PWN
|
|
|
484
642
|
# Supported Method Parameters::
|
|
485
643
|
# PWN::AI::Agent::Curriculum.calibrate(predicted:, actual:, engine:)
|
|
486
644
|
|
|
645
|
+
# ----------------------------------------------------------------
|
|
646
|
+
# P1 — Outer curriculum KPI: does practice cut live [REPEATING]?
|
|
647
|
+
# ----------------------------------------------------------------
|
|
648
|
+
# Snapshot unresolved repeating counts before/after practice nights
|
|
649
|
+
# into ~/.pwn/curriculum_kpi.jsonl so week-over-week delta is visible
|
|
650
|
+
# without scraping Mistakes by hand. practice() always appends a row.
|
|
651
|
+
|
|
652
|
+
KPI_FILE = File.join(Dir.home, '.pwn', 'curriculum_kpi.jsonl')
|
|
653
|
+
|
|
654
|
+
public_class_method def self.practice_kpi(opts = {})
|
|
655
|
+
results = Array(opts[:results])
|
|
656
|
+
top = defined?(Mistakes) ? Mistakes.top(limit: 50, unresolved_only: true) : []
|
|
657
|
+
repeating = top.select { |m| m[:count].to_i >= 3 }
|
|
658
|
+
budgetish = repeating.count do |m|
|
|
659
|
+
t = m[:tool].to_s
|
|
660
|
+
e = m[:error].to_s.downcase
|
|
661
|
+
t == 'agent_loop' || t == 'assistant_answer' ||
|
|
662
|
+
e.include?('budget') || e.include?('iteration budget')
|
|
663
|
+
end
|
|
664
|
+
row = {
|
|
665
|
+
at: Time.now.utc.iso8601,
|
|
666
|
+
unresolved_total: top.length,
|
|
667
|
+
repeating_n: repeating.length,
|
|
668
|
+
repeating_sum_count: repeating.sum { |m| m[:count].to_i },
|
|
669
|
+
budget_repeating_n: budgetish,
|
|
670
|
+
practiced: results.length,
|
|
671
|
+
resolved_tonight: results.count { |r| r[:resolved] },
|
|
672
|
+
mean_holdout: if results.empty?
|
|
673
|
+
nil
|
|
674
|
+
else
|
|
675
|
+
(results.sum { |r| r[:mean_score].to_f } / results.length).round(3)
|
|
676
|
+
end
|
|
677
|
+
}
|
|
678
|
+
begin
|
|
679
|
+
FileUtils.mkdir_p(File.dirname(KPI_FILE))
|
|
680
|
+
File.open(KPI_FILE, 'a') { |f| f.puts(JSON.generate(row)) }
|
|
681
|
+
rescue StandardError
|
|
682
|
+
nil
|
|
683
|
+
end
|
|
684
|
+
trend = repeating_trend
|
|
685
|
+
row.merge(trend: trend)
|
|
686
|
+
rescue StandardError => e
|
|
687
|
+
{ error: "#{e.class}: #{e.message}" }
|
|
688
|
+
end
|
|
689
|
+
|
|
690
|
+
# Week-over-week (or last-N snapshots) delta on repeating_n.
|
|
691
|
+
# Positive delta_repeating = getting worse; negative = practice working.
|
|
692
|
+
public_class_method def self.repeating_trend(opts = {})
|
|
693
|
+
limit = (opts[:limit] || 14).to_i
|
|
694
|
+
return { samples: 0, delta_repeating: nil, status: :no_data } unless File.exist?(KPI_FILE)
|
|
695
|
+
|
|
696
|
+
rows = File.readlines(KPI_FILE).last(limit).filter_map do |l|
|
|
697
|
+
JSON.parse(l, symbolize_names: true)
|
|
698
|
+
rescue StandardError
|
|
699
|
+
nil
|
|
700
|
+
end
|
|
701
|
+
return { samples: 0, delta_repeating: nil, status: :no_data } if rows.empty?
|
|
702
|
+
return { samples: rows.length, delta_repeating: 0, status: :baseline, latest: rows.last } if rows.length < 2
|
|
703
|
+
|
|
704
|
+
first = rows.first
|
|
705
|
+
last = rows.last
|
|
706
|
+
d_rep = last[:repeating_n].to_i - first[:repeating_n].to_i
|
|
707
|
+
d_budget = last[:budget_repeating_n].to_i - first[:budget_repeating_n].to_i
|
|
708
|
+
status = if d_rep <= -2 then :improving
|
|
709
|
+
elsif d_rep >= 2 then :regressing
|
|
710
|
+
else :flat
|
|
711
|
+
end
|
|
712
|
+
{
|
|
713
|
+
samples: rows.length,
|
|
714
|
+
from: first[:at],
|
|
715
|
+
to: last[:at],
|
|
716
|
+
delta_repeating: d_rep,
|
|
717
|
+
delta_budget_repeating: d_budget,
|
|
718
|
+
latest_repeating_n: last[:repeating_n],
|
|
719
|
+
status: status
|
|
720
|
+
}
|
|
721
|
+
rescue StandardError => e
|
|
722
|
+
{ samples: 0, status: :error, error: "#{e.class}: #{e.message}" }
|
|
723
|
+
end
|
|
724
|
+
|
|
487
725
|
public_class_method def self.calibrate(opts = {})
|
|
488
726
|
p = opts[:predicted].to_f.clamp(0.0, 1.0)
|
|
489
727
|
a = opts[:actual].to_f.clamp(0.0, 1.0)
|
|
@@ -496,6 +734,70 @@ module PWN
|
|
|
496
734
|
# privates
|
|
497
735
|
# ----------------------------------------------------------------
|
|
498
736
|
|
|
737
|
+
# P24 — single-shot text critic (no tools, no persona Loop.run).
|
|
738
|
+
private_class_method def self.critic_text_only(opts = {})
|
|
739
|
+
req = opts[:request].to_s
|
|
740
|
+
final = opts[:final].to_s
|
|
741
|
+
return { verdict: :pass, source: :text_only_empty } if final.strip.empty?
|
|
742
|
+
|
|
743
|
+
reply = if reflect_available?
|
|
744
|
+
Reflect.on(
|
|
745
|
+
request: "You are a strict critic. Given REQUEST and ANSWER, reply PASS or FLAW: <one line>.\nREQUEST:\n#{req[0, 600]}\nANSWER:\n#{final[0, 1_200]}",
|
|
746
|
+
suppress_pii_warning: true
|
|
747
|
+
).to_s
|
|
748
|
+
else
|
|
749
|
+
# heuristic fallback: empty / self-reported failure / budget stop
|
|
750
|
+
if final.match?(/\[pwn-ai\].*budget|iteration budget exhausted|i (was )?unable to|failed to\b/i)
|
|
751
|
+
'FLAW: answer reports failure or budget exhaustion'
|
|
752
|
+
else
|
|
753
|
+
'PASS'
|
|
754
|
+
end
|
|
755
|
+
end
|
|
756
|
+
if reply.to_s.strip.upcase.start_with?('PASS')
|
|
757
|
+
{ verdict: :pass, source: :text_only, confidence: 0.55 }
|
|
758
|
+
else
|
|
759
|
+
flaw = reply.to_s.sub(/\AFLAW:\s*/i, '').strip[0, 300]
|
|
760
|
+
# Do NOT Mistakes.record assistant_answer thrash from text_only path —
|
|
761
|
+
# that was feeding the budget-exhaustion mistake pile.
|
|
762
|
+
{ verdict: :flaw, flaw: flaw, source: :text_only, confidence: 0.55 }
|
|
763
|
+
end
|
|
764
|
+
rescue StandardError => e
|
|
765
|
+
{ verdict: :pass, source: :text_only_error, error: e.message }
|
|
766
|
+
end
|
|
767
|
+
|
|
768
|
+
# P9 — turn a critic flaw into a DPO-chosen REVISED answer (trajectory
|
|
769
|
+
# shaped), not "CORRECTION: <flaw>" commentary. Best-effort: ask Reflect
|
|
770
|
+
# to rewrite; fall back to a structured scaffold that still contains the
|
|
771
|
+
# original answer + the concrete fix.
|
|
772
|
+
private_class_method def self.revise_after_flaw(opts = {})
|
|
773
|
+
req = opts[:request].to_s
|
|
774
|
+
final = opts[:final].to_s
|
|
775
|
+
flaw = opts[:flaw].to_s
|
|
776
|
+
revised = nil
|
|
777
|
+
if reflect_available?
|
|
778
|
+
prompt = <<~P
|
|
779
|
+
Revise the ANSWER so it no longer has this flaw. Return the FULL
|
|
780
|
+
corrected answer only (no preamble, no "CORRECTION:" prefix).
|
|
781
|
+
REQUEST: #{req[0, 600]}
|
|
782
|
+
FLAW: #{flaw[0, 300]}
|
|
783
|
+
ANSWER: #{final[0, 1_500]}
|
|
784
|
+
P
|
|
785
|
+
revised = Reflect.on(request: prompt, suppress_pii_warning: true).to_s.strip
|
|
786
|
+
end
|
|
787
|
+
if revised.to_s.strip.empty? || revised.strip == final.strip || revised.match?(/\A\s*CORRECTION:/i)
|
|
788
|
+
revised = <<~REV.strip
|
|
789
|
+
REVISED ANSWER (addresses: #{flaw[0, 180]}):
|
|
790
|
+
#{final[0, 1_200]}
|
|
791
|
+
|
|
792
|
+
Fix applied: #{flaw[0, 300]}
|
|
793
|
+
Do not repeat the flawed claim or step above; prefer verified evidence from tools.
|
|
794
|
+
REV
|
|
795
|
+
end
|
|
796
|
+
revised.to_s[0, 4_000]
|
|
797
|
+
rescue StandardError
|
|
798
|
+
"REVISED ANSWER (addresses: #{opts[:flaw].to_s[0, 180]}):\n#{opts[:final].to_s[0, 1_200]}"
|
|
799
|
+
end
|
|
800
|
+
|
|
499
801
|
private_class_method def self.generate_reproducers(opts = {})
|
|
500
802
|
m = opts[:mistake]
|
|
501
803
|
count = (opts[:count] || 2).to_i
|
|
@@ -581,13 +883,34 @@ module PWN
|
|
|
581
883
|
'Use pwn_eval to list PWN::AI::Agent constants',
|
|
582
884
|
'Return Dir.pwd from pwn_eval'
|
|
583
885
|
]
|
|
584
|
-
|
|
886
|
+
when 'agent_loop', 'assistant_answer'
|
|
887
|
+
# P17 — dominant live failure: iteration / critic budget exhaustion.
|
|
888
|
+
# Practise finishing under a tight tool budget, not shell shapes.
|
|
585
889
|
[
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
890
|
+
'Answer in one shell call: print kernel release with uname -r',
|
|
891
|
+
'In at most two tools, show cwd and ruby version then stop',
|
|
892
|
+
'Give a final answer with no tools: what is 7 times 8?',
|
|
893
|
+
'Finish under three iterations: list /tmp and report file count',
|
|
894
|
+
'Do not explore — one pwn_eval of Dir.pwd and return the path',
|
|
895
|
+
'Short plan then one command: show free disk with df -h /'
|
|
590
896
|
]
|
|
897
|
+
else
|
|
898
|
+
if err.include?('budget exhausted') || err.include?('iteration budget') ||
|
|
899
|
+
(err.include?('handler_error') && err.include?('budget'))
|
|
900
|
+
[
|
|
901
|
+
'Answer in one shell call: print kernel release with uname -r',
|
|
902
|
+
'In at most two tools, show cwd and ruby version then stop',
|
|
903
|
+
'Give a final answer with no tools: what is 7 times 8?',
|
|
904
|
+
'Finish under three iterations: list /tmp and report file count'
|
|
905
|
+
]
|
|
906
|
+
else
|
|
907
|
+
[
|
|
908
|
+
"Demonstrate a correct use of the #{tool} tool on this host",
|
|
909
|
+
"Use #{tool} to answer a simple factual question about this system",
|
|
910
|
+
"Show a minimal successful #{tool} call with valid arguments",
|
|
911
|
+
"Recover from a bad #{tool} invocation without retrying the same args"
|
|
912
|
+
]
|
|
913
|
+
end
|
|
591
914
|
end
|
|
592
915
|
Array.new(n) { |i| pool[i % pool.length] }
|
|
593
916
|
end
|
|
@@ -654,8 +977,29 @@ module PWN
|
|
|
654
977
|
|
|
655
978
|
private_class_method def self.self_play(opts = {})
|
|
656
979
|
sid = PWN::Sessions.create(title: "curriculum #{opts[:tag]}")[:id]
|
|
657
|
-
|
|
658
|
-
|
|
980
|
+
# P23 — short-horizon graded tasks: cap iters so practice actually
|
|
981
|
+
# teaches finish-under-N instead of letting DEFAULT_MAX_ITERS mask it.
|
|
982
|
+
prompt = opts[:prompt].to_s
|
|
983
|
+
short = prompt.match?(/\b(one shell|at most two|no tools|three iterations|finish under|do not explore|short plan)\b/i)
|
|
984
|
+
prev_max = :__unset__
|
|
985
|
+
capped = false
|
|
986
|
+
if short && defined?(PWN::Env) && PWN::Env.is_a?(Hash) && PWN::Env[:ai].is_a?(Hash) && PWN::Env[:ai][:agent].is_a?(Hash) && !PWN::Env[:ai][:agent].frozen?
|
|
987
|
+
prev_max = PWN::Env[:ai][:agent].key?(:max_iters) ? PWN::Env[:ai][:agent][:max_iters] : :__unset__
|
|
988
|
+
PWN::Env[:ai][:agent][:max_iters] = 5
|
|
989
|
+
capped = true
|
|
990
|
+
end
|
|
991
|
+
begin
|
|
992
|
+
final = Loop.run(request: prompt, session_id: sid, enabled_toolsets: %w[terminal pwn memory learning])
|
|
993
|
+
ensure
|
|
994
|
+
if capped && defined?(PWN::Env) && PWN::Env[:ai].is_a?(Hash) && PWN::Env[:ai][:agent].is_a?(Hash) && !PWN::Env[:ai][:agent].frozen?
|
|
995
|
+
if prev_max == :__unset__
|
|
996
|
+
PWN::Env[:ai][:agent].delete(:max_iters)
|
|
997
|
+
else
|
|
998
|
+
PWN::Env[:ai][:agent][:max_iters] = prev_max
|
|
999
|
+
end
|
|
1000
|
+
end
|
|
1001
|
+
end
|
|
1002
|
+
v = Reward.judge(request: prompt, final: final, session_id: sid, commit: false)
|
|
659
1003
|
# 2.4 — capture tool trace for structured_fix.winning_trace
|
|
660
1004
|
trace = begin
|
|
661
1005
|
if defined?(PWN::Sessions)
|
|
@@ -673,21 +1017,25 @@ module PWN
|
|
|
673
1017
|
end
|
|
674
1018
|
|
|
675
1019
|
private_class_method def self.score_branch(opts = {})
|
|
676
|
-
|
|
677
|
-
|
|
1020
|
+
score_branch_detailed(**opts)[:score]
|
|
1021
|
+
end
|
|
1022
|
+
|
|
1023
|
+
# Returns {score:, mode: :real_dispatch|:imagined|:default, trace:}.
|
|
1024
|
+
# Callers that need honest advantage estimation should inspect :mode.
|
|
1025
|
+
private_class_method def self.score_branch_detailed(opts = {})
|
|
678
1026
|
branch = opts[:branch].to_s
|
|
679
1027
|
request = opts[:request].to_s
|
|
680
1028
|
real = try_real_dispatch_score(branch: branch)
|
|
681
|
-
return real if real
|
|
1029
|
+
return { score: real, mode: :real_dispatch, trace: branch.to_s[0, 500] } if real
|
|
682
1030
|
|
|
683
|
-
return 0.5 unless reflect_available?
|
|
1031
|
+
return { score: 0.5, mode: :default, trace: nil } unless reflect_available?
|
|
684
1032
|
|
|
685
1033
|
req = "Goal: #{request}\nProposed next action: #{branch}\nOn a scale 0.0-1.0, how likely is this to advance the goal on a Kali Linux host? Reply with ONLY the number."
|
|
686
1034
|
imagined = Reflect.on(request: req, suppress_pii_warning: true).to_s[/[01](?:\.\d+)?/].to_f.clamp(0.0, 1.0)
|
|
687
1035
|
# haircut imagined scores so they never outrank a real dispatch
|
|
688
|
-
(imagined * 0.6).clamp(0.0, 0.6)
|
|
1036
|
+
{ score: (imagined * 0.6).clamp(0.0, 0.6), mode: :imagined, trace: nil }
|
|
689
1037
|
rescue StandardError
|
|
690
|
-
0.5
|
|
1038
|
+
{ score: 0.5, mode: :default, trace: nil }
|
|
691
1039
|
end
|
|
692
1040
|
|
|
693
1041
|
# Best-effort: if branch names a registered tool + args, run ONE Dispatch
|
|
@@ -767,6 +1115,114 @@ module PWN
|
|
|
767
1115
|
{ baseline: opts[:baseline], candidate: opts[:candidate], baseline_resolved: baseline, candidate_resolved: candid, evalset_size: evalset.length }
|
|
768
1116
|
end
|
|
769
1117
|
|
|
1118
|
+
# P11 — promote only when candidate beats baseline on (a) resolved count
|
|
1119
|
+
# with margin, (b) mean judge score, and (c) no smoke regression. Smoke
|
|
1120
|
+
# set is fixed natural tasks, not Mistakes.top, so eval-set memorisation
|
|
1121
|
+
# cannot self-promote.
|
|
1122
|
+
private_class_method def self.ab_gate_v2(opts = {})
|
|
1123
|
+
evalset = Array(opts[:evalset])
|
|
1124
|
+
baseline = replay_on_detailed(tag: opts[:baseline], evalset: evalset)
|
|
1125
|
+
candid = replay_on_detailed(tag: opts[:candidate], evalset: evalset)
|
|
1126
|
+
smoke = smoke_eval_set
|
|
1127
|
+
base_smoke = replay_on_detailed(tag: opts[:baseline], evalset: smoke)
|
|
1128
|
+
cand_smoke = replay_on_detailed(tag: opts[:candidate], evalset: smoke)
|
|
1129
|
+
|
|
1130
|
+
b_res = baseline[:resolved].to_i
|
|
1131
|
+
c_res = candid[:resolved].to_i
|
|
1132
|
+
n = [evalset.length, 1].max
|
|
1133
|
+
delta = c_res - b_res
|
|
1134
|
+
rel = delta.to_f / n
|
|
1135
|
+
resolved_win = delta >= 1 && (n < 10 || rel >= 0.05)
|
|
1136
|
+
|
|
1137
|
+
b_mean = baseline[:mean_score].to_f
|
|
1138
|
+
c_mean = candid[:mean_score].to_f
|
|
1139
|
+
mean_win = c_mean + 1e-9 >= b_mean
|
|
1140
|
+
|
|
1141
|
+
smoke_ok = cand_smoke[:resolved].to_i >= base_smoke[:resolved].to_i &&
|
|
1142
|
+
cand_smoke[:mean_score].to_f + 0.05 >= base_smoke[:mean_score].to_f
|
|
1143
|
+
|
|
1144
|
+
promote = resolved_win && mean_win && smoke_ok
|
|
1145
|
+
{
|
|
1146
|
+
baseline: opts[:baseline],
|
|
1147
|
+
candidate: opts[:candidate],
|
|
1148
|
+
baseline_resolved: b_res,
|
|
1149
|
+
candidate_resolved: c_res,
|
|
1150
|
+
baseline_mean: b_mean.round(3),
|
|
1151
|
+
candidate_mean: c_mean.round(3),
|
|
1152
|
+
delta_resolved: delta,
|
|
1153
|
+
relative_delta: rel.round(3),
|
|
1154
|
+
smoke: {
|
|
1155
|
+
baseline_resolved: base_smoke[:resolved],
|
|
1156
|
+
candidate_resolved: cand_smoke[:resolved],
|
|
1157
|
+
baseline_mean: base_smoke[:mean_score],
|
|
1158
|
+
candidate_mean: cand_smoke[:mean_score]
|
|
1159
|
+
},
|
|
1160
|
+
resolved_win: resolved_win,
|
|
1161
|
+
mean_win: mean_win,
|
|
1162
|
+
smoke_ok: smoke_ok,
|
|
1163
|
+
promote: promote,
|
|
1164
|
+
evalset_size: evalset.length,
|
|
1165
|
+
gate_version: 2
|
|
1166
|
+
}
|
|
1167
|
+
end
|
|
1168
|
+
|
|
1169
|
+
private_class_method def self.smoke_eval_set
|
|
1170
|
+
[
|
|
1171
|
+
{ signature: 'smoke_uname', prompt: 'Print the kernel release with uname -r' },
|
|
1172
|
+
{ signature: 'smoke_pwd', prompt: 'Show the current working directory' },
|
|
1173
|
+
{ signature: 'smoke_ruby', prompt: 'Display the active ruby version' }
|
|
1174
|
+
]
|
|
1175
|
+
end
|
|
1176
|
+
|
|
1177
|
+
# P19 — weight promote requires scrubbed trajectory diversity, not just
|
|
1178
|
+
# A/B eval win. export_ready remains the correct posture otherwise.
|
|
1179
|
+
private_class_method def self.preference_diet_gate(opts = {})
|
|
1180
|
+
return { ok: false, reason: 'no Reward' } unless defined?(Reward)
|
|
1181
|
+
|
|
1182
|
+
bal = if Reward.respond_to?(:preference_balance)
|
|
1183
|
+
Reward.preference_balance(limit: opts[:limit] || 10_000, scrub: true)
|
|
1184
|
+
else
|
|
1185
|
+
preference_balance(limit: opts[:limit] || 10_000)
|
|
1186
|
+
end
|
|
1187
|
+
total = bal[:kept].to_i
|
|
1188
|
+
total = bal[:total].to_i if total <= 0
|
|
1189
|
+
return { ok: false, reason: 'too_few_pairs', total: total, min: 12 } if total < 12
|
|
1190
|
+
|
|
1191
|
+
frac = bal[:fractions] || {}
|
|
1192
|
+
max_share = frac.values.map(&:to_f).max || 1.0
|
|
1193
|
+
traj_frac = bal[:trajectory_fraction].to_f
|
|
1194
|
+
ok = !bal[:monoculture] && max_share <= 0.45 && traj_frac >= 0.30
|
|
1195
|
+
{
|
|
1196
|
+
ok: ok,
|
|
1197
|
+
total: total,
|
|
1198
|
+
monoculture: bal[:monoculture],
|
|
1199
|
+
max_source_share: max_share.round(3),
|
|
1200
|
+
trajectory_fraction: traj_frac.round(3),
|
|
1201
|
+
by_source: bal[:by_source],
|
|
1202
|
+
reason: ok ? 'diet_ok' : 'need_trajectory_diversity_or_rebalance'
|
|
1203
|
+
}
|
|
1204
|
+
rescue StandardError => e
|
|
1205
|
+
{ ok: false, reason: "#{e.class}: #{e.message}" }
|
|
1206
|
+
end
|
|
1207
|
+
|
|
1208
|
+
private_class_method def self.replay_on_detailed(opts = {})
|
|
1209
|
+
tag = opts[:tag].to_s
|
|
1210
|
+
return { resolved: 0, mean_score: 0.0, scores: [] } if tag.empty?
|
|
1211
|
+
|
|
1212
|
+
scores = []
|
|
1213
|
+
with_ollama_model(tag: tag) do
|
|
1214
|
+
Array(opts[:evalset]).each do |e|
|
|
1215
|
+
r = self_play(prompt: e[:prompt], tag: "gate:#{tag}")
|
|
1216
|
+
scores << r[:score].to_f
|
|
1217
|
+
end
|
|
1218
|
+
end
|
|
1219
|
+
resolved = scores.count { |s| s >= 0.7 }
|
|
1220
|
+
mean = scores.empty? ? 0.0 : (scores.sum / scores.length)
|
|
1221
|
+
{ resolved: resolved, mean_score: mean.round(3), scores: scores }
|
|
1222
|
+
rescue StandardError
|
|
1223
|
+
{ resolved: 0, mean_score: 0.0, scores: [] }
|
|
1224
|
+
end
|
|
1225
|
+
|
|
770
1226
|
private_class_method def self.replay_on(opts = {})
|
|
771
1227
|
tag = opts[:tag].to_s
|
|
772
1228
|
return 0 if tag.empty?
|
|
@@ -1118,7 +1574,7 @@ module PWN
|
|
|
1118
1574
|
puts <<~USAGE
|
|
1119
1575
|
USAGE:
|
|
1120
1576
|
# Tier 4 — self-play
|
|
1121
|
-
PWN::AI::Agent::Curriculum.practice(limit: 3) # S1
|
|
1577
|
+
PWN::AI::Agent::Curriculum.practice(limit: 3) # S1 + P14 trajectory DPO pairs + P17 budget-first
|
|
1122
1578
|
PWN::AI::Agent::Curriculum.offline_judge(since_hours: 24) # P3 offline ORM/PRM fill
|
|
1123
1579
|
PWN::AI::Agent::Curriculum.preference_balance # P5 W1 diversity report
|
|
1124
1580
|
PWN::AI::Agent::Curriculum.counterfactual(request:, name:, args:, error:, hint:) # S2 A/B → DPO pair
|
|
@@ -1127,7 +1583,7 @@ module PWN
|
|
|
1127
1583
|
PWN::AI::Agent::Curriculum.hindsight(request:, final:, session_id:) # C3 HER soft-relabel
|
|
1128
1584
|
|
|
1129
1585
|
# Tier 5 — close the weight loop
|
|
1130
|
-
PWN::AI::Agent::Curriculum.train_and_gate(dry_run: true) # W2 export-ready; promote only with trainer+dry_run:false
|
|
1586
|
+
PWN::AI::Agent::Curriculum.train_and_gate(dry_run: true) # W2 export-ready; P11 gate v2 promote only with trainer+dry_run:false
|
|
1131
1587
|
PWN::AI::Agent::Curriculum.calibrate(predicted: 0.8, actual: 1.0) # W3 Brier → Metrics[:calibration]
|
|
1132
1588
|
|
|
1133
1589
|
Cron self-improvement (seeded by PWN::Cron.install_defaults):
|