pwn 0.5.643 → 0.5.650
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/documentation/Agent-Tool-Registry.md +10 -5
- data/documentation/Cron.md +3 -2
- data/documentation/Home.md +2 -2
- data/documentation/How-PWN-Works.md +1 -1
- data/documentation/Mistakes.md +13 -0
- data/documentation/Reinforcement-Learning.md +72 -3
- data/documentation/Skills-Memory-Learning.md +5 -3
- data/documentation/What-is-PWN.md +1 -1
- data/documentation/diagrams/dot/pwn-ai-feedback-learning-loop.dot +3 -3
- data/documentation/diagrams/dot/reinforcement-learning.dot +8 -8
- data/documentation/diagrams/pwn-ai-feedback-learning-loop.svg +318 -319
- data/documentation/diagrams/reinforcement-learning.svg +188 -187
- data/documentation/pwn-ai-Agent.md +6 -5
- data/lib/pwn/ai/agent/curriculum.rb +491 -35
- data/lib/pwn/ai/agent/extrospection.rb +18 -4
- data/lib/pwn/ai/agent/learning.rb +308 -54
- data/lib/pwn/ai/agent/loop.rb +145 -11
- data/lib/pwn/ai/agent/metrics.rb +200 -11
- data/lib/pwn/ai/agent/mistakes.rb +26 -8
- data/lib/pwn/ai/agent/registry.rb +24 -2
- data/lib/pwn/ai/agent/reward.rb +447 -7
- data/lib/pwn/ai/agent/tools/curriculum.rb +23 -0
- data/lib/pwn/ai/agent/tools/reward.rb +86 -0
- data/lib/pwn/ai/agent/tools/ruby_eval.rb +29 -6
- data/lib/pwn/cron.rb +14 -2
- data/lib/pwn/version.rb +1 -1
- data/spec/integration/reinforced_feedback_loop_spec.rb +806 -12
- data/spec/lib/pwn/ai/agent/reward_spec.rb +64 -1
- data/spec/lib/pwn/ai/agent/tools/ruby_eval_spec.rb +39 -1
- data/third_party/pwn_rdoc.jsonl +33 -4
- metadata +1 -1
|
@@ -1722,14 +1722,24 @@ module PWN
|
|
|
1722
1722
|
c = opts[:claim].to_s
|
|
1723
1723
|
return :doc if opts[:url] || c =~ URI::DEFAULT_PARSER.make_regexp(%w[http https])
|
|
1724
1724
|
return :cve if c =~ /CVE-\d{4}-\d{4,}/i
|
|
1725
|
-
|
|
1725
|
+
# P26 — full semver (x.y.z) OR "latest X is [v]N.N" product claims.
|
|
1726
|
+
# Bare two-part floats alone stay :generic (metrics like "cap 0.2").
|
|
1727
|
+
return :version if c =~ /\b[A-Za-z][\w.+-]{2,}\s+v?\d+\.\d+\.\d+\b/
|
|
1728
|
+
return :version if c =~ /\blatest\b.+\bis\b\s+v?\d+\.\d+/i
|
|
1726
1729
|
|
|
1727
1730
|
:generic
|
|
1728
1731
|
end
|
|
1729
1732
|
|
|
1730
1733
|
private_class_method def self.commit_verdict(opts = {})
|
|
1731
|
-
claim = opts[:claim]
|
|
1734
|
+
claim = opts[:claim].to_s
|
|
1732
1735
|
ev = Array(opts[:evidence]).first
|
|
1736
|
+
# P26 — never pollute learning.jsonl with :unknown on metric crumbs
|
|
1737
|
+
# or sub-8-char blobs; those are not human-reviewable world claims.
|
|
1738
|
+
junk_unknown = opts[:verdict].to_s == 'unknown' && (
|
|
1739
|
+
claim.length < 12 ||
|
|
1740
|
+
claim.match?(/\A(?:cap|share|proxy|judge|success|only|now|clears|gap|score|rate|mean|brier|overconf|distrust|trajectory_fraction|handler|orm|prm|delta|limit|window|pct|percent|ms|iter|budget|conf|confidence|ruby|python|linux|kernel|host|cwd|e\.g)\b/i) ||
|
|
1741
|
+
(claim.match?(/\d+\.\d+/) && !claim.match?(/CVE-|\d+\.\d+\.\d+/i))
|
|
1742
|
+
)
|
|
1733
1743
|
case opts[:verdict]
|
|
1734
1744
|
when :refuted
|
|
1735
1745
|
Mistakes.record(tool: 'assumption', error: "REFUTED (extro_verify #{opts[:kind]}, conf=#{opts[:confidence].round(2)}): #{claim}", args: ev&.dig(:final_url), source: :model) if defined?(Mistakes)
|
|
@@ -1739,8 +1749,12 @@ module PWN
|
|
|
1739
1749
|
observe(source: 'extro_verify', category: :intel, target: ev&.dig(:final_url), data: claim, tags: %w[verify confirmed], ttl: 30 * 24 * 3600)
|
|
1740
1750
|
:extro_observe
|
|
1741
1751
|
else
|
|
1742
|
-
|
|
1743
|
-
|
|
1752
|
+
if junk_unknown
|
|
1753
|
+
:skipped_junk
|
|
1754
|
+
else
|
|
1755
|
+
Learning.note_outcome(task: "extro_verify: #{claim[0, 120]}", success: false, details: 'verdict :unknown - needs human review', tags: %w[needs_human extro_verify]) if defined?(Learning) && Learning.respond_to?(:note_outcome)
|
|
1756
|
+
:learning_note
|
|
1757
|
+
end
|
|
1744
1758
|
end
|
|
1745
1759
|
rescue StandardError
|
|
1746
1760
|
nil
|
|
@@ -25,8 +25,41 @@ module PWN
|
|
|
25
25
|
module Learning
|
|
26
26
|
LEARNING_FILE = File.join(Dir.home, '.pwn', 'learning.jsonl')
|
|
27
27
|
FINETUNE_DIR = File.join(Dir.home, '.pwn', 'finetune')
|
|
28
|
+
# P0 — post-answer introspect must not train "stop early" while
|
|
29
|
+
# spending the iteration budget after the final. Soft cap skips
|
|
30
|
+
# expensive stages (tool critic, PRM-LLM, reflect, extrospect);
|
|
31
|
+
# hard cap keeps only note_outcome + judge(heuristic) + sentinel.
|
|
32
|
+
INTROSPECT_SOFT_MS = 2_500
|
|
33
|
+
INTROSPECT_HARD_MS = 8_000
|
|
34
|
+
INTROSPECT_MIN_STAGES = %i[judge note_outcome fold_judge sentinel].freeze
|
|
35
|
+
|
|
28
36
|
MAX_MEMORY_ENTRIES = 200
|
|
29
|
-
|
|
37
|
+
# E3/P26 — only CVE-ids or software-name + full semver (x.y.z).
|
|
38
|
+
# Two-part floats ("cap 0.2", "proxy 1.0", "judge 37.0") are RL
|
|
39
|
+
# metric crumbs that were scraped by verify_as_reward and flooded
|
|
40
|
+
# learning.jsonl with extro_verify :unknown failures.
|
|
41
|
+
CLAIM_METRIC_WORDS = %w[
|
|
42
|
+
cap share proxy judge success only now clears gap score rate mean
|
|
43
|
+
brier overconf distrust trajectory_fraction handler orm prm delta
|
|
44
|
+
limit window pct percent ms iter budget conf confidence n
|
|
45
|
+
ruby python linux kernel host cwd e.g e.g.
|
|
46
|
+
].freeze
|
|
47
|
+
CLAIM_RX = /
|
|
48
|
+
CVE-\d{4}-\d{4,7}
|
|
49
|
+
|
|
|
50
|
+
\b
|
|
51
|
+
(?!
|
|
52
|
+
(?:cap|share|proxy|judge|success|only|now|clears|gap|score|rate|mean|
|
|
53
|
+
brier|overconf|distrust|trajectory_fraction|handler|orm|prm|delta|
|
|
54
|
+
limit|window|pct|percent|ms|iter|budget|conf|confidence|
|
|
55
|
+
ruby|python|linux|kernel|host|cwd|e\.g)
|
|
56
|
+
\b
|
|
57
|
+
)
|
|
58
|
+
[A-Za-z][\w.+-]{2,}
|
|
59
|
+
\s+
|
|
60
|
+
v?\d+\.\d+\.\d+(?:[-+][\w.]+)?
|
|
61
|
+
\b
|
|
62
|
+
/x
|
|
30
63
|
|
|
31
64
|
# Supported Method Parameters::
|
|
32
65
|
# entry = PWN::AI::Agent::Learning.note_outcome(
|
|
@@ -168,7 +201,10 @@ module PWN
|
|
|
168
201
|
# C2 — strict success:true only (excludes HER success:'soft'). Also
|
|
169
202
|
# down-weight any residual hindsight-tagged rows so partial failures
|
|
170
203
|
# never launder into full-strength few-shot exemplars.
|
|
204
|
+
# P20 — strict success:true AND prefer high judge scores. Drop rows
|
|
205
|
+
# with explicit low ORM score so proxy-true / judge-low cannot be few-shot.
|
|
171
206
|
pool = outcomes(limit: 500, success: true).reject { |r| r[:session_id].to_s.empty? }
|
|
207
|
+
pool = pool.reject { |r| r.key?(:score) && r[:score].to_f < 0.6 }
|
|
172
208
|
scored = pool.map do |r|
|
|
173
209
|
sim = tokens.count { |t| r[:task].to_s.downcase.include?(t) }.to_f / tokens.length
|
|
174
210
|
age_d = (now - Time.parse(r[:timestamp].to_s)) / 86_400.0
|
|
@@ -202,26 +238,65 @@ module PWN
|
|
|
202
238
|
# Modelfile` over the export - the only path to ACTUAL parity with a
|
|
203
239
|
# frontier model, because it changes the weights not just the scaffold.
|
|
204
240
|
|
|
241
|
+
# P12 — SFT quality gate (as hard as DPO source-cap): drop HER/soft,
|
|
242
|
+
# low judge_score, auto-only noise without score, and PRM-compress
|
|
243
|
+
# trajectories so LoRA is not 5MB of "how we flailed".
|
|
244
|
+
SFT_MIN_SCORE = 0.6
|
|
245
|
+
SFT_MAX_TOOL_CHARS = 1_200
|
|
246
|
+
|
|
205
247
|
public_class_method def self.export_finetune(opts = {})
|
|
206
248
|
fmt = (opts[:format] || :sharegpt).to_sym
|
|
207
249
|
min_tools = (opts[:min_tools] || 1).to_i
|
|
250
|
+
min_score = (opts[:min_score] || SFT_MIN_SCORE).to_f
|
|
251
|
+
compress = opts.key?(:compress) ? opts[:compress] : true
|
|
208
252
|
FileUtils.mkdir_p(FINETUNE_DIR)
|
|
209
253
|
out = opts[:out] || File.join(FINETUNE_DIR, "pwn-#{Time.now.utc.strftime('%Y%m%d')}.jsonl")
|
|
210
254
|
|
|
211
|
-
# 4.1 — exclude HER soft-success
|
|
255
|
+
# 4.1 / P12 — exclude HER soft-success + low-score + untagged auto flail
|
|
212
256
|
gold = outcomes(limit: 10_000, success: true).reject do |r|
|
|
213
|
-
r[:
|
|
214
|
-
|
|
257
|
+
tags = Array(r[:tags]).map(&:to_s)
|
|
258
|
+
soft = r[:success].to_s == 'soft' || tags.intersect?(%w[hindsight her soft])
|
|
259
|
+
low = !r[:score].nil? && r[:score].to_f < min_score
|
|
260
|
+
# require a score when present in corpus era that has scores
|
|
261
|
+
soft || low
|
|
215
262
|
end
|
|
216
|
-
|
|
263
|
+
# prefer highest-score outcome per session
|
|
264
|
+
by_sid = {}
|
|
265
|
+
gold.each do |r|
|
|
266
|
+
sid = r[:session_id].to_s
|
|
267
|
+
next if sid.empty?
|
|
268
|
+
|
|
269
|
+
prev = by_sid[sid]
|
|
270
|
+
by_sid[sid] = r if prev.nil? || r[:score].to_f >= prev[:score].to_f
|
|
271
|
+
end
|
|
272
|
+
sids = by_sid.keys
|
|
217
273
|
rows = 0
|
|
274
|
+
dropped = { tools: 0, empty: 0, load: 0 }
|
|
218
275
|
File.open(out, 'w') do |f|
|
|
219
276
|
sids.each do |sid|
|
|
220
|
-
t =
|
|
221
|
-
|
|
277
|
+
t = begin
|
|
278
|
+
PWN::Sessions.load(session_id: sid)
|
|
279
|
+
rescue StandardError
|
|
280
|
+
dropped[:load] += 1
|
|
281
|
+
next
|
|
282
|
+
end
|
|
283
|
+
tool_n = t.count { |e| e[:role].to_s == 'tool' }
|
|
284
|
+
if tool_n < min_tools
|
|
285
|
+
dropped[:tools] += 1
|
|
286
|
+
next
|
|
287
|
+
end
|
|
288
|
+
|
|
289
|
+
conv = if compress
|
|
290
|
+
compress_finetune_trace(transcript: t, max_tool_chars: SFT_MAX_TOOL_CHARS)
|
|
291
|
+
else
|
|
292
|
+
t.map { |e| { role: e[:role].to_s, content: e[:content].to_s } }
|
|
293
|
+
.reject { |e| e[:role] == 'system' && e[:content].start_with?('Session started') }
|
|
294
|
+
end
|
|
295
|
+
if conv.nil? || conv.empty? || conv.none? { |m| m[:role].to_s == 'assistant' }
|
|
296
|
+
dropped[:empty] += 1
|
|
297
|
+
next
|
|
298
|
+
end
|
|
222
299
|
|
|
223
|
-
conv = t.map { |e| { role: e[:role].to_s, content: e[:content].to_s } }
|
|
224
|
-
.reject { |e| e[:role] == 'system' && e[:content].start_with?('Session started') }
|
|
225
300
|
line = case fmt
|
|
226
301
|
when :openai_jsonl then { messages: conv }
|
|
227
302
|
else { conversations: conv.map { |m| { from: sharegpt_role(role: m[:role]), value: m[:content] } } }
|
|
@@ -230,7 +305,11 @@ module PWN
|
|
|
230
305
|
rows += 1
|
|
231
306
|
end
|
|
232
307
|
end
|
|
233
|
-
{
|
|
308
|
+
{
|
|
309
|
+
path: out, format: fmt, sessions: sids.length, samples: rows,
|
|
310
|
+
bytes: File.size(out), min_score: min_score, compressed: compress,
|
|
311
|
+
dropped: dropped
|
|
312
|
+
}
|
|
234
313
|
end
|
|
235
314
|
|
|
236
315
|
# Supported Method Parameters::
|
|
@@ -313,10 +392,33 @@ module PWN
|
|
|
313
392
|
return unless session_id
|
|
314
393
|
return unless auto_introspect_enabled?
|
|
315
394
|
|
|
395
|
+
t0 = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
396
|
+
stages_run = []
|
|
397
|
+
stages_skipped = []
|
|
398
|
+
budget_hot = begin
|
|
399
|
+
defined?(Loop) && Loop.respond_to?(:budget_exhaustion_hot?, true) &&
|
|
400
|
+
Loop.send(:budget_exhaustion_hot?)
|
|
401
|
+
rescue StandardError
|
|
402
|
+
false
|
|
403
|
+
end
|
|
404
|
+
|
|
405
|
+
elapsed_ms = lambda do
|
|
406
|
+
((Process.clock_gettime(Process::CLOCK_MONOTONIC) - t0) * 1000).round
|
|
407
|
+
end
|
|
408
|
+
# soft = skip expensive; hard = stop almost everything
|
|
409
|
+
over_soft = lambda do
|
|
410
|
+
ms = elapsed_ms.call
|
|
411
|
+
ms >= INTROSPECT_SOFT_MS || budget_hot
|
|
412
|
+
end
|
|
413
|
+
over_hard = lambda do
|
|
414
|
+
ms = elapsed_ms.call
|
|
415
|
+
ms >= INTROSPECT_HARD_MS
|
|
416
|
+
end
|
|
417
|
+
|
|
316
418
|
proxy_ok = infer_success(session_id: session_id, final: opts[:final])
|
|
317
|
-
|
|
318
|
-
#
|
|
319
|
-
#
|
|
419
|
+
|
|
420
|
+
# S3 critic — BEFORE reward so verdict is evidence.
|
|
421
|
+
# P24/P0 — budget_hot or soft-cap → text_only or skip.
|
|
320
422
|
force_critic = begin
|
|
321
423
|
eng = (PWN::Env.dig(:ai, :active) if defined?(PWN::Env))
|
|
322
424
|
cal = defined?(Metrics) ? Metrics.calibration(engine: eng) : { n: 0 }
|
|
@@ -324,35 +426,63 @@ module PWN
|
|
|
324
426
|
rescue StandardError
|
|
325
427
|
false
|
|
326
428
|
end
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
429
|
+
# P0 — when W1 mix urgently needs :critic pairs, prefer running critic
|
|
430
|
+
need_critic_mix = begin
|
|
431
|
+
mix = defined?(Reward) && Reward.respond_to?(:generator_mix) ? Reward.generator_mix : {}
|
|
432
|
+
Array(mix[:urgent]).include?('critic')
|
|
433
|
+
rescue StandardError
|
|
434
|
+
false
|
|
435
|
+
end
|
|
436
|
+
|
|
437
|
+
crit = { verdict: :pass, source: :skipped }
|
|
438
|
+
# Single skip path (Lint/DuplicateBranch): hard-budget OR no Curriculum.
|
|
439
|
+
if !defined?(Curriculum) || (over_hard.call && !need_critic_mix)
|
|
440
|
+
stages_skipped << :critic
|
|
441
|
+
elsif budget_hot || (over_soft.call && !force_critic && !need_critic_mix)
|
|
442
|
+
stages_run << :critic_text_only
|
|
443
|
+
crit = Curriculum.critic(
|
|
444
|
+
request: opts[:request],
|
|
445
|
+
final: opts[:final],
|
|
446
|
+
session_id: session_id,
|
|
447
|
+
text_only: true
|
|
448
|
+
)
|
|
449
|
+
elsif force_critic
|
|
450
|
+
stages_run << :critic_forced
|
|
451
|
+
prev = (PWN::Env[:ai][:agent][:critic] if defined?(PWN::Env) && PWN::Env[:ai].is_a?(Hash) && PWN::Env[:ai][:agent].is_a?(Hash))
|
|
452
|
+
begin
|
|
453
|
+
PWN::Env[:ai][:agent][:critic] = true if defined?(PWN::Env) && PWN::Env[:ai].is_a?(Hash) && PWN::Env[:ai][:agent].is_a?(Hash) && !PWN::Env[:ai][:agent].frozen?
|
|
454
|
+
crit = Curriculum.critic(request: opts[:request], final: opts[:final], session_id: session_id)
|
|
455
|
+
ensure
|
|
456
|
+
PWN::Env[:ai][:agent][:critic] = prev if defined?(PWN::Env) && PWN::Env[:ai].is_a?(Hash) && PWN::Env[:ai][:agent].is_a?(Hash) && !PWN::Env[:ai][:agent].frozen?
|
|
457
|
+
end
|
|
458
|
+
else
|
|
459
|
+
stages_run << :critic
|
|
460
|
+
crit = Curriculum.critic(request: opts[:request], final: opts[:final], session_id: session_id)
|
|
461
|
+
end
|
|
462
|
+
|
|
463
|
+
# R1 judge — always attempt (heuristic is cheap; LLM gated inside)
|
|
464
|
+
stages_run << :judge
|
|
343
465
|
v = Reward.judge(request: opts[:request], final: opts[:final], session_id: session_id, proxy_ok: proxy_ok) if defined?(Reward)
|
|
344
466
|
v ||= { score: proxy_ok ? 1.0 : 0.0, success: proxy_ok, verdict: proxy_ok ? :solved : :wrong }
|
|
345
467
|
v[:score] = [v[:score], 0.3].min if crit[:verdict] == :flaw
|
|
346
468
|
ok = v[:score] >= 0.6
|
|
347
469
|
|
|
348
|
-
# W1
|
|
349
|
-
# user correction on the previous turn.
|
|
470
|
+
# W1 pending user_correction pair
|
|
350
471
|
pend = Thread.current[:pwn_pending_pref]
|
|
351
472
|
if pend && ok && defined?(Reward)
|
|
352
|
-
|
|
473
|
+
stages_run << :user_correction_pref
|
|
474
|
+
Reward.record_preference(
|
|
475
|
+
prompt: pend[:prompt],
|
|
476
|
+
rejected: pend[:rejected],
|
|
477
|
+
chosen: opts[:final].to_s,
|
|
478
|
+
source: :user_correction,
|
|
479
|
+
shape: :revised_answer,
|
|
480
|
+
force: true
|
|
481
|
+
)
|
|
353
482
|
Thread.current[:pwn_pending_pref] = nil
|
|
354
483
|
end
|
|
355
484
|
|
|
485
|
+
stages_run << :note_outcome
|
|
356
486
|
note_outcome(
|
|
357
487
|
task: opts[:request].to_s[0, 120],
|
|
358
488
|
success: ok,
|
|
@@ -361,32 +491,69 @@ module PWN
|
|
|
361
491
|
session_id: session_id,
|
|
362
492
|
tags: ['auto', 'loop', v[:verdict].to_s]
|
|
363
493
|
)
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
494
|
+
|
|
495
|
+
stages_run << :fold_judge
|
|
496
|
+
fold_judge_into_metrics(session_id: session_id, score: v[:score], confidence: v[:confidence])
|
|
497
|
+
|
|
498
|
+
# R2 PRM — skip under hard cap (expensive LLM); keep under soft if heuristic path
|
|
499
|
+
if over_hard.call
|
|
500
|
+
stages_skipped << :prm
|
|
501
|
+
elsif defined?(Reward)
|
|
502
|
+
stages_run << :prm
|
|
503
|
+
Reward.prm(request: opts[:request], session_id: session_id)
|
|
504
|
+
end
|
|
505
|
+
|
|
506
|
+
# C3 HER — only on failure; skip hard
|
|
507
|
+
if !ok && defined?(Curriculum) && !over_hard.call
|
|
508
|
+
stages_run << :hindsight
|
|
509
|
+
Curriculum.hindsight(request: opts[:request], final: opts[:final], session_id: session_id)
|
|
510
|
+
else
|
|
511
|
+
stages_skipped << :hindsight unless ok
|
|
512
|
+
end
|
|
513
|
+
|
|
514
|
+
# W3 calibrate — cheap; always when predicted available
|
|
515
|
+
predicted = opts[:predicted]
|
|
516
|
+
predicted = recover_predicted_from_session(session_id: session_id) if predicted.nil?
|
|
517
|
+
if !predicted.nil? && defined?(Curriculum)
|
|
518
|
+
stages_run << :calibrate
|
|
519
|
+
Curriculum.calibrate(predicted: predicted, actual: v[:score], engine: PWN::Env.dig(:ai, :active))
|
|
520
|
+
end
|
|
521
|
+
|
|
522
|
+
# reflect on success — skip soft/hard (LLM + memory writes)
|
|
523
|
+
if ok && !over_soft.call
|
|
524
|
+
stages_run << :reflect
|
|
525
|
+
reflect(session_id: session_id)
|
|
526
|
+
elsif ok
|
|
527
|
+
stages_skipped << :reflect
|
|
528
|
+
end
|
|
529
|
+
|
|
530
|
+
# R3 sentinel — cheap disk math; always
|
|
531
|
+
if defined?(Reward)
|
|
532
|
+
stages_run << :sentinel
|
|
533
|
+
Reward.sentinel
|
|
534
|
+
end
|
|
535
|
+
|
|
536
|
+
# E ambient extrospect — skip soft (can launch probes)
|
|
537
|
+
if defined?(Extrospection) && !over_soft.call
|
|
538
|
+
stages_run << :extrospect
|
|
539
|
+
Extrospection.auto_extrospect(session_id: session_id)
|
|
540
|
+
else
|
|
541
|
+
stages_skipped << :extrospect
|
|
542
|
+
end
|
|
543
|
+
|
|
544
|
+
{
|
|
545
|
+
ok: ok,
|
|
546
|
+
score: v[:score],
|
|
547
|
+
elapsed_ms: elapsed_ms.call,
|
|
548
|
+
budget_hot: budget_hot,
|
|
549
|
+
stages_run: stages_run,
|
|
550
|
+
stages_skipped: stages_skipped
|
|
551
|
+
}
|
|
374
552
|
rescue StandardError => e
|
|
375
553
|
warn "[pwn-ai/learning] auto_introspect swallowed: #{e.class}: #{e.message}"
|
|
376
554
|
nil
|
|
377
555
|
end
|
|
378
556
|
|
|
379
|
-
# Supported Method Parameters::
|
|
380
|
-
# PWN::AI::Agent::Learning.flip_last_outcome(
|
|
381
|
-
# session_id: 'optional - only flip if the newest outcome belongs to this session',
|
|
382
|
-
# reason: 'optional - why it is being flipped (usually the user correction text)'
|
|
383
|
-
# )
|
|
384
|
-
#
|
|
385
|
-
# Rewrites the most-recently-appended learning.jsonl entry from
|
|
386
|
-
# success:true to success:false. Called by Mistakes.check_user_correction
|
|
387
|
-
# when the user's next message rejects the previous answer, so the
|
|
388
|
-
# 100 %-success illusion is broken and the failure enters the corpus.
|
|
389
|
-
|
|
390
557
|
public_class_method def self.flip_last_outcome(opts = {})
|
|
391
558
|
return { flipped: false } unless File.exist?(LEARNING_FILE)
|
|
392
559
|
|
|
@@ -473,6 +640,47 @@ module PWN
|
|
|
473
640
|
# privates
|
|
474
641
|
# -------------------------------------------------------------
|
|
475
642
|
|
|
643
|
+
# P20 — attribute episode judge score to every tool used in session.
|
|
644
|
+
private_class_method def self.fold_judge_into_metrics(opts = {})
|
|
645
|
+
return unless defined?(Metrics) && Metrics.respond_to?(:record_judge)
|
|
646
|
+
|
|
647
|
+
sid = opts[:session_id]
|
|
648
|
+
score = opts[:score]
|
|
649
|
+
conf = opts[:confidence]
|
|
650
|
+
return if sid.to_s.empty? || score.nil?
|
|
651
|
+
return unless defined?(PWN::Sessions)
|
|
652
|
+
|
|
653
|
+
names = PWN::Sessions.load(session_id: sid)
|
|
654
|
+
.select { |e| e[:role].to_s == 'tool' }
|
|
655
|
+
.map { |e| e[:content].to_s[/\A([a-z0-9_]+)\s*→/i, 1] }
|
|
656
|
+
.compact
|
|
657
|
+
.uniq
|
|
658
|
+
names.each { |n| Metrics.record_judge(name: n, score: score, confidence: conf) }
|
|
659
|
+
rescue StandardError
|
|
660
|
+
nil
|
|
661
|
+
end
|
|
662
|
+
|
|
663
|
+
# P22 — recover plan_first p(success)= from session transcript when
|
|
664
|
+
# the live return floated away (rescue path / degrade).
|
|
665
|
+
private_class_method def self.recover_predicted_from_session(opts = {})
|
|
666
|
+
# Prefer live stash from plan_first in this process.
|
|
667
|
+
stash = Thread.current[:pwn_plan_predicted]
|
|
668
|
+
return stash.to_f.clamp(0.0, 1.0) unless stash.nil?
|
|
669
|
+
|
|
670
|
+
sid = opts[:session_id]
|
|
671
|
+
return nil if sid.to_s.empty? || !defined?(PWN::Sessions)
|
|
672
|
+
|
|
673
|
+
entries = PWN::Sessions.load(session_id: sid)
|
|
674
|
+
plan = entries.reverse.find do |e|
|
|
675
|
+
e[:role].to_s == 'assistant' && e[:content].to_s.match?(/\bPLAN:\b|p\(success\)\s*=/i)
|
|
676
|
+
end
|
|
677
|
+
return nil unless plan
|
|
678
|
+
|
|
679
|
+
plan[:content].to_s[/p\(success\)\s*=\s*([01](?:\.\d+)?)/i, 1]&.to_f
|
|
680
|
+
rescue StandardError
|
|
681
|
+
nil
|
|
682
|
+
end
|
|
683
|
+
|
|
476
684
|
private_class_method def self.auto_introspect_enabled?
|
|
477
685
|
return false unless defined?(PWN::Env) && PWN::Env.is_a?(Hash)
|
|
478
686
|
|
|
@@ -562,6 +770,23 @@ module PWN
|
|
|
562
770
|
lessons.uniq.first(5)
|
|
563
771
|
end
|
|
564
772
|
|
|
773
|
+
# P26 — is this CLAIM_RX hit worth spending a headless browser on?
|
|
774
|
+
# Reject metric crumbs, OS/runtime banner lines, and bare versions.
|
|
775
|
+
private_class_method def self.checkable_claim?(opts = {})
|
|
776
|
+
c = opts[:claim].to_s.strip
|
|
777
|
+
return false if c.empty? || c.length < 8
|
|
778
|
+
return true if c.match?(/CVE-\d{4}-\d{4,7}/i)
|
|
779
|
+
|
|
780
|
+
head = c[/\A[A-Za-z][\w.+-]*/].to_s
|
|
781
|
+
return false if CLAIM_METRIC_WORDS.any? { |w| head.casecmp?(w) }
|
|
782
|
+
# require full semver x.y.z for non-CVE claims
|
|
783
|
+
return false unless c.match?(/\bv?\d+\.\d+\.\d+/)
|
|
784
|
+
|
|
785
|
+
true
|
|
786
|
+
rescue StandardError
|
|
787
|
+
false
|
|
788
|
+
end
|
|
789
|
+
|
|
565
790
|
# Auto fact-check post-filter: local models hallucinate CVEs /
|
|
566
791
|
# versions ~5-10x more than frontier ones. When the active engine is
|
|
567
792
|
# :ollama, scan the final for CVE / version-shaped claims and hand
|
|
@@ -572,12 +797,41 @@ module PWN
|
|
|
572
797
|
return unless defined?(PWN::Env) && PWN::Env.dig(:ai, :active).to_s.downcase.to_sym == :ollama
|
|
573
798
|
return unless defined?(Extrospection) && Extrospection.respond_to?(:verify)
|
|
574
799
|
|
|
575
|
-
claims = opts[:final].to_s.scan(CLAIM_RX).flatten.compact.uniq
|
|
800
|
+
claims = opts[:final].to_s.scan(CLAIM_RX).flatten.compact.uniq
|
|
801
|
+
claims = claims.select { |c| checkable_claim?(claim: c) }.first(3)
|
|
576
802
|
claims.each { |c| Extrospection.verify(claim: c, commit: true) }
|
|
577
803
|
rescue StandardError => e
|
|
578
804
|
warn "[pwn-ai/learning] fact_check swallowed: #{e.class}: #{e.message}"
|
|
579
805
|
end
|
|
580
806
|
|
|
807
|
+
# P12 — PRM-aware SFT trace: keep user + reward>0 tools (else first N) +
|
|
808
|
+
# final assistant, truncate tool payloads. Drops system noise.
|
|
809
|
+
private_class_method def self.compress_finetune_trace(opts = {})
|
|
810
|
+
t = Array(opts[:transcript])
|
|
811
|
+
cap = (opts[:max_tool_chars] || SFT_MAX_TOOL_CHARS).to_i
|
|
812
|
+
fin_idx = t.rindex { |e| e[:role].to_s == 'assistant' && !e[:content].to_s.strip.empty? }
|
|
813
|
+
return [] if fin_idx.nil?
|
|
814
|
+
|
|
815
|
+
user_idx = t[0...fin_idx].rindex { |e| e[:role].to_s == 'user' }
|
|
816
|
+
return [] if user_idx.nil?
|
|
817
|
+
|
|
818
|
+
window = t[user_idx..fin_idx]
|
|
819
|
+
tools = window.select { |e| e[:role].to_s == 'tool' }
|
|
820
|
+
rewarded = tools.select { |e| e[:step_reward].to_i.positive? }
|
|
821
|
+
tools = rewarded unless rewarded.empty?
|
|
822
|
+
tools = tools.first(8)
|
|
823
|
+
|
|
824
|
+
out = []
|
|
825
|
+
out << { role: 'user', content: t[user_idx][:content].to_s[0, 2_000] }
|
|
826
|
+
tools.each do |e|
|
|
827
|
+
out << { role: 'tool', content: e[:content].to_s[0, cap] }
|
|
828
|
+
end
|
|
829
|
+
out << { role: 'assistant', content: t[fin_idx][:content].to_s[0, 4_000] }
|
|
830
|
+
out
|
|
831
|
+
rescue StandardError
|
|
832
|
+
[]
|
|
833
|
+
end
|
|
834
|
+
|
|
581
835
|
private_class_method def self.compress_exemplar(opts = {})
|
|
582
836
|
sid = opts[:session_id]
|
|
583
837
|
cap = opts[:max_msgs] || 6
|