pwn 0.5.680 → 0.5.682
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.gitignore +1 -0
- data/documentation/AI-Integration.md +1 -1
- data/documentation/Agent-Tool-Registry.md +1 -1
- data/documentation/Configuration.md +3 -6
- data/documentation/How-PWN-Works.md +2 -2
- data/documentation/Reinforcement-Learning.md +1 -1
- data/documentation/diagrams/dot/task-summarizer.dot +3 -3
- data/documentation/pwn-ai-Agent.md +16 -35
- data/lib/pwn/ai/agent/loop.rb +230 -191
- data/lib/pwn/ai/agent/mistakes.rb +17 -0
- data/lib/pwn/ai/agent/policy.rb +55 -5
- data/lib/pwn/ai/agent/prompt_builder.rb +38 -12
- data/lib/pwn/ai/agent/registry.rb +17 -9
- data/lib/pwn/ai/agent/task_summarizer.rb +374 -503
- data/lib/pwn/config.rb +1 -1
- data/lib/pwn/version.rb +1 -1
- data/spec/integration/prompt_builder_spec.rb +6 -4
- data/spec/lib/pwn/ai/agent/loop_spec.rb +251 -33
- data/spec/lib/pwn/ai/agent/mistakes_spec.rb +14 -0
- data/spec/lib/pwn/ai/agent/policy_spec.rb +52 -1
- data/spec/lib/pwn/ai/agent/prompt_builder_spec.rb +4 -9
- data/spec/lib/pwn/ai/agent/registry_spec.rb +22 -2
- data/spec/lib/pwn/ai/agent/signal_hygiene_spec.rb +4 -5
- data/spec/lib/pwn/ai/agent/task_summarizer_spec.rb +321 -90
- data/third_party/pwn_rdoc.jsonl +24 -9
- metadata +1 -1
|
@@ -7,22 +7,12 @@ module PWN
|
|
|
7
7
|
module Agent
|
|
8
8
|
# High-level executive brief of the work the agent is about to do.
|
|
9
9
|
#
|
|
10
|
-
#
|
|
11
|
-
#
|
|
12
|
-
#
|
|
13
|
-
#
|
|
14
|
-
#
|
|
15
|
-
#
|
|
16
|
-
# 1. plan(request:) — on user submit, break the goal into an ordered
|
|
17
|
-
# list of plain-English tasks via the active LLM (each task is a
|
|
18
|
-
# coherent unit that may require many tool calls). ONLY autonomous goals (not bare statements/questions).
|
|
19
|
-
# Works for any — no static per-domain task lists.
|
|
20
|
-
# 2. about_to(tools:) — per tool-batch brief led by the active
|
|
21
|
-
# "task k/n: <english>" item (same vocabulary as emit_plan!).
|
|
22
|
-
# Tool counts/intents are a secondary "via …" suffix only.
|
|
23
|
-
# 3. active_task_prompt / plan_context — injected into Loop messages
|
|
24
|
-
# so generated tasks steer tool choice, not only the TUI.
|
|
25
|
-
# 4. record! emits an advancement brief when plan_idx moves forward.
|
|
10
|
+
# Every pwn-ai request is a goal. There is no statement/question/goal
|
|
11
|
+
# request type. English tangible tasks are an advisory compass:
|
|
12
|
+
# 1. plan(request:) — break the goal into ordered plain-English tasks
|
|
13
|
+
# 2. about_to(tools:) — per tool-batch brief led by "task k/n"
|
|
14
|
+
# 3. active_task_prompt — injected into Loop as a compass only
|
|
15
|
+
# 4. record! emits an advancement brief when plan_idx moves
|
|
26
16
|
#
|
|
27
17
|
# Never dumps raw commands or tool results into the task row — those
|
|
28
18
|
# stay on the per-tool lines the REPL already prints.
|
|
@@ -51,10 +41,8 @@ module PWN
|
|
|
51
41
|
}.freeze
|
|
52
42
|
|
|
53
43
|
PLAN_SYSTEM = <<~SYS
|
|
54
|
-
You are the pwn-ai Task Planner
|
|
55
|
-
|
|
56
|
-
accomplish (not a bare statement or a question). Break it into an
|
|
57
|
-
ordered list of tangible work units. Rules:
|
|
44
|
+
You are the pwn-ai Task Planner.
|
|
45
|
+
Break the user request into an ordered list of tangible work units. Rules:
|
|
58
46
|
- 2..12 tasks. Each task is one coherent unit of work (may need many tools).
|
|
59
47
|
- Plain US English. Imperative mood. No tool names, paths, or shell commands.
|
|
60
48
|
- Cover discovery/recon, the core work, verification, and the requested
|
|
@@ -113,10 +101,11 @@ module PWN
|
|
|
113
101
|
# request: 'optional - original user goal string'
|
|
114
102
|
# )
|
|
115
103
|
public_class_method def self.fresh(opts = {})
|
|
116
|
-
req = opts[:request]
|
|
104
|
+
req = canonical_request(request: opts[:request])
|
|
105
|
+
req = opts[:request].to_s if req.empty?
|
|
117
106
|
{
|
|
118
107
|
request: req,
|
|
119
|
-
|
|
108
|
+
original_request: req,
|
|
120
109
|
events: [],
|
|
121
110
|
since_emit: 0,
|
|
122
111
|
last_emit_at: Time.now,
|
|
@@ -140,7 +129,8 @@ module PWN
|
|
|
140
129
|
focus_injected_idx: nil,
|
|
141
130
|
last_advanced_from: nil,
|
|
142
131
|
last_advance_brief: nil,
|
|
143
|
-
tools_on_task: 0
|
|
132
|
+
tools_on_task: 0,
|
|
133
|
+
evidence_blob: ''
|
|
144
134
|
}
|
|
145
135
|
end
|
|
146
136
|
|
|
@@ -276,322 +266,10 @@ module PWN
|
|
|
276
266
|
line
|
|
277
267
|
end
|
|
278
268
|
|
|
279
|
-
#
|
|
280
|
-
# Top-level request kind (statement | question | autonomous_goal).
|
|
281
|
-
# General statements and questions must NOT be multi-step planned.
|
|
282
|
-
# Autonomous goals MUST decompose into ordered tangible work units.
|
|
283
|
-
#
|
|
284
|
-
# Classification priority:
|
|
285
|
-
# 1. Injected kind / llm_kind (tests)
|
|
286
|
-
# 2. Cheap deterministic intent (greeting/howto/recall/recon)
|
|
287
|
-
# 3. LLM classify when request_kind_llm/task_summary_llm is on
|
|
288
|
-
# 4. Regex / structure heuristics (offline fallback)
|
|
289
|
-
#
|
|
290
|
-
# Supported Method Parameters::
|
|
291
|
-
# kind = PWN::AI::Agent::TaskSummarizer.request_kind(
|
|
292
|
-
# request: 'required - user text',
|
|
293
|
-
# kind: 'optional - precomputed symbol',
|
|
294
|
-
# llm_kind: 'optional - injected LLM label (tests)',
|
|
295
|
-
# heuristic_only: 'optional - skip LLM (Boolean)'
|
|
296
|
-
# )
|
|
297
|
-
# ------------------------------------------------------------------
|
|
298
|
-
KIND_LABELS = %i[statement question autonomous_goal].freeze
|
|
299
|
-
|
|
300
|
-
# Review / opinion asks are questions unless the user also asks to implement.
|
|
301
|
-
OPINION_REVIEW_RX = /
|
|
302
|
-
\b(?:
|
|
303
|
-
suggest\s+(?:areas?|ways?|ideas?)|
|
|
304
|
-
areas?\s+for\s+improvement|
|
|
305
|
-
what\s+(?:do\s+you\s+think|would\s+you\s+(?:change|improve))|
|
|
306
|
-
your\s+(?:take|opinion|assessment|review)\b|
|
|
307
|
-
briefly\s+describe|
|
|
308
|
-
review\s+(?:the|this|our)
|
|
309
|
-
)
|
|
310
|
-
/ix
|
|
311
|
-
|
|
312
|
-
IMPLEMENT_RX = /
|
|
313
|
-
\b(?:implement|fix\s+all|apply|patch|land|ship)\b
|
|
314
|
-
/ix
|
|
315
|
-
|
|
316
|
-
STATEMENT_RX = /
|
|
317
|
-
\A\s*(
|
|
318
|
-
(?:fyi|note|noted|heads\s*up|for\s+the\s+record|just\s+so\s+you\s+know)\b |
|
|
319
|
-
(?:i\s+(?:think|believe|notice|see|observed)|looks\s+like)\b |
|
|
320
|
-
(?:the|this|that)\s+(?:\w+\s+){0,6}(?:is|are|was|were|seems|looks)\b
|
|
321
|
-
)
|
|
322
|
-
/ix
|
|
323
|
-
|
|
324
|
-
# Interrogative openers / trailing ? / how-to phrases.
|
|
325
|
-
# Also match mid-line "what/why/..." after a short preface
|
|
326
|
-
# ("excellent - what is my hostname?").
|
|
327
|
-
# Do NOT treat bare "Do three things..." as a question (imperative goal).
|
|
328
|
-
QUESTION_RX = /
|
|
329
|
-
\A\s*(?:
|
|
330
|
-
(?:what|why|when|where|who|whom|whose|which|how)\b
|
|
331
|
-
|
|
|
332
|
-
(?:is|are|was|were|do|does|did|can|could|would|should|may|might|will)\s+
|
|
333
|
-
(?:i|you|we|they|he|she|it|this|that|there|the|a|an|my|your|our|any|
|
|
334
|
-
anyone|someone|everyone|something|anything|everything)\b
|
|
335
|
-
)
|
|
336
|
-
|
|
|
337
|
-
(?:\A|\.\s+|[-–—:]\s+|\b)(?:what|why|when|where|who|whom|whose|which|how)\b
|
|
338
|
-
|
|
|
339
|
-
\?\s*\z
|
|
340
|
-
|
|
|
341
|
-
\b(?:how\s+to|how\s+do\s+i|how\s+can\s+i|explain\s+how|usage\s+of|syntax\s+for)\b
|
|
342
|
-
/ix
|
|
343
|
-
|
|
344
|
-
# Strong "agent, do this" signals — avoid bare nouns like "build"/"scan"
|
|
345
|
-
# inside ordinary prose ("the build is green", "ping scan flags").
|
|
346
|
-
AUTONOMOUS_GOAL_RX = /
|
|
347
|
-
\b(?:
|
|
348
|
-
(?:please\s+)(?:\w+\s+){0,4}(?:implement|fix|patch|refactor|rewrite|migrate|
|
|
349
|
-
deploy|install|compile|commit|push|create|remove|delete|update|change|edit|
|
|
350
|
-
write|run|execute|scan|probe|enumerate|discover|find|locate|audit|harden|
|
|
351
|
-
resolve|debug|diagnose|investigate|optimize|configure|enable|disable|
|
|
352
|
-
start|stop|restart|ship|finish|complete|make|ensure|verify)
|
|
353
|
-
|
|
|
354
|
-
(?:can|could|would)\s+you\s+(?:please\s+)?(?:\w+\s+){0,3}(?:do|run|fix|
|
|
355
|
-
implement|scan|find|make|update|change|write|patch|refactor|add|remove|
|
|
356
|
-
create|locate|resolve|debug|ship|finish)
|
|
357
|
-
|
|
|
358
|
-
\bi\s+need\s+you\s+to\b |
|
|
359
|
-
\byour\s+job\s+is\s+to\b |
|
|
360
|
-
\bgo\s+ahead\b |
|
|
361
|
-
\bdo\s+it\b |
|
|
362
|
-
\bship\s+it\b |
|
|
363
|
-
\bmake\s+it\s+so\b |
|
|
364
|
-
\bmake\s+sure\b |
|
|
365
|
-
# Imperative at start of request (common operator style)
|
|
366
|
-
\A\s*(?:implement|fix|patch|refactor|rewrite|migrate|deploy|install|
|
|
367
|
-
compile|commit|push|create|add|remove|delete|update|change|edit|write|
|
|
368
|
-
run|execute|scan|probe|enumerate|discover|find|locate|audit|harden|
|
|
369
|
-
resolve|debug|diagnose|investigate|optimize|configure|enable|disable|
|
|
370
|
-
start|stop|restart|ship|finish|complete|break|decompose|ensure|verify|
|
|
371
|
-
please)\b
|
|
372
|
-
)
|
|
373
|
-
/ix
|
|
374
|
-
|
|
375
|
-
# Interrogatives that need a live local lookup (hostname, cwd, whoami…).
|
|
376
|
-
# These are autonomous_goal so Loop uses tools — not text-only Q&A.
|
|
377
|
-
NEEDS_LOCAL_EVIDENCE_RX = /
|
|
378
|
-
\b(?:what(?:'?s|\s+is)|show|print|tell\s+me|get|echo|display|check)\b
|
|
379
|
-
.{0,80}\b(?:
|
|
380
|
-
(?:my\s+)?host\s*name|
|
|
381
|
-
(?:my\s+)?(?:ip|ipv4|ipv6)(?:\s+address)?|
|
|
382
|
-
(?:my\s+)?(?:cwd|pwd|working\s+directory|present\s+working\s+directory)|
|
|
383
|
-
whoami|(?:my\s+)?user(?:name)?|(?:logged[- ]?in\s+)?user|
|
|
384
|
-
(?:my\s+)?kernel|uname\b|
|
|
385
|
-
uptime|disk\s+usage|free\s+space|memory\s+usage|
|
|
386
|
-
listening\s+ports?|default\s+route|gateway|
|
|
387
|
-
(?:this\s+)?(?:machine|host|box|system)\s+(?:name|hostname)
|
|
388
|
-
)\b
|
|
389
|
-
/ix
|
|
390
|
-
|
|
391
|
-
KIND_SYSTEM = <<~SYS
|
|
392
|
-
You classify ONE user request for the pwn-ai agent.
|
|
393
|
-
Return ONLY one token from this set:
|
|
394
|
-
statement
|
|
395
|
-
question
|
|
396
|
-
autonomous_goal
|
|
397
|
-
|
|
398
|
-
Definitions:
|
|
399
|
-
- statement: FYI, observation, ack, or greeting. No work requested. No answer needed beyond a brief note.
|
|
400
|
-
- question: asks for knowledge, explanation, syntax, or prior-turn recall that can be answered without the agent performing multi-step host/code work. Do NOT plan tools.
|
|
401
|
-
- autonomous_goal: the agent must DO something — implement/fix/scan/run commands, or answer a fact that requires a live local lookup (hostname, cwd, whoami, listening ports, etc.).
|
|
402
|
-
|
|
403
|
-
Rules:
|
|
404
|
-
- "can you fix/implement/scan...?" is autonomous_goal even with a trailing ?
|
|
405
|
-
- "how to ..." / "what flags does X use" without "do it here" is question
|
|
406
|
-
- "what is my hostname?" / "what is my ip?" is autonomous_goal (needs tools)
|
|
407
|
-
- "FYI the build is green" is statement
|
|
408
|
-
- "suggest areas for improvement" / "what do you think" / review-opinion asks are question unless the user also says implement/fix/apply
|
|
409
|
-
- Prefer autonomous_goal when unsure whether live action is required
|
|
410
|
-
|
|
411
|
-
Output: a single label token. No punctuation, no JSON, no prose.
|
|
412
|
-
SYS
|
|
413
|
-
|
|
414
|
-
# LLM request-kind classifier is ON by default (same knob family as plan LLM).
|
|
415
|
-
# PWN::Env[:ai][:agent][:request_kind_llm] = false disables it (tests/airgap).
|
|
416
|
-
# When unset, follows :task_summary_llm (false in unit specs).
|
|
417
|
-
public_class_method def self.llm_kind_enabled?
|
|
418
|
-
v = PWN::Env.dig(:ai, :agent, :request_kind_llm)
|
|
419
|
-
return !!v unless v.nil?
|
|
420
|
-
|
|
421
|
-
llm_plan_enabled?
|
|
422
|
-
rescue StandardError
|
|
423
|
-
true
|
|
424
|
-
end
|
|
425
|
-
|
|
426
|
-
public_class_method def self.request_kind(opts = {})
|
|
427
|
-
req = opts[:request].to_s
|
|
428
|
-
return :statement if req.strip.empty?
|
|
429
|
-
|
|
430
|
-
# Injected precompute (tests / Loop hand-off).
|
|
431
|
-
if opts.key?(:kind) && !opts[:kind].nil?
|
|
432
|
-
parsed = parse_kind_label(raw: opts[:kind])
|
|
433
|
-
return parsed if parsed
|
|
434
|
-
end
|
|
435
|
-
if opts.key?(:llm_kind)
|
|
436
|
-
parsed = parse_kind_label(raw: opts[:llm_kind])
|
|
437
|
-
return parsed if parsed
|
|
438
|
-
end
|
|
439
|
-
|
|
440
|
-
# Cheap deterministic intent (greeting/howto/recall/recon) — single
|
|
441
|
-
# source with Loop.request_intent; avoids LLM on pure short-circuits.
|
|
442
|
-
intent = nil
|
|
443
|
-
if defined?(PWN::AI::Agent::Loop) && PWN::AI::Agent::Loop.respond_to?(:request_intent)
|
|
444
|
-
begin
|
|
445
|
-
intent = PWN::AI::Agent::Loop.request_intent(request: req)
|
|
446
|
-
case intent
|
|
447
|
-
when :greeting, :empty
|
|
448
|
-
return :statement
|
|
449
|
-
when :howto, :recall
|
|
450
|
-
return :question
|
|
451
|
-
when :recon_act
|
|
452
|
-
return :autonomous_goal
|
|
453
|
-
end
|
|
454
|
-
# :act falls through — may still be statement/question/goal
|
|
455
|
-
rescue StandardError
|
|
456
|
-
# offline / load-order safe
|
|
457
|
-
end
|
|
458
|
-
end
|
|
459
|
-
|
|
460
|
-
# Numbered operator checklist is always an autonomous goal.
|
|
461
|
-
enumerated = begin
|
|
462
|
-
extract_enumerated_steps(goal: req)
|
|
463
|
-
rescue StandardError
|
|
464
|
-
[]
|
|
465
|
-
end
|
|
466
|
-
return :autonomous_goal if enumerated.length >= MIN_PLAN_TASKS
|
|
467
|
-
|
|
468
|
-
# Review / opinion / "suggest areas" stay questions unless implement.
|
|
469
|
-
return :question if req.match?(OPINION_REVIEW_RX) && !req.match?(IMPLEMENT_RX)
|
|
470
|
-
|
|
471
|
-
# Strong agent-do / host-evidence before LLM (stable + offline).
|
|
472
|
-
return :autonomous_goal if req.match?(AUTONOMOUS_GOAL_RX)
|
|
473
|
-
return :autonomous_goal if req.match?(NEEDS_LOCAL_EVIDENCE_RX)
|
|
474
|
-
|
|
475
|
-
# LLM classify ambiguous residual (prefer over brittle regex defaults).
|
|
476
|
-
if !opts[:heuristic_only] && llm_kind_enabled?
|
|
477
|
-
llm_k = llm_classify_kind(request: req)
|
|
478
|
-
return llm_k if llm_k
|
|
479
|
-
end
|
|
480
|
-
# Explicit heuristic_only or LLM miss → offline rules.
|
|
481
|
-
heuristic_request_kind(request: req, intent: intent)
|
|
482
|
-
rescue StandardError
|
|
483
|
-
:autonomous_goal
|
|
484
|
-
end
|
|
485
|
-
|
|
486
|
-
# Offline / fallback classifier (regex + length heuristics).
|
|
487
|
-
public_class_method def self.heuristic_request_kind(opts = {})
|
|
488
|
-
req = opts[:request].to_s
|
|
489
|
-
return :statement if req.strip.empty?
|
|
490
|
-
|
|
491
|
-
return :question if req.match?(OPINION_REVIEW_RX) && !req.match?(IMPLEMENT_RX)
|
|
492
|
-
return :autonomous_goal if req.match?(AUTONOMOUS_GOAL_RX)
|
|
493
|
-
return :autonomous_goal if req.match?(NEEDS_LOCAL_EVIDENCE_RX)
|
|
494
|
-
return :question if req.match?(QUESTION_RX)
|
|
495
|
-
return :statement if req.match?(STATEMENT_RX)
|
|
496
|
-
|
|
497
|
-
stripped = req.gsub(/\s+/, ' ').strip
|
|
498
|
-
# Short bare remarks are statements unless they look like work or Qs.
|
|
499
|
-
if stripped.length < 48 && !stripped.match?(/\b(please|need|want|should|must)\b/i)
|
|
500
|
-
return :question if stripped.include?('?')
|
|
501
|
-
|
|
502
|
-
return :statement
|
|
503
|
-
end
|
|
504
|
-
|
|
505
|
-
:autonomous_goal
|
|
506
|
-
rescue StandardError
|
|
507
|
-
:autonomous_goal
|
|
508
|
-
end
|
|
509
|
-
|
|
510
|
-
# Normalize LLM / caller labels → kind symbol or nil.
|
|
511
|
-
public_class_method def self.parse_kind_label(opts = {})
|
|
512
|
-
raw = opts[:raw]
|
|
513
|
-
return nil if raw.nil?
|
|
514
|
-
|
|
515
|
-
s = raw.to_s.strip.downcase
|
|
516
|
-
return nil if s.empty?
|
|
517
|
-
|
|
518
|
-
# Prefer exact kind tokens anywhere in the response ("Label: question").
|
|
519
|
-
tokens = s.gsub(/[^a-z_]/, ' ').split
|
|
520
|
-
aliases = {
|
|
521
|
-
'statement' => :statement, 'statements' => :statement,
|
|
522
|
-
'fyi' => :statement, 'observation' => :statement,
|
|
523
|
-
'ack' => :statement, 'greeting' => :statement,
|
|
524
|
-
'question' => :question, 'questions' => :question,
|
|
525
|
-
'query' => :question, 'howto' => :question,
|
|
526
|
-
'how_to' => :question, 'recall' => :question,
|
|
527
|
-
'autonomous_goal' => :autonomous_goal, 'goal' => :autonomous_goal,
|
|
528
|
-
'task' => :autonomous_goal, 'act' => :autonomous_goal,
|
|
529
|
-
'action' => :autonomous_goal, 'work' => :autonomous_goal,
|
|
530
|
-
'do' => :autonomous_goal
|
|
531
|
-
}
|
|
532
|
-
tokens.each do |t|
|
|
533
|
-
return aliases[t] if aliases.key?(t)
|
|
534
|
-
end
|
|
535
|
-
|
|
536
|
-
joined = tokens.join('_')
|
|
537
|
-
return aliases[joined] if aliases.key?(joined)
|
|
538
|
-
|
|
539
|
-
sym = s.to_sym
|
|
540
|
-
KIND_LABELS.include?(sym) ? sym : nil
|
|
541
|
-
rescue StandardError
|
|
542
|
-
nil
|
|
543
|
-
end
|
|
544
|
-
|
|
545
|
-
# LLM classify path — Reflect/engine chat, no tools. Returns kind or nil.
|
|
546
|
-
public_class_method def self.llm_classify_kind(opts = {})
|
|
547
|
-
return nil unless llm_kind_enabled?
|
|
548
|
-
|
|
549
|
-
req = opts[:request].to_s
|
|
550
|
-
return :statement if req.strip.empty?
|
|
551
|
-
|
|
552
|
-
raw = chat_for_kind(request: req)
|
|
553
|
-
parse_kind_label(raw: raw)
|
|
554
|
-
rescue StandardError => e
|
|
555
|
-
warn "[pwn-ai/task_summarizer] llm_classify_kind swallowed: #{e.class}: #{e.message}"
|
|
556
|
-
nil
|
|
557
|
-
end
|
|
558
|
-
|
|
559
|
-
# Public so specs can stub the LLM boundary (mirrors chat_for_plan).
|
|
560
|
-
public_class_method def self.chat_for_kind(opts = {})
|
|
561
|
-
req = opts[:request].to_s
|
|
562
|
-
system = KIND_SYSTEM
|
|
563
|
-
user = "USER REQUEST:\n#{req}\n\nLabel:"
|
|
564
|
-
if reflect_available?
|
|
565
|
-
resp = Reflect.on(
|
|
566
|
-
request: user,
|
|
567
|
-
system_role_content: system,
|
|
568
|
-
suppress_pii_warning: true,
|
|
569
|
-
spinner: false,
|
|
570
|
-
timeout: sidecar_timeout,
|
|
571
|
-
quiet: true
|
|
572
|
-
)
|
|
573
|
-
text = reflect_text(resp: resp)
|
|
574
|
-
return text unless text.to_s.strip.empty?
|
|
575
|
-
end
|
|
576
|
-
|
|
577
|
-
engine_chat(request: user, system_role_content: system)
|
|
578
|
-
rescue StandardError => e
|
|
579
|
-
warn "[pwn-ai/task_summarizer] chat_for_kind swallowed: #{e.class}: #{e.message}"
|
|
580
|
-
''
|
|
581
|
-
end
|
|
582
|
-
|
|
583
|
-
# True only for autonomous goals — statements/questions skip multi-step plans.
|
|
584
|
-
#
|
|
585
|
-
# Supported Method Parameters::
|
|
586
|
-
# yes = PWN::AI::Agent::TaskSummarizer.needs_task_breakdown?(
|
|
587
|
-
# request: 'optional - user text',
|
|
588
|
-
# kind: 'optional - precomputed request_kind'
|
|
589
|
-
# )
|
|
269
|
+
# Every request gets a task compass. There is no request type.
|
|
590
270
|
public_class_method def self.needs_task_breakdown?(opts = {})
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
kind.to_sym == :autonomous_goal
|
|
594
|
-
rescue StandardError
|
|
271
|
+
return true if opts.is_a?(Hash)
|
|
272
|
+
|
|
595
273
|
true
|
|
596
274
|
end
|
|
597
275
|
|
|
@@ -613,26 +291,21 @@ module PWN
|
|
|
613
291
|
# )
|
|
614
292
|
# ------------------------------------------------------------------
|
|
615
293
|
public_class_method def self.plan(opts = {})
|
|
616
|
-
|
|
294
|
+
raw = opts[:request].to_s
|
|
295
|
+
raw = (opts[:state][:original_request] || opts[:state][:request]).to_s if raw.strip.empty? && opts[:state].is_a?(Hash)
|
|
296
|
+
goal = canonical_request(request: raw)
|
|
297
|
+
goal = raw.gsub(/\s+/, ' ').strip if goal.empty?
|
|
617
298
|
return [] if goal.empty?
|
|
618
299
|
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
# Statements and questions: never multi-step task breakdown.
|
|
625
|
-
# (Injected :tasks still honored for tests.)
|
|
626
|
-
unless needs_task_breakdown?(kind: kind) || opts.key?(:tasks) || opts.key?(:llm_tasks)
|
|
627
|
-
tasks = []
|
|
628
|
-
source = :"no_breakdown_#{kind}"
|
|
300
|
+
if defined?(PWN::AI::Agent::Loop) &&
|
|
301
|
+
PWN::AI::Agent::Loop.respond_to?(:world_knowledge?) &&
|
|
302
|
+
!opts.key?(:tasks) && !opts.key?(:llm_tasks) &&
|
|
303
|
+
PWN::AI::Agent::Loop.world_knowledge?(request: goal)
|
|
629
304
|
if opts[:state].is_a?(Hash)
|
|
630
|
-
opts[:state][:plan] =
|
|
631
|
-
opts[:state][:
|
|
632
|
-
opts[:state][:request_kind] = kind
|
|
633
|
-
opts[:state][:plan_source] = source
|
|
305
|
+
opts[:state][:plan] = []
|
|
306
|
+
opts[:state][:plan_source] = :no_host_work
|
|
634
307
|
end
|
|
635
|
-
return
|
|
308
|
+
return []
|
|
636
309
|
end
|
|
637
310
|
|
|
638
311
|
source = nil
|
|
@@ -651,7 +324,7 @@ module PWN
|
|
|
651
324
|
tasks = llm_decompose(goal: goal, llm_tasks: opts[:llm_tasks], has_llm_tasks: opts.key?(:llm_tasks))
|
|
652
325
|
source = tasks.any? ? :llm : nil
|
|
653
326
|
if tasks.length < MIN_PLAN_TASKS
|
|
654
|
-
tasks = fallback_decompose(goal: goal
|
|
327
|
+
tasks = fallback_decompose(goal: goal)
|
|
655
328
|
source = :fallback
|
|
656
329
|
end
|
|
657
330
|
end
|
|
@@ -659,24 +332,15 @@ module PWN
|
|
|
659
332
|
tasks = normalize_task_list(tasks: tasks, goal: goal)
|
|
660
333
|
if opts[:state].is_a?(Hash)
|
|
661
334
|
opts[:state][:plan] = tasks
|
|
335
|
+
opts[:state][:original_request] = goal if opts[:state][:original_request].to_s.empty?
|
|
662
336
|
opts[:state][:request] = goal if opts[:state][:request].to_s.empty?
|
|
663
|
-
opts[:state][:request_kind] = kind
|
|
664
337
|
opts[:state][:plan_source] = source
|
|
665
338
|
end
|
|
666
339
|
tasks
|
|
667
340
|
rescue StandardError
|
|
668
341
|
goal = opts[:request].to_s.gsub(/\s+/, ' ').strip
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
rescue StandardError
|
|
672
|
-
:autonomous_goal
|
|
673
|
-
end
|
|
674
|
-
if goal.empty? || !needs_task_breakdown?(kind: kind)
|
|
675
|
-
if opts[:state].is_a?(Hash)
|
|
676
|
-
opts[:state][:plan] = []
|
|
677
|
-
opts[:state][:request_kind] = kind
|
|
678
|
-
opts[:state][:plan_source] = :"no_breakdown_#{kind}"
|
|
679
|
-
end
|
|
342
|
+
if goal.empty?
|
|
343
|
+
opts[:state][:plan] = [] if opts[:state].is_a?(Hash)
|
|
680
344
|
[]
|
|
681
345
|
else
|
|
682
346
|
normalize_task_list(tasks: ["Carry out: #{goal}"], goal: goal)
|
|
@@ -709,8 +373,12 @@ module PWN
|
|
|
709
373
|
list = ['Explain the requested tool usage with concrete examples', 'Present the answer clearly'] if list.empty?
|
|
710
374
|
return list.first(MAX_PLAN_TASKS)
|
|
711
375
|
end
|
|
712
|
-
#
|
|
713
|
-
|
|
376
|
+
# Close with a present/report step — never a test-runner verify.
|
|
377
|
+
# Appending "Verify the result…" used to classify as :verify and
|
|
378
|
+
# block Loop.run until rspec/rubocop printed 0 failures.
|
|
379
|
+
list << 'Present the result and report completion' unless list.last.to_s.match?(
|
|
380
|
+
/verif|test|confirm|rubocop|rake|accept|done|close|summar|json|yaml|table|present|format|convert|report completion|report results/i
|
|
381
|
+
)
|
|
714
382
|
list.first(MAX_PLAN_TASKS)
|
|
715
383
|
end
|
|
716
384
|
|
|
@@ -725,27 +393,10 @@ module PWN
|
|
|
725
393
|
public_class_method def self.format_plan(opts = {})
|
|
726
394
|
list = Array(opts[:tasks]).map(&:to_s).reject(&:empty?)
|
|
727
395
|
goal = opts[:request].to_s.gsub(/\s+/, ' ').strip
|
|
728
|
-
|
|
729
|
-
kind = request_kind(request: goal) if kind.nil? && !goal.empty?
|
|
730
|
-
kind = kind.to_s.empty? ? nil : kind.to_sym
|
|
731
|
-
|
|
732
|
-
# Statements / questions: no multi-step breakdown banner.
|
|
733
|
-
if list.empty?
|
|
734
|
-
return '' if goal.empty?
|
|
735
|
-
|
|
736
|
-
case kind
|
|
737
|
-
when :statement
|
|
738
|
-
return "Request type: statement — no multi-step task breakdown\nNote: #{goal}"
|
|
739
|
-
when :question
|
|
740
|
-
return "Request type: question — no multi-step task breakdown\nQuestion: #{goal}"
|
|
741
|
-
else
|
|
742
|
-
return ''
|
|
743
|
-
end
|
|
744
|
-
end
|
|
396
|
+
return '' if list.empty?
|
|
745
397
|
|
|
746
398
|
n = list.length
|
|
747
399
|
lines = []
|
|
748
|
-
lines << 'Request type: autonomous_goal' if kind == :autonomous_goal || kind.nil?
|
|
749
400
|
lines << "Goal: #{goal}" unless goal.empty?
|
|
750
401
|
lines << "Tangible tasks (#{n}) — each task may leverage one or more tools to complete its objective(s):"
|
|
751
402
|
list.each_with_index do |t, i|
|
|
@@ -763,14 +414,14 @@ module PWN
|
|
|
763
414
|
return nil unless state.is_a?(Hash)
|
|
764
415
|
return state[:plan_text] if state[:plan_emitted] && !state[:plan_text].nil?
|
|
765
416
|
|
|
766
|
-
request = state[:
|
|
417
|
+
request = state[:original_request].to_s
|
|
418
|
+
request = state[:request].to_s if request.empty?
|
|
767
419
|
request = opts[:request].to_s if request.empty?
|
|
768
|
-
|
|
769
|
-
|
|
420
|
+
request = canonical_request(request: request)
|
|
421
|
+
request = opts[:request].to_s.gsub(/\s+/, ' ').strip if request.empty?
|
|
770
422
|
tasks = state[:plan]
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
text = format_plan(tasks: tasks, request: request, request_kind: kind)
|
|
423
|
+
tasks = plan(request: request, state: state) if tasks.nil? || Array(tasks).empty?
|
|
424
|
+
text = format_plan(tasks: tasks, request: request)
|
|
774
425
|
state[:plan] = Array(tasks)
|
|
775
426
|
state[:plan_text] = text
|
|
776
427
|
state[:plan_emitted] = true
|
|
@@ -809,7 +460,7 @@ module PWN
|
|
|
809
460
|
inline = goal.split(/(?=(?:^|\s)\d+\.\s+)/).map(&:strip).reject(&:empty?)
|
|
810
461
|
chunks = inline.map { |c| c.sub(/\A\d+\.\s+/, '') }.reject(&:empty?) if inline.length >= MIN_PLAN_TASKS
|
|
811
462
|
end
|
|
812
|
-
chunks
|
|
463
|
+
reject_scaffold_tasks(tasks: chunks)
|
|
813
464
|
rescue StandardError
|
|
814
465
|
[]
|
|
815
466
|
end
|
|
@@ -989,25 +640,24 @@ module PWN
|
|
|
989
640
|
# goal: 'required - user goal string'
|
|
990
641
|
# )
|
|
991
642
|
public_class_method def self.fallback_decompose(opts = {})
|
|
992
|
-
goal_text = opts[:goal]
|
|
643
|
+
goal_text = canonical_request(request: opts[:goal])
|
|
644
|
+
goal_text = opts[:goal].to_s if goal_text.empty?
|
|
993
645
|
goal_lc = goal_text.downcase
|
|
994
646
|
tasks = []
|
|
995
647
|
|
|
996
|
-
# Statements / questions never get a synthetic multi-step fallback.
|
|
997
|
-
kind = opts[:request_kind] || request_kind(request: goal_text)
|
|
998
|
-
return [] unless needs_task_breakdown?(kind: kind)
|
|
999
|
-
|
|
1000
648
|
# If the operator already spelled improvement bullets, surface them.
|
|
1001
649
|
bullets = goal_text.scan(/(?:^|\s)(?:\d+\.|[-*])\s*([^.;]+)/).flatten.map(&:strip)
|
|
650
|
+
bullets = reject_scaffold_tasks(tasks: bullets)
|
|
1002
651
|
if bullets.length >= MIN_PLAN_TASKS
|
|
1003
652
|
bullets.first(8).each { |b| tasks << b.sub(/\Athe\s+/i, '').sub(/\.\s*\z/, '') }
|
|
1004
653
|
return tasks
|
|
1005
654
|
end
|
|
1006
655
|
|
|
1007
656
|
if howto_goal?(goal: goal_text)
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
657
|
+
return [
|
|
658
|
+
'Explain the requested tool usage with concrete examples',
|
|
659
|
+
'Present the answer clearly'
|
|
660
|
+
]
|
|
1011
661
|
end
|
|
1012
662
|
|
|
1013
663
|
tasks << "Understand the request: #{truncate_goal(goal: goal_text)}"
|
|
@@ -1095,13 +745,8 @@ module PWN
|
|
|
1095
745
|
|
|
1096
746
|
tools = normalize_tools(tools: opts[:tools], name: opts[:name], args: opts[:args])
|
|
1097
747
|
names = tools.map { |t| t[:name] }
|
|
1098
|
-
# Ensure plan exists
|
|
1099
|
-
|
|
1100
|
-
if state.is_a?(Hash) && Array(state[:plan]).empty? && !request.to_s.strip.empty?
|
|
1101
|
-
kind = state[:request_kind] || request_kind(request: request)
|
|
1102
|
-
state[:request_kind] ||= kind
|
|
1103
|
-
plan(request: request, state: state, request_kind: kind) if needs_task_breakdown?(kind: kind)
|
|
1104
|
-
end
|
|
748
|
+
# Ensure a plan exists so about_to can name the active task.
|
|
749
|
+
plan(request: request, state: state) if state.is_a?(Hash) && Array(state[:plan]).empty? && !request.to_s.strip.empty?
|
|
1105
750
|
|
|
1106
751
|
caps = capabilities_for(names: names)
|
|
1107
752
|
counts = tool_counts_phrase(names: names)
|
|
@@ -1203,6 +848,8 @@ module PWN
|
|
|
1203
848
|
return nil if plan.empty?
|
|
1204
849
|
|
|
1205
850
|
idx = active_plan_index(state: state)
|
|
851
|
+
left = unfinished_tasks(state: state, messages: opts[:messages])
|
|
852
|
+
idx = left.first[:idx] if left.any? && left.none? { |task| task[:idx] == idx }
|
|
1206
853
|
item = plan[idx].to_s
|
|
1207
854
|
return nil if item.empty?
|
|
1208
855
|
|
|
@@ -1217,6 +864,49 @@ module PWN
|
|
|
1217
864
|
nil
|
|
1218
865
|
end
|
|
1219
866
|
|
|
867
|
+
# Remaining English work units that lack tool-result evidence.
|
|
868
|
+
# Discover/map items need some tool evidence; implement/fix needs a
|
|
869
|
+
# mutation signal; verify needs a spec/lint pass. Tool-count success
|
|
870
|
+
# JSON is not enough — that was the premature-final / skipped-task bug.
|
|
871
|
+
#
|
|
872
|
+
# Supported Method Parameters::
|
|
873
|
+
# left = PWN::AI::Agent::TaskSummarizer.unfinished_tasks(
|
|
874
|
+
# state: 'required - fresh() hash',
|
|
875
|
+
# messages: 'optional - Loop message array for extra coverage'
|
|
876
|
+
# )
|
|
877
|
+
# => [{ idx:, item:, label: }, ...]
|
|
878
|
+
public_class_method def self.unfinished_tasks(opts = {})
|
|
879
|
+
state = opts[:state]
|
|
880
|
+
return [] unless state.is_a?(Hash)
|
|
881
|
+
|
|
882
|
+
plan = Array(state[:plan]).map { |t| t.to_s.strip }
|
|
883
|
+
return [] if plan.length < 2
|
|
884
|
+
|
|
885
|
+
blob = coverage_blob(state: state, messages: opts[:messages])
|
|
886
|
+
n = plan.length
|
|
887
|
+
plan.each_with_index.filter_map do |item, i|
|
|
888
|
+
next if item.empty? || item_covered?(item: item, blob: blob)
|
|
889
|
+
|
|
890
|
+
{ idx: i, item: item, label: "task #{i + 1}/#{n}: #{item}" }
|
|
891
|
+
end
|
|
892
|
+
rescue StandardError
|
|
893
|
+
[]
|
|
894
|
+
end
|
|
895
|
+
|
|
896
|
+
# True while a multi-step English plan still has uncovered work.
|
|
897
|
+
# Loop uses this to refuse a text-only final.
|
|
898
|
+
#
|
|
899
|
+
# Supported Method Parameters::
|
|
900
|
+
# open = PWN::AI::Agent::TaskSummarizer.plan_open?(
|
|
901
|
+
# state: 'required - fresh() hash',
|
|
902
|
+
# messages: 'optional - Loop message array'
|
|
903
|
+
# )
|
|
904
|
+
public_class_method def self.plan_open?(opts = {})
|
|
905
|
+
unfinished_tasks(opts).any?
|
|
906
|
+
rescue StandardError
|
|
907
|
+
false
|
|
908
|
+
end
|
|
909
|
+
|
|
1220
910
|
# Short block for engine messages: full plan + focus on active English task.
|
|
1221
911
|
# Primary steering surface so tools follow generated tasks, not only TUI.
|
|
1222
912
|
#
|
|
@@ -1234,9 +924,11 @@ module PWN
|
|
|
1234
924
|
info = active_task(state: state)
|
|
1235
925
|
n = plan.length
|
|
1236
926
|
lines = []
|
|
1237
|
-
lines << '[pwn-ai/tasks] English tangible tasks are the
|
|
1238
|
-
lines << '
|
|
1239
|
-
lines << '
|
|
927
|
+
lines << '[pwn-ai/tasks] English tangible tasks are an advisory compass — the original request is the completion signal.'
|
|
928
|
+
lines << 'Prefer CORE_TOOLS (shell, pwn_eval, memory, mistakes, learning) until that request is done or blocked.'
|
|
929
|
+
lines << 'The task list is a breakdown, not a gate. Do not skip useful work; do not grind a covered item.'
|
|
930
|
+
req_line = immutable_request_line(opts)
|
|
931
|
+
lines << req_line if req_line
|
|
1240
932
|
lines << 'Original goal stays in context; the English tasks below are the work breakdown.'
|
|
1241
933
|
lines << "Active: #{info[:label]}" if info
|
|
1242
934
|
lines << "Tangible tasks (#{n}):"
|
|
@@ -1285,8 +977,9 @@ module PWN
|
|
|
1285
977
|
public_class_method def self.active_task_prompt(opts = {})
|
|
1286
978
|
state = opts[:state]
|
|
1287
979
|
return nil unless state.is_a?(Hash)
|
|
980
|
+
return nil unless plan_open?(state: state, messages: opts[:messages])
|
|
1288
981
|
|
|
1289
|
-
info = active_task(state: state)
|
|
982
|
+
info = active_task(state: state, messages: opts[:messages])
|
|
1290
983
|
return nil unless info
|
|
1291
984
|
|
|
1292
985
|
force = !opts[:force].nil?
|
|
@@ -1296,7 +989,7 @@ module PWN
|
|
|
1296
989
|
state[:focus_injected_idx] = info[:idx]
|
|
1297
990
|
# First injection after plan: full plan_context. Later: compact focus.
|
|
1298
991
|
if prev.nil? || force
|
|
1299
|
-
plan_context(state: state)
|
|
992
|
+
plan_context(state: state, request: opts[:request])
|
|
1300
993
|
else
|
|
1301
994
|
done_bit =
|
|
1302
995
|
if state[:last_advanced_from]
|
|
@@ -1306,14 +999,86 @@ module PWN
|
|
|
1306
999
|
else
|
|
1307
1000
|
''
|
|
1308
1001
|
end
|
|
1309
|
-
"#{done_bit}[pwn-ai/tasks]
|
|
1310
|
-
|
|
1311
|
-
|
|
1002
|
+
focus = "#{done_bit}[pwn-ai/tasks] Compass: #{info[:label]}. " \
|
|
1003
|
+
'Finish the original request with CORE_TOOLS; this task list is advisory.'
|
|
1004
|
+
req_line = immutable_request_line(opts)
|
|
1005
|
+
req_line ? "#{focus} #{req_line}" : focus
|
|
1312
1006
|
end
|
|
1313
1007
|
rescue StandardError
|
|
1314
1008
|
nil
|
|
1315
1009
|
end
|
|
1316
1010
|
|
|
1011
|
+
# Pull the operator's original ask out of curriculum / critic / GOAL+PLAN
|
|
1012
|
+
# wrappers so planning and model-facing prompts never treat a PLAN:
|
|
1013
|
+
# tool-call scaffold as the user request.
|
|
1014
|
+
#
|
|
1015
|
+
# Supported Method Parameters::
|
|
1016
|
+
# text = PWN::AI::Agent::TaskSummarizer.canonical_request(
|
|
1017
|
+
# request: 'required - raw user or wrapper string'
|
|
1018
|
+
# )
|
|
1019
|
+
public_class_method def self.canonical_request(opts = {})
|
|
1020
|
+
text = opts[:request].to_s
|
|
1021
|
+
return '' if text.strip.empty?
|
|
1022
|
+
|
|
1023
|
+
body = text.sub(/\A\s*REQUEST:\s*/i, '')
|
|
1024
|
+
if (m = body.match(/\AGOAL:\s*(.+?)(?=(?:\n\s*|\s+)(?:PLAN|ANSWER|FLAW|PATCH)\s*:|\z)/mi))
|
|
1025
|
+
extracted = squeeze_request_ws(text: m[1])
|
|
1026
|
+
return extracted unless extracted.empty?
|
|
1027
|
+
end
|
|
1028
|
+
if (idx = body =~ /\n\s*PLAN\s*:\s*(?:\n|\z)/i)
|
|
1029
|
+
head = squeeze_request_ws(text: body[0...idx])
|
|
1030
|
+
return head unless head.empty?
|
|
1031
|
+
end
|
|
1032
|
+
if (m = body.match(/\A(.+?)\s+PLAN\s*:\s*\d+[.):]/i))
|
|
1033
|
+
head = squeeze_request_ws(text: m[1])
|
|
1034
|
+
return head unless head.empty?
|
|
1035
|
+
end
|
|
1036
|
+
|
|
1037
|
+
squeeze_request_ws(text: body)
|
|
1038
|
+
rescue StandardError
|
|
1039
|
+
opts[:request].to_s.gsub(/\s+/, ' ').strip
|
|
1040
|
+
end
|
|
1041
|
+
|
|
1042
|
+
private_class_method def self.squeeze_request_ws(opts = {})
|
|
1043
|
+
opts[:text].to_s.gsub(/\s+/, ' ').strip
|
|
1044
|
+
end
|
|
1045
|
+
|
|
1046
|
+
private_class_method def self.plan_scaffold_item?(opts = {})
|
|
1047
|
+
s = opts[:item].to_s.gsub(/\s+/, ' ').strip
|
|
1048
|
+
return true if s.empty?
|
|
1049
|
+
return true if s.match?(/\A(?:GOAL|PLAN|REQUEST|ANSWER|FLAW|PATCH)\s*:/i)
|
|
1050
|
+
return true if s.match?(/\A\w+\s+command\s*=/i)
|
|
1051
|
+
|
|
1052
|
+
false
|
|
1053
|
+
rescue StandardError
|
|
1054
|
+
false
|
|
1055
|
+
end
|
|
1056
|
+
|
|
1057
|
+
private_class_method def self.reject_scaffold_tasks(opts = {})
|
|
1058
|
+
Array(opts[:tasks]).map { |t| t.to_s.gsub(/\s+/, ' ').strip }.reject(&:empty?).reject do |item|
|
|
1059
|
+
plan_scaffold_item?(item: item) || tool_jargon_task?(item: item)
|
|
1060
|
+
end
|
|
1061
|
+
rescue StandardError
|
|
1062
|
+
[]
|
|
1063
|
+
end
|
|
1064
|
+
|
|
1065
|
+
# Resolve the original user request for model-facing task prompts.
|
|
1066
|
+
# Prefer an explicit opts[:request] override, else the pinned
|
|
1067
|
+
# state[:original_request], else state[:request]. Wrappers are stripped.
|
|
1068
|
+
private_class_method def self.immutable_request_line(opts = {})
|
|
1069
|
+
state = opts[:state]
|
|
1070
|
+
request = opts[:request].to_s.strip
|
|
1071
|
+
request = canonical_request(request: request) unless request.empty?
|
|
1072
|
+
if request.empty? && state.is_a?(Hash)
|
|
1073
|
+
request = state[:original_request].to_s.strip
|
|
1074
|
+
request = state[:request].to_s.strip if request.empty?
|
|
1075
|
+
request = canonical_request(request: request) unless request.empty?
|
|
1076
|
+
end
|
|
1077
|
+
return nil if request.empty?
|
|
1078
|
+
|
|
1079
|
+
"Original request (immutable): #{request}"
|
|
1080
|
+
end
|
|
1081
|
+
|
|
1317
1082
|
private_class_method def self.active_plan_index(opts = {})
|
|
1318
1083
|
state = opts[:state]
|
|
1319
1084
|
plan = Array(state[:plan])
|
|
@@ -1481,7 +1246,9 @@ module PWN
|
|
|
1481
1246
|
end
|
|
1482
1247
|
|
|
1483
1248
|
# Advance or hold plan_idx from an R2 step batch.
|
|
1484
|
-
#
|
|
1249
|
+
# Advance ONLY when the current English task has completion evidence
|
|
1250
|
+
# (mutate/verify) or the batch clearly hands off to the NEXT task's
|
|
1251
|
+
# phase. A +1 search streak is telemetry — never task completion.
|
|
1485
1252
|
# Any -1 or mistake fingerprint on the batch -> do not advance.
|
|
1486
1253
|
#
|
|
1487
1254
|
# Supported Method Parameters::
|
|
@@ -1490,6 +1257,7 @@ module PWN
|
|
|
1490
1257
|
# rewards: 'required - Array of -1|0|1 (batch order)',
|
|
1491
1258
|
# intents: 'optional - Array of intent verb strings for the batch',
|
|
1492
1259
|
# names: 'optional - tool names in the batch',
|
|
1260
|
+
# result: 'optional - latest tool result string',
|
|
1493
1261
|
# mistake: 'optional - truthy when a mistake fingerprint hit this batch'
|
|
1494
1262
|
# )
|
|
1495
1263
|
public_class_method def self.apply_prm_advancement!(opts = {})
|
|
@@ -1513,86 +1281,208 @@ module PWN
|
|
|
1513
1281
|
end
|
|
1514
1282
|
|
|
1515
1283
|
pos = rewards.count(&:positive?)
|
|
1516
|
-
neu = rewards.count(&:zero?)
|
|
1517
|
-
# Require a clear +1 presence (not all-neutral).
|
|
1518
1284
|
if pos.zero?
|
|
1519
1285
|
state[:prm_pos_streak] = 0
|
|
1520
1286
|
state[:last_prm_signal] = :hold_neutral
|
|
1521
1287
|
return idx
|
|
1522
1288
|
end
|
|
1523
1289
|
|
|
1524
|
-
|
|
1290
|
+
# Streak is telemetry only — never an advance trigger.
|
|
1291
|
+
state[:prm_pos_streak] = state[:prm_pos_streak].to_i + pos
|
|
1292
|
+
|
|
1293
|
+
item = plan[idx].to_s
|
|
1525
1294
|
intents = Array(opts[:intents]).map { |iv| iv.to_s.downcase }.reject(&:empty?)
|
|
1526
1295
|
names = Array(opts[:names]).map(&:to_s)
|
|
1296
|
+
intent_s = (intents + names).join(' ')
|
|
1527
1297
|
|
|
1528
|
-
|
|
1529
|
-
|
|
1530
|
-
|
|
1531
|
-
|
|
1532
|
-
|
|
1533
|
-
|
|
1534
|
-
|
|
1535
|
-
|
|
1536
|
-
|
|
1537
|
-
unless matched
|
|
1538
|
-
state[:last_prm_signal] = :hold_intent_mismatch
|
|
1539
|
-
return idx
|
|
1298
|
+
if task_complete_enough?(
|
|
1299
|
+
item: item,
|
|
1300
|
+
result: opts[:result],
|
|
1301
|
+
names: names,
|
|
1302
|
+
intents: intents,
|
|
1303
|
+
state: state
|
|
1304
|
+
)
|
|
1305
|
+
return bump_plan!(state: state, idx: idx, signal: :advance_complete)
|
|
1540
1306
|
end
|
|
1541
1307
|
|
|
1542
|
-
|
|
1543
|
-
state
|
|
1544
|
-
# Advance after a streak of >=2 positive steps (or a single full +1 batch of size>=2).
|
|
1545
|
-
should = streak >= 2 || (pos >= 2 && neu.zero?)
|
|
1546
|
-
if should
|
|
1547
|
-
state[:last_advanced_from] = idx
|
|
1548
|
-
state[:plan_idx] = idx + 1
|
|
1549
|
-
state[:prm_pos_streak] = 0
|
|
1550
|
-
state[:tools_on_task] = 0
|
|
1551
|
-
state[:last_prm_signal] = :advance
|
|
1552
|
-
return idx + 1
|
|
1553
|
-
end
|
|
1308
|
+
nxt = plan[idx + 1].to_s
|
|
1309
|
+
return bump_plan!(state: state, idx: idx, signal: :advance_handoff) if handoff_to_next?(state: state, item: item, next_item: nxt, intent: intent_s)
|
|
1554
1310
|
|
|
1555
|
-
state[:last_prm_signal] = :
|
|
1311
|
+
state[:last_prm_signal] = :hold_open
|
|
1556
1312
|
idx
|
|
1557
1313
|
rescue StandardError
|
|
1558
1314
|
opts[:state].is_a?(Hash) ? opts[:state][:plan_idx].to_i : 0
|
|
1559
1315
|
end
|
|
1560
1316
|
|
|
1561
|
-
|
|
1317
|
+
MUTATION_DONE_RX = /
|
|
1318
|
+
patched|wrote\s|file\.write|fileutils|sed\s+-i|binwrite|
|
|
1319
|
+
syntax\sok|changed\s+\d+\s+lines
|
|
1320
|
+
/ix
|
|
1321
|
+
VERIFY_DONE_RX = /
|
|
1322
|
+
0\s+offenses|0\s+failures|examples?,\s*0|
|
|
1323
|
+
all\s+examples?\s+passed|\d+\s+runs?,\s*0\s+failures
|
|
1324
|
+
/ix
|
|
1325
|
+
# Ran the verifier — green or red. "4 offenses" still completes the
|
|
1326
|
+
# verify English task; remaining defects belong to implement/fix.
|
|
1327
|
+
VERIFY_RAN_RX = /
|
|
1328
|
+
\d+\s+offenses?|\d+\s+failures|examples?,\s*\d+|
|
|
1329
|
+
finished\s+in\s+\d|\d+\s+runs?,\s*\d+\s+failures
|
|
1330
|
+
/ix
|
|
1331
|
+
HANDOFF_MIN_TOOLS = 2
|
|
1332
|
+
DISCOVER_MIN_TOOLS = 3
|
|
1333
|
+
|
|
1334
|
+
private_class_method def self.task_phase(opts = {})
|
|
1335
|
+
s = opts[:item].to_s.downcase
|
|
1336
|
+
# Strict test/lint verify first — stems must not sit inside \b...\b
|
|
1337
|
+
# ("\bverif\b" never matches "verify"; that was the stuck-plan bug).
|
|
1338
|
+
return :verify if s.match?(
|
|
1339
|
+
/\b(rspec|rubocop|rake|lint|end-to-end)\b|run spec|run test|confirm the final/
|
|
1340
|
+
)
|
|
1341
|
+
return :present if s.match?(/\bpresent\b|\breport completion\b|\breport results\b/)
|
|
1342
|
+
# Soft "verify the result and report" is a closer, not a test runner.
|
|
1343
|
+
return :present if s.match?(/\bverif\w*\b/) && !s.match?(/\b(rspec|rubocop|rake|lint|spec|test)\b/)
|
|
1344
|
+
return :mutate if s.match?(
|
|
1345
|
+
/\b(implement\w*|fix|patch\w*|chang\w*|improv\w*|write|apply|wire|refactor\w*)\b/
|
|
1346
|
+
)
|
|
1347
|
+
return :discover if s.match?(
|
|
1348
|
+
/\b(locat\w*|find|read|inspect|recon\w*|understand|decompos\w*|map|identif\w*|gather|discover|enumerat\w*|scan|probe|determin\w*|root cause|where and why|track)\b/
|
|
1349
|
+
)
|
|
1350
|
+
|
|
1351
|
+
:generic
|
|
1352
|
+
rescue StandardError
|
|
1353
|
+
:generic
|
|
1354
|
+
end
|
|
1355
|
+
|
|
1356
|
+
private_class_method def self.intent_phase(opts = {})
|
|
1357
|
+
s = opts[:intent].to_s.downcase
|
|
1358
|
+
return :verify if s.match?(/test|rubocop|rake|rspec|offenses|failures/)
|
|
1359
|
+
return :mutate if s.match?(/edit|mutate|write|patch|sed\s+-i|file.write|refactor/)
|
|
1360
|
+
return :discover if s.match?(/search|read|recon|extro|sessions|memory|find|list|scan|inspect|locat/)
|
|
1361
|
+
|
|
1362
|
+
:generic
|
|
1363
|
+
rescue StandardError
|
|
1364
|
+
:generic
|
|
1365
|
+
end
|
|
1366
|
+
|
|
1367
|
+
# Exclusive phase match. A search tool must not "match" an implement
|
|
1368
|
+
# task, and "plans" in an English item must not count as discovery.
|
|
1369
|
+
# :present closers accept any non-empty intent (the work already ran).
|
|
1562
1370
|
private_class_method def self.task_intent_match?(opts = {})
|
|
1563
1371
|
item = opts[:item].to_s.downcase
|
|
1564
1372
|
intent = opts[:intent].to_s.downcase
|
|
1565
|
-
return
|
|
1373
|
+
return false if item.empty? || intent.empty?
|
|
1566
1374
|
|
|
1567
|
-
|
|
1568
|
-
|
|
1569
|
-
return true if stems.empty?
|
|
1375
|
+
phase = task_phase(item: item)
|
|
1376
|
+
return true if phase == :present
|
|
1570
1377
|
|
|
1571
|
-
|
|
1572
|
-
return
|
|
1378
|
+
ip = intent_phase(intent: intent)
|
|
1379
|
+
return false if ip == :generic && phase != :generic
|
|
1573
1380
|
|
|
1574
|
-
|
|
1575
|
-
|
|
1576
|
-
|
|
1577
|
-
|
|
1578
|
-
runish = /run|script|eval|shell|pwn_eval|search|read|edit|test|mutate/
|
|
1381
|
+
phase == ip || phase == :generic
|
|
1382
|
+
rescue StandardError
|
|
1383
|
+
false
|
|
1384
|
+
end
|
|
1579
1385
|
|
|
1580
|
-
|
|
1581
|
-
|
|
1582
|
-
|
|
1386
|
+
private_class_method def self.coverage_blob(opts = {})
|
|
1387
|
+
state = opts[:state]
|
|
1388
|
+
parts = []
|
|
1389
|
+
parts << state[:evidence_blob].to_s if state.is_a?(Hash)
|
|
1390
|
+
Array(opts[:messages]).each do |msg|
|
|
1391
|
+
next unless msg.is_a?(Hash)
|
|
1392
|
+
|
|
1393
|
+
role = msg[:role].to_s
|
|
1394
|
+
if role == 'tool'
|
|
1395
|
+
parts << msg[:name].to_s
|
|
1396
|
+
parts << msg[:content].to_s[0, 2_000]
|
|
1397
|
+
elsif role == 'assistant'
|
|
1398
|
+
Array(msg[:tool_calls]).each do |tc|
|
|
1399
|
+
parts << tc.dig(:function, :name).to_s
|
|
1400
|
+
parts << tc.dig(:function, :arguments).to_s[0, 1_000]
|
|
1401
|
+
end
|
|
1402
|
+
end
|
|
1403
|
+
end
|
|
1404
|
+
parts.join("\n").downcase
|
|
1405
|
+
rescue StandardError
|
|
1406
|
+
''
|
|
1407
|
+
end
|
|
1408
|
+
|
|
1409
|
+
private_class_method def self.item_covered?(opts = {})
|
|
1410
|
+
item = opts[:item].to_s
|
|
1411
|
+
blob = opts[:blob].to_s.downcase
|
|
1412
|
+
phase = task_phase(item: item)
|
|
1413
|
+
case phase
|
|
1414
|
+
when :mutate
|
|
1415
|
+
blob.match?(MUTATION_DONE_RX)
|
|
1416
|
+
when :verify
|
|
1417
|
+
blob.match?(VERIFY_DONE_RX) || blob.match?(VERIFY_RAN_RX)
|
|
1418
|
+
else
|
|
1419
|
+
# discover / present / generic: some real tool evidence, not empty / tiny JSON
|
|
1420
|
+
blob.strip.length >= 40
|
|
1421
|
+
end
|
|
1422
|
+
rescue StandardError
|
|
1423
|
+
false
|
|
1424
|
+
end
|
|
1583
1425
|
|
|
1584
|
-
|
|
1585
|
-
|
|
1586
|
-
|
|
1426
|
+
private_class_method def self.task_complete_enough?(opts = {})
|
|
1427
|
+
phase = task_phase(item: opts[:item])
|
|
1428
|
+
blob = [
|
|
1429
|
+
opts[:result],
|
|
1430
|
+
Array(opts[:names]).join(' '),
|
|
1431
|
+
Array(opts[:intents]).join(' ')
|
|
1432
|
+
]
|
|
1433
|
+
blob << opts[:state][:evidence_blob] if opts[:state].is_a?(Hash)
|
|
1434
|
+
joined = blob.join(' ')
|
|
1435
|
+
|
|
1436
|
+
case phase
|
|
1437
|
+
when :mutate, :verify
|
|
1438
|
+
item_covered?(item: opts[:item], blob: joined)
|
|
1439
|
+
when :discover, :generic, :present
|
|
1440
|
+
on_task = opts[:state].is_a?(Hash) ? opts[:state][:tools_on_task].to_i : 0
|
|
1441
|
+
return false if on_task < DISCOVER_MIN_TOOLS
|
|
1442
|
+
return false unless item_covered?(item: opts[:item], blob: joined)
|
|
1443
|
+
|
|
1444
|
+
intent_s = (Array(opts[:intents]) + Array(opts[:names])).join(' ')
|
|
1445
|
+
task_intent_match?(item: opts[:item], intent: intent_s) || phase == :present
|
|
1446
|
+
else
|
|
1447
|
+
false
|
|
1448
|
+
end
|
|
1587
1449
|
rescue StandardError
|
|
1450
|
+
false
|
|
1451
|
+
end
|
|
1452
|
+
|
|
1453
|
+
private_class_method def self.handoff_to_next?(opts = {})
|
|
1454
|
+
nxt = opts[:next_item].to_s
|
|
1455
|
+
return false if nxt.strip.empty?
|
|
1456
|
+
|
|
1457
|
+
on_task = opts[:state].is_a?(Hash) ? opts[:state][:tools_on_task].to_i : 0
|
|
1458
|
+
return false if on_task < HANDOFF_MIN_TOOLS
|
|
1459
|
+
|
|
1460
|
+
cur_p = task_phase(item: opts[:item])
|
|
1461
|
+
nxt_p = task_phase(item: nxt)
|
|
1462
|
+
return false if nxt_p == :generic
|
|
1463
|
+
return false if cur_p == nxt_p
|
|
1464
|
+
return false unless task_intent_match?(item: nxt, intent: opts[:intent]) || nxt_p == :present
|
|
1465
|
+
|
|
1588
1466
|
true
|
|
1467
|
+
rescue StandardError
|
|
1468
|
+
false
|
|
1589
1469
|
end
|
|
1590
1470
|
|
|
1591
|
-
|
|
1471
|
+
private_class_method def self.bump_plan!(opts = {})
|
|
1472
|
+
state = opts[:state]
|
|
1473
|
+
idx = opts[:idx].to_i
|
|
1474
|
+
state[:last_advanced_from] = idx
|
|
1475
|
+
state[:plan_idx] = idx + 1
|
|
1476
|
+
state[:prm_pos_streak] = 0
|
|
1477
|
+
state[:tools_on_task] = 0
|
|
1478
|
+
state[:last_prm_signal] = opts[:signal] || :advance
|
|
1479
|
+
idx + 1
|
|
1480
|
+
end
|
|
1481
|
+
|
|
1482
|
+
# Nudge plan_idx forward only on a real next-task phase handoff.
|
|
1483
|
+
# Never advance because "enough tools landed" — that skipped English work.
|
|
1592
1484
|
private_class_method def self.maybe_advance_plan!(opts = {})
|
|
1593
1485
|
state = opts[:state]
|
|
1594
|
-
names = opts[:names]
|
|
1595
|
-
intent = opts[:intent].to_s
|
|
1596
1486
|
return unless state.is_a?(Hash)
|
|
1597
1487
|
|
|
1598
1488
|
plan = Array(state[:plan])
|
|
@@ -1601,37 +1491,16 @@ module PWN
|
|
|
1601
1491
|
idx = state[:plan_idx].to_i
|
|
1602
1492
|
return if idx >= plan.length - 1
|
|
1603
1493
|
|
|
1604
|
-
|
|
1605
|
-
|
|
1606
|
-
|
|
1607
|
-
|
|
1608
|
-
|
|
1609
|
-
|
|
1610
|
-
|
|
1611
|
-
|
|
1612
|
-
|
|
1613
|
-
|
|
1614
|
-
phase_shift =
|
|
1615
|
-
(item.match?(/locat|find|read|inspect|recon|understand|decompos|plan|determin|identif|gather|discover/) &&
|
|
1616
|
-
intent.match?(/edit|mutate|test|vcs|run|script|eval/)) ||
|
|
1617
|
-
(item.match?(/implement|fix|patch|chang|improv|display|wire|carry out|core work|probe|scan|enumerat/) &&
|
|
1618
|
-
intent.match?(/test|rubocop|rake|eval|mutate/)) ||
|
|
1619
|
-
(item.match?(/aggregat|collect|combin/) &&
|
|
1620
|
-
intent.match?(/eval|mutate|script|run/)) ||
|
|
1621
|
-
(item.match?(/json|yaml|table|format|convert|present|report|verif/) &&
|
|
1622
|
-
intent.match?(/eval|mutate|script|test|run/)) ||
|
|
1623
|
-
(name_s.match?(/extro_/) && item.match?(/code|source|implement|rubocop/)) ||
|
|
1624
|
-
(name_s.match?(/memory_|sessions_/) && item.match?(/scan|probe|discover|recon/))
|
|
1625
|
-
end
|
|
1626
|
-
# Advance when phase clearly moves after >=1 tool on this task, OR
|
|
1627
|
-
# after >=3 successful-ish tools still on this task (completion quota).
|
|
1628
|
-
should = (phase_shift && on_task >= 2) || (on_task >= 3 && matched)
|
|
1629
|
-
if should
|
|
1630
|
-
state[:last_advanced_from] = idx
|
|
1631
|
-
state[:plan_idx] = idx + 1
|
|
1632
|
-
state[:tools_on_task] = 0
|
|
1633
|
-
state[:last_prm_signal] = :advance_phase if state[:last_prm_signal] != :advance
|
|
1634
|
-
end
|
|
1494
|
+
names = Array(opts[:names])
|
|
1495
|
+
intent_s = "#{opts[:intent]} #{names.join(' ')}"
|
|
1496
|
+
return unless handoff_to_next?(
|
|
1497
|
+
state: state,
|
|
1498
|
+
item: plan[idx].to_s,
|
|
1499
|
+
next_item: plan[idx + 1].to_s,
|
|
1500
|
+
intent: intent_s
|
|
1501
|
+
)
|
|
1502
|
+
|
|
1503
|
+
bump_plan!(state: state, idx: idx, signal: :advance_phase)
|
|
1635
1504
|
rescue StandardError
|
|
1636
1505
|
nil
|
|
1637
1506
|
end
|
|
@@ -1741,6 +1610,9 @@ module PWN
|
|
|
1741
1610
|
state[:since_emit] += 1
|
|
1742
1611
|
state[:tools_on_task] = state[:tools_on_task].to_i + 1
|
|
1743
1612
|
state[:emitted_for_batch] = false
|
|
1613
|
+
chunk = "#{name} #{preview} #{rs.to_s[0, 800]}"
|
|
1614
|
+
state[:evidence_blob] = "#{state[:evidence_blob]} #{chunk}"
|
|
1615
|
+
state[:evidence_blob] = state[:evidence_blob][-16_000..] if state[:evidence_blob].to_s.length > 20_000
|
|
1744
1616
|
intent = intent_phrase(tools: [{ name: name.to_s, args: args }])
|
|
1745
1617
|
# Live R2-local signal from tool outcome (executive idx only).
|
|
1746
1618
|
# Full ORM/PRM credit stays in Reward during auto_introspect.
|
|
@@ -1756,10 +1628,13 @@ module PWN
|
|
|
1756
1628
|
rewards: [local_r],
|
|
1757
1629
|
intents: [intent],
|
|
1758
1630
|
names: [name.to_s],
|
|
1631
|
+
result: rs,
|
|
1759
1632
|
mistake: mistake_hit
|
|
1760
1633
|
)
|
|
1761
1634
|
# Heuristic phase-shift remains as a backstop when PRM streak has not fired.
|
|
1762
|
-
if state[:last_prm_signal]
|
|
1635
|
+
if state[:last_prm_signal].to_s.start_with?('advance')
|
|
1636
|
+
# already moved this record
|
|
1637
|
+
else
|
|
1763
1638
|
maybe_advance_plan!(
|
|
1764
1639
|
state: state,
|
|
1765
1640
|
names: [name.to_s],
|
|
@@ -1853,11 +1728,7 @@ module PWN
|
|
|
1853
1728
|
puts <<~USAGE
|
|
1854
1729
|
USAGE:
|
|
1855
1730
|
state = PWN::AI::Agent::TaskSummarizer.fresh(request: 'ship task briefs to execs')
|
|
1856
|
-
# On user submit — autonomous goals only: LLM breaks into tangible tasks:
|
|
1857
|
-
kind = PWN::AI::Agent::TaskSummarizer.request_kind(request: state[:request])
|
|
1858
|
-
# => :statement | :question | :autonomous_goal
|
|
1859
1731
|
plan = PWN::AI::Agent::TaskSummarizer.plan(request: state[:request], state: state)
|
|
1860
|
-
# statements/questions => []; autonomous goals => ordered work units
|
|
1861
1732
|
text = PWN::AI::Agent::TaskSummarizer.emit_plan!(state: state)
|
|
1862
1733
|
# → "Goal: ...\nTangible tasks (N) — each may use many tools:\n task 1/N: ...\n task 2/N: ..."
|
|
1863
1734
|
# UI: on_tool.call('task', text, '') # full text, no truncation
|