pwn 0.5.680 → 0.5.682
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.gitignore +1 -0
- data/documentation/AI-Integration.md +1 -1
- data/documentation/Agent-Tool-Registry.md +1 -1
- data/documentation/Configuration.md +3 -6
- data/documentation/How-PWN-Works.md +2 -2
- data/documentation/Reinforcement-Learning.md +1 -1
- data/documentation/diagrams/dot/task-summarizer.dot +3 -3
- data/documentation/pwn-ai-Agent.md +16 -35
- data/lib/pwn/ai/agent/loop.rb +230 -191
- data/lib/pwn/ai/agent/mistakes.rb +17 -0
- data/lib/pwn/ai/agent/policy.rb +55 -5
- data/lib/pwn/ai/agent/prompt_builder.rb +38 -12
- data/lib/pwn/ai/agent/registry.rb +17 -9
- data/lib/pwn/ai/agent/task_summarizer.rb +374 -503
- data/lib/pwn/config.rb +1 -1
- data/lib/pwn/version.rb +1 -1
- data/spec/integration/prompt_builder_spec.rb +6 -4
- data/spec/lib/pwn/ai/agent/loop_spec.rb +251 -33
- data/spec/lib/pwn/ai/agent/mistakes_spec.rb +14 -0
- data/spec/lib/pwn/ai/agent/policy_spec.rb +52 -1
- data/spec/lib/pwn/ai/agent/prompt_builder_spec.rb +4 -9
- data/spec/lib/pwn/ai/agent/registry_spec.rb +22 -2
- data/spec/lib/pwn/ai/agent/signal_hygiene_spec.rb +4 -5
- data/spec/lib/pwn/ai/agent/task_summarizer_spec.rb +321 -90
- data/third_party/pwn_rdoc.jsonl +24 -9
- metadata +1 -1
|
@@ -51,6 +51,8 @@ describe PWN::AI::Agent::TaskSummarizer do
|
|
|
51
51
|
expect(described_class).to respond_to :active_task_prompt
|
|
52
52
|
expect(described_class).to respond_to :relevance_query
|
|
53
53
|
expect(described_class).to respond_to :tool_jargon_task?
|
|
54
|
+
expect(described_class).to respond_to :plan_open?
|
|
55
|
+
expect(described_class).to respond_to :unfinished_tasks
|
|
54
56
|
end
|
|
55
57
|
|
|
56
58
|
it 'format_plan method should exist' do
|
|
@@ -330,7 +332,7 @@ describe PWN::AI::Agent::TaskSummarizer do
|
|
|
330
332
|
expect(state[:plan_source]).to eq :llm
|
|
331
333
|
end
|
|
332
334
|
|
|
333
|
-
it 'about_to
|
|
335
|
+
it 'about_to keeps task k/n without restating the full goal every batch' do
|
|
334
336
|
expect(described_class).to receive(:chat_for_plan).with(request: subnet_req).and_return(llm_subnet_json)
|
|
335
337
|
state = described_class.fresh(request: subnet_req)
|
|
336
338
|
plan_text = described_class.emit_plan!(state: state)
|
|
@@ -346,13 +348,15 @@ describe PWN::AI::Agent::TaskSummarizer do
|
|
|
346
348
|
expect(first).to match(/via |shell/i)
|
|
347
349
|
expect(first).not_to include(subnet_req)
|
|
348
350
|
|
|
349
|
-
#
|
|
351
|
+
# Search-only tools must NOT consume the English plan. A later batch
|
|
352
|
+
# with a different tool still restates task k/n, never the full goal.
|
|
350
353
|
3.times do |i|
|
|
351
354
|
described_class.record!(state: state, name: 'shell', args: "probe #{i}", result: '{success:true}')
|
|
352
355
|
end
|
|
356
|
+
expect(state[:plan_idx]).to eq 0
|
|
353
357
|
|
|
354
358
|
second = described_class.about_to(
|
|
355
|
-
tools: [{ name: '
|
|
359
|
+
tools: [{ name: 'pwn_eval', args: { 'code' => '1+1' } }],
|
|
356
360
|
state: state
|
|
357
361
|
)
|
|
358
362
|
expect(second).to be_a(String)
|
|
@@ -360,17 +364,13 @@ describe PWN::AI::Agent::TaskSummarizer do
|
|
|
360
364
|
expect(second).not_to include(subnet_req)
|
|
361
365
|
end
|
|
362
366
|
|
|
363
|
-
it 'how-to questions get no
|
|
367
|
+
it 'how-to questions still get a task compass — there is no request type' do
|
|
364
368
|
allow(PWN::Env).to receive(:dig).and_call_original
|
|
365
369
|
allow(PWN::Env).to receive(:dig).with(:ai, :agent, :task_summary_llm).and_return(false)
|
|
366
370
|
req = 'how to do a ping sweep of a subnet using hping3?'
|
|
367
|
-
expect(described_class.
|
|
368
|
-
expect(described_class
|
|
369
|
-
|
|
370
|
-
expect(tasks).to eq([])
|
|
371
|
-
banner = described_class.format_plan(tasks: tasks, request: req)
|
|
372
|
-
expect(banner).to match(/Request type: question/i)
|
|
373
|
-
expect(banner).not_to match(/Tangible tasks \(\d+\)/)
|
|
371
|
+
expect(described_class.needs_task_breakdown?(request: req)).to eq(true)
|
|
372
|
+
expect(described_class).not_to respond_to :request_kind
|
|
373
|
+
expect(described_class.plan(request: 'what color is a cherry')).to eq([])
|
|
374
374
|
end
|
|
375
375
|
|
|
376
376
|
it 'falls back generically when LLM is disabled (no static domain scripts)' do
|
|
@@ -462,23 +462,26 @@ describe PWN::AI::Agent::TaskSummarizer do
|
|
|
462
462
|
expect(info[:item]).to include('locate')
|
|
463
463
|
|
|
464
464
|
ctx = described_class.plan_context(state: state)
|
|
465
|
-
expect(ctx).to include('
|
|
465
|
+
expect(ctx).to include('advisory compass')
|
|
466
|
+
expect(ctx).not_to include('SOLE driver')
|
|
466
467
|
expect(ctx).to include('▶ task 1/3:')
|
|
467
468
|
expect(ctx).to include('task 2/3:')
|
|
469
|
+
expect(ctx).to include("Original request (immutable): #{goal}")
|
|
468
470
|
|
|
469
471
|
first = described_class.active_task_prompt(state: state, force: true)
|
|
470
472
|
expect(first).to include('Active:')
|
|
471
473
|
expect(first).to include('task 1/3:')
|
|
474
|
+
expect(first).to include("Original request (immutable): #{goal}")
|
|
472
475
|
# second call same idx → nil (no spam)
|
|
473
476
|
expect(described_class.active_task_prompt(state: state)).to be_nil
|
|
474
477
|
|
|
475
478
|
state[:plan_idx] = 1
|
|
476
479
|
nxt = described_class.active_task_prompt(state: state)
|
|
477
|
-
expect(nxt).to match(%r{
|
|
480
|
+
expect(nxt).to match(%r{Compass: task 2/3:}i)
|
|
481
|
+
expect(nxt).to include("Original request (immutable): #{goal}")
|
|
478
482
|
end
|
|
479
483
|
|
|
480
|
-
it 'record!
|
|
481
|
-
# Drive PRM streak on locate task with search intents
|
|
484
|
+
it 'record! does not treat two successful searches as finishing the whole plan' do
|
|
482
485
|
line = nil
|
|
483
486
|
2.times do
|
|
484
487
|
line = described_class.record!(
|
|
@@ -488,7 +491,70 @@ describe PWN::AI::Agent::TaskSummarizer do
|
|
|
488
491
|
result: '{"success":true,"result":{"stdout":"hit","exit":0}}'
|
|
489
492
|
)
|
|
490
493
|
end
|
|
491
|
-
expect(state[:plan_idx]).to be
|
|
494
|
+
expect(state[:plan_idx]).to be < (state[:plan].length - 1)
|
|
495
|
+
expect(described_class.plan_open?(state: state)).to eq true
|
|
496
|
+
leftover = described_class.unfinished_tasks(state: state).map { |t| t[:idx] }
|
|
497
|
+
expect(leftover).to include(2)
|
|
498
|
+
end
|
|
499
|
+
|
|
500
|
+
it 'two ls/find calls do not skip to the last English task' do
|
|
501
|
+
described_class.active_task_prompt(state: state, force: true)
|
|
502
|
+
briefs = [
|
|
503
|
+
'ls /opt/pwn/lib/pwn/ai/agent',
|
|
504
|
+
'find /opt/pwn/lib -name task_summarizer.rb'
|
|
505
|
+
].map do |cmd|
|
|
506
|
+
described_class.record!(
|
|
507
|
+
state: state,
|
|
508
|
+
name: 'shell',
|
|
509
|
+
args: { 'command' => cmd },
|
|
510
|
+
result: '{"success":true,"result":{"stdout":"task_summarizer.rb","exit":0}}'
|
|
511
|
+
)
|
|
512
|
+
end
|
|
513
|
+
|
|
514
|
+
expect(state[:plan_idx]).to be < (state[:plan].length - 1)
|
|
515
|
+
expect(briefs.join).not_to match(/Completed task/i)
|
|
516
|
+
|
|
517
|
+
info = described_class.active_task(state: state)
|
|
518
|
+
leftover = described_class.unfinished_tasks(state: state).map { |t| t[:idx] }
|
|
519
|
+
expect(described_class.plan_open?(state: state)).to eq true
|
|
520
|
+
expect(leftover).to include(2)
|
|
521
|
+
# Covered locate must not keep focus on task 1 while later work is open.
|
|
522
|
+
expect(info[:idx]).to eq(leftover.min)
|
|
523
|
+
expect(info[:idx]).not_to eq(state[:plan].length - 1)
|
|
524
|
+
end
|
|
525
|
+
|
|
526
|
+
it 'active_task_prompt stays quiet once every English task is covered' do
|
|
527
|
+
st = described_class.fresh(request: 'what is my hostname?')
|
|
528
|
+
st[:plan] = [
|
|
529
|
+
'Determine the local hostname',
|
|
530
|
+
'Present the result and report completion'
|
|
531
|
+
]
|
|
532
|
+
st[:plan_idx] = 0
|
|
533
|
+
st[:plan_emitted] = true
|
|
534
|
+
described_class.record!(
|
|
535
|
+
state: st,
|
|
536
|
+
name: 'shell',
|
|
537
|
+
args: { 'command' => 'hostname' },
|
|
538
|
+
result: '{"success":true,"result":{"stdout":"kali-box","exit":0}}'
|
|
539
|
+
)
|
|
540
|
+
expect(described_class.plan_open?(state: st)).to eq false
|
|
541
|
+
expect(described_class.active_task_prompt(state: st, force: true)).to be_nil
|
|
542
|
+
end
|
|
543
|
+
|
|
544
|
+
it 'record! emits advancement brief only on real handoff to the next English task' do
|
|
545
|
+
described_class.record!(
|
|
546
|
+
state: state,
|
|
547
|
+
name: 'shell',
|
|
548
|
+
args: { 'command' => 'rg TaskSummarizer lib' },
|
|
549
|
+
result: '{"success":true,"result":{"stdout":"lib/pwn/ai/agent/task_summarizer.rb","exit":0}}'
|
|
550
|
+
)
|
|
551
|
+
line = described_class.record!(
|
|
552
|
+
state: state,
|
|
553
|
+
name: 'shell',
|
|
554
|
+
args: { 'command' => 'sed -i s/truncation/fixed/ lib/pwn/ai/agent/task_summarizer.rb' },
|
|
555
|
+
result: '{"success":true,"result":{"stdout":"patched task_summarizer.rb","exit":0}}'
|
|
556
|
+
)
|
|
557
|
+
expect(state[:plan_idx]).to eq 1
|
|
492
558
|
expect(line).to be_a(String)
|
|
493
559
|
expect(line).to match(%r{Advanced past task 1/3}i)
|
|
494
560
|
expect(line).to match(%r{now task 2/3:}i)
|
|
@@ -574,8 +640,7 @@ describe PWN::AI::Agent::TaskSummarizer do
|
|
|
574
640
|
expect(q).to include('run rspec to verify')
|
|
575
641
|
end
|
|
576
642
|
|
|
577
|
-
it 'apply_prm_advancement!
|
|
578
|
-
# first +1 alone is only a streak, not yet advance
|
|
643
|
+
it 'apply_prm_advancement! holds on +1 search streak and on -1; advances on next-task handoff' do
|
|
579
644
|
described_class.apply_prm_advancement!(
|
|
580
645
|
state: state,
|
|
581
646
|
rewards: [1],
|
|
@@ -583,7 +648,7 @@ describe PWN::AI::Agent::TaskSummarizer do
|
|
|
583
648
|
names: ['shell']
|
|
584
649
|
)
|
|
585
650
|
expect(state[:plan_idx]).to eq 0
|
|
586
|
-
expect(
|
|
651
|
+
expect(state[:last_prm_signal].to_s).to match(/hold|streak/)
|
|
587
652
|
|
|
588
653
|
described_class.apply_prm_advancement!(
|
|
589
654
|
state: state,
|
|
@@ -591,8 +656,19 @@ describe PWN::AI::Agent::TaskSummarizer do
|
|
|
591
656
|
intents: %w[search read],
|
|
592
657
|
names: ['shell']
|
|
593
658
|
)
|
|
659
|
+
# Two successful searches must NOT complete "locate the source".
|
|
660
|
+
expect(state[:plan_idx]).to eq 0
|
|
661
|
+
|
|
662
|
+
state[:tools_on_task] = 2
|
|
663
|
+
described_class.apply_prm_advancement!(
|
|
664
|
+
state: state,
|
|
665
|
+
rewards: [1],
|
|
666
|
+
intents: ['edit'],
|
|
667
|
+
names: ['shell'],
|
|
668
|
+
result: 'patched task_summarizer.rb'
|
|
669
|
+
)
|
|
594
670
|
expect(state[:plan_idx]).to eq 1
|
|
595
|
-
expect(state[:last_prm_signal]).to
|
|
671
|
+
expect(state[:last_prm_signal].to_s).to match(/advance/)
|
|
596
672
|
|
|
597
673
|
hold_idx = state[:plan_idx]
|
|
598
674
|
described_class.apply_prm_advancement!(
|
|
@@ -643,105 +719,260 @@ describe PWN::AI::Agent::TaskSummarizer do
|
|
|
643
719
|
end
|
|
644
720
|
end
|
|
645
721
|
|
|
646
|
-
describe '
|
|
722
|
+
describe 'task compass (no request type)' do
|
|
647
723
|
before do
|
|
648
724
|
allow(PWN::Env).to receive(:dig).and_call_original
|
|
649
725
|
allow(PWN::Env).to receive(:dig).with(:ai, :agent, :task_summary_llm).and_return(false)
|
|
650
|
-
allow(PWN::Env).to receive(:dig).with(:ai, :agent, :request_kind_llm).and_return(false)
|
|
651
726
|
end
|
|
652
727
|
|
|
653
|
-
it '
|
|
654
|
-
expect(described_class).
|
|
728
|
+
it 'has no request_kind classifier' do
|
|
729
|
+
expect(described_class).not_to respond_to :request_kind
|
|
730
|
+
expect(described_class).not_to respond_to :heuristic_request_kind
|
|
731
|
+
expect(described_class).not_to respond_to :parse_kind_label
|
|
732
|
+
expect(described_class).not_to respond_to :chat_for_kind
|
|
655
733
|
expect(described_class).to respond_to :needs_task_breakdown?
|
|
656
|
-
expect(described_class).to
|
|
657
|
-
expect(described_class).to respond_to :heuristic_request_kind
|
|
658
|
-
expect(described_class).to respond_to :llm_classify_kind
|
|
659
|
-
expect(described_class).to respond_to :chat_for_kind
|
|
660
|
-
expect(described_class).to respond_to :parse_kind_label
|
|
661
|
-
end
|
|
662
|
-
|
|
663
|
-
it 'classifies general statements without multi-step plans' do
|
|
664
|
-
req = 'FYI the staging build is green.'
|
|
665
|
-
expect(described_class.request_kind(request: req)).to eq(:statement)
|
|
666
|
-
expect(described_class.needs_task_breakdown?(request: req)).to eq(false)
|
|
667
|
-
st = described_class.fresh(request: req)
|
|
668
|
-
expect(st[:request_kind]).to eq(:statement)
|
|
669
|
-
expect(described_class.plan(request: req, state: st)).to eq([])
|
|
670
|
-
expect(st[:plan_source].to_s).to match(/no_breakdown/)
|
|
671
|
-
text = described_class.emit_plan!(state: st)
|
|
672
|
-
expect(text.to_s).to match(/Request type: statement/i)
|
|
673
|
-
expect(text.to_s).not_to match(%r{task 1/\d+:})
|
|
674
|
-
end
|
|
675
|
-
|
|
676
|
-
it 'classifies questions without multi-step plans' do
|
|
677
|
-
req = 'what is the default GQRX remote-control port?'
|
|
678
|
-
expect(described_class.request_kind(request: req)).to eq(:question)
|
|
679
|
-
expect(described_class.needs_task_breakdown?(kind: :question)).to eq(false)
|
|
680
|
-
expect(described_class.plan(request: req)).to eq([])
|
|
681
|
-
end
|
|
682
|
-
|
|
683
|
-
it 'classifies host-evidence questions as autonomous goals' do
|
|
684
|
-
req = 'what is my hostname?'
|
|
685
|
-
expect(described_class.request_kind(request: req)).to eq(:autonomous_goal)
|
|
686
|
-
expect(described_class.needs_task_breakdown?(request: req)).to eq(true)
|
|
687
|
-
expect(described_class.request_kind(request: 'excellent - what is my hostname?')).to eq(:autonomous_goal)
|
|
734
|
+
expect(described_class.needs_task_breakdown?(request: 'anything')).to eq(true)
|
|
688
735
|
end
|
|
689
736
|
|
|
690
|
-
it '
|
|
737
|
+
it 'fresh state has no request_kind and format_plan has no Request type line' do
|
|
691
738
|
req = 'refactor the authentication middleware and add unit tests'
|
|
692
|
-
|
|
693
|
-
expect(
|
|
739
|
+
st = described_class.fresh(request: req)
|
|
740
|
+
expect(st).not_to have_key(:request_kind)
|
|
694
741
|
steps = [
|
|
695
742
|
'locate the authentication middleware source',
|
|
696
743
|
'refactor the middleware for clarity and safety',
|
|
697
744
|
'add unit tests covering the new behavior',
|
|
698
745
|
'run the test suite and report completion'
|
|
699
746
|
]
|
|
700
|
-
st = described_class.fresh(request: req)
|
|
701
747
|
tasks = described_class.plan(request: req, state: st, tasks: steps)
|
|
702
748
|
expect(tasks.length).to be >= 3
|
|
703
|
-
|
|
704
|
-
text
|
|
705
|
-
expect(text).to match(/Request type: autonomous_goal/)
|
|
749
|
+
text = described_class.format_plan(tasks: tasks, request: req)
|
|
750
|
+
expect(text).not_to match(/Request type/)
|
|
706
751
|
expect(text).to match(/Tangible tasks/)
|
|
707
752
|
expect(text).to match(%r{task 1/\d+:})
|
|
708
|
-
# each work unit may use many tools — banner still says so
|
|
709
|
-
expect(text).to match(/one or more tools/i)
|
|
710
753
|
end
|
|
711
754
|
|
|
712
|
-
it '
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
755
|
+
it 'plan_open? stays true until mutate and verify English tasks have evidence' do
|
|
756
|
+
st = described_class.fresh(request: 'locate, fix, verify')
|
|
757
|
+
st[:plan] = [
|
|
758
|
+
'locate the TaskSummarizer source',
|
|
759
|
+
'fix the truncation bug',
|
|
760
|
+
'run rspec to verify'
|
|
761
|
+
]
|
|
762
|
+
st[:plan_idx] = 0
|
|
763
|
+
expect(described_class.plan_open?(state: st)).to eq true
|
|
764
|
+
|
|
765
|
+
described_class.record!(
|
|
766
|
+
state: st,
|
|
767
|
+
name: 'shell',
|
|
768
|
+
args: { 'command' => 'rg TaskSummarizer lib' },
|
|
769
|
+
result: '{"success":true,"result":{"stdout":"lib/pwn/ai/agent/task_summarizer.rb","exit":0}}'
|
|
770
|
+
)
|
|
771
|
+
expect(described_class.plan_open?(state: st)).to eq true
|
|
772
|
+
expect(described_class.unfinished_tasks(state: st).map { |t| t[:idx] }).to include(1, 2)
|
|
773
|
+
|
|
774
|
+
described_class.record!(
|
|
775
|
+
state: st,
|
|
776
|
+
name: 'shell',
|
|
777
|
+
args: { 'command' => 'sed -i s/bug/fix/ lib/pwn/ai/agent/task_summarizer.rb' },
|
|
778
|
+
result: '{"success":true,"result":{"stdout":"patched task_summarizer.rb","exit":0}}'
|
|
779
|
+
)
|
|
780
|
+
open_idx = described_class.unfinished_tasks(state: st).map { |t| t[:idx] }
|
|
781
|
+
expect(open_idx).to include(2)
|
|
782
|
+
expect(open_idx).not_to include(1)
|
|
783
|
+
expect(described_class.plan_open?(state: st)).to eq true
|
|
784
|
+
|
|
785
|
+
described_class.record!(
|
|
786
|
+
state: st,
|
|
787
|
+
name: 'shell',
|
|
788
|
+
args: { 'command' => 'bundle exec rspec spec/lib/pwn/ai/agent/task_summarizer_spec.rb' },
|
|
789
|
+
result: '{"success":true,"result":{"stdout":"12 examples, 0 failures","exit":0}}'
|
|
790
|
+
)
|
|
791
|
+
expect(described_class.plan_open?(state: st)).to eq false
|
|
792
|
+
expect(described_class.unfinished_tasks(state: st)).to eq([])
|
|
793
|
+
end
|
|
794
|
+
|
|
795
|
+
it 'task_phase classifies English stems (locate/verify/change), not only exact \b-words' do
|
|
796
|
+
expect(described_class.send(:task_phase, item: 'locate the TaskSummarizer source')).to eq(:discover)
|
|
797
|
+
expect(described_class.send(:task_phase, item: 'identify where tasks are skipped')).to eq(:discover)
|
|
798
|
+
expect(described_class.send(:task_phase, item: 'Determine the local hostname')).to eq(:discover)
|
|
799
|
+
expect(described_class.send(:task_phase, item: 'change the advancement logic')).to eq(:mutate)
|
|
800
|
+
expect(described_class.send(:task_phase, item: 'improve completion handling')).to eq(:mutate)
|
|
801
|
+
expect(described_class.send(:task_phase, item: 'Refactor Loop.run')).to eq(:mutate)
|
|
802
|
+
expect(described_class.send(:task_phase, item: 'Verify the result and report completion')).to eq(:present)
|
|
803
|
+
expect(described_class.send(:task_phase, item: 'Present the result and report completion')).to eq(:present)
|
|
804
|
+
expect(described_class.send(:task_phase, item: 'Verify full task completion end-to-end')).to eq(:verify)
|
|
805
|
+
expect(described_class.send(:task_phase, item: 'run rspec to verify')).to eq(:verify)
|
|
806
|
+
end
|
|
807
|
+
|
|
808
|
+
it 'host-evidence plans close after a live lookup (forced closer is not a test-runner verify)' do
|
|
809
|
+
st = described_class.fresh(request: 'what is my hostname?')
|
|
810
|
+
closer = described_class.send(
|
|
811
|
+
:normalize_task_list,
|
|
812
|
+
tasks: ['Determine the local hostname'],
|
|
813
|
+
goal: 'what is my hostname?'
|
|
814
|
+
)
|
|
815
|
+
expect(closer.join(' ')).not_to match(/\brspec\b|\brubocop\b/)
|
|
816
|
+
expect(described_class.send(:task_phase, item: closer.last)).not_to eq(:verify)
|
|
817
|
+
st[:plan] = closer
|
|
818
|
+
st[:plan_idx] = 0
|
|
819
|
+
described_class.record!(
|
|
820
|
+
state: st,
|
|
821
|
+
name: 'shell',
|
|
822
|
+
args: { 'command' => 'hostname' },
|
|
823
|
+
result: '{"success":true,"result":{"stdout":"kali-box","exit":0}}'
|
|
824
|
+
)
|
|
825
|
+
expect(described_class.plan_open?(state: st)).to eq false
|
|
826
|
+
expect(described_class.unfinished_tasks(state: st)).to eq([])
|
|
716
827
|
end
|
|
717
828
|
|
|
718
|
-
it '
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
829
|
+
it 'three on-task discover tools advance one English task and do not jump to the last' do
|
|
830
|
+
st = described_class.fresh(request: 'map then fix then verify')
|
|
831
|
+
st[:plan] = [
|
|
832
|
+
'locate the TaskSummarizer source',
|
|
833
|
+
'identify where planned tasks are skipped',
|
|
834
|
+
'Implement the completion fix',
|
|
835
|
+
'Verify full task completion end-to-end'
|
|
836
|
+
]
|
|
837
|
+
st[:plan_idx] = 0
|
|
838
|
+
3.times do |i|
|
|
839
|
+
described_class.record!(
|
|
840
|
+
state: st,
|
|
841
|
+
name: 'shell',
|
|
842
|
+
args: { 'command' => "rg TaskSummarizer lib #{i}" },
|
|
843
|
+
result: '{"success":true,"result":{"stdout":"lib/pwn/ai/agent/task_summarizer.rb","exit":0}}'
|
|
844
|
+
)
|
|
845
|
+
end
|
|
846
|
+
expect(st[:plan_idx]).to eq 1
|
|
847
|
+
expect(st[:plan_idx]).to be < (st[:plan].length - 1)
|
|
848
|
+
expect(described_class.plan_open?(state: st)).to eq true
|
|
726
849
|
end
|
|
727
850
|
|
|
728
|
-
it '
|
|
729
|
-
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
851
|
+
it 'a verify task is covered after the verifier ran, even with remaining offenses' do
|
|
852
|
+
st = described_class.fresh(request: 'run rubocop')
|
|
853
|
+
st[:plan] = [
|
|
854
|
+
'Refactor Loop.run',
|
|
855
|
+
'run rubocop to verify'
|
|
856
|
+
]
|
|
857
|
+
st[:plan_idx] = 1
|
|
858
|
+
described_class.record!(
|
|
859
|
+
state: st,
|
|
860
|
+
name: 'shell',
|
|
861
|
+
args: { 'command' => 'bundle exec rubocop lib/pwn/ai/agent/loop.rb' },
|
|
862
|
+
result: '{"success":true,"result":{"stdout":"4 offenses detected","exit":1}}'
|
|
863
|
+
)
|
|
864
|
+
leftover = described_class.unfinished_tasks(state: st).map { |t| t[:idx] }
|
|
865
|
+
expect(leftover).not_to include(1)
|
|
866
|
+
end
|
|
867
|
+
|
|
868
|
+
it 'five successful ls/find shells never skip a 5-step mapping plan to the last task' do
|
|
869
|
+
st = described_class.fresh(request: 'map then fix autonomous_goal completion')
|
|
870
|
+
st[:plan] = [
|
|
871
|
+
'Map how autonomous_goal plans, tracks, and completes work units',
|
|
872
|
+
'Identify where and why planned tasks are skipped',
|
|
873
|
+
'Determine the root cause of premature finalized responses',
|
|
874
|
+
'Implement fixes so every planned task is finished',
|
|
875
|
+
'Verify full task completion end-to-end'
|
|
876
|
+
]
|
|
877
|
+
st[:plan_idx] = 0
|
|
878
|
+
st[:plan_emitted] = true
|
|
879
|
+
5.times do |i|
|
|
880
|
+
described_class.record!(
|
|
881
|
+
state: st,
|
|
882
|
+
name: 'shell',
|
|
883
|
+
args: { 'command' => "ls /opt/pwn && find lib -name '*#{i}*'" },
|
|
884
|
+
result: '{"success":true,"result":{"stdout":"/opt/pwn","exit":0}}'
|
|
885
|
+
)
|
|
886
|
+
end
|
|
887
|
+
expect(st[:plan_idx]).to be < (st[:plan].length - 1)
|
|
888
|
+
expect(described_class.plan_open?(state: st)).to eq true
|
|
889
|
+
unfinished = described_class.unfinished_tasks(state: st).map { |t| t[:item] }
|
|
890
|
+
expect(unfinished.join(' ')).to match(/Implement|Verify/i)
|
|
736
891
|
end
|
|
737
892
|
|
|
738
|
-
it 'about_to
|
|
893
|
+
it 'about_to plans a question-shaped request with no request type' do
|
|
739
894
|
req = 'what flags does nmap use for a ping scan?'
|
|
740
895
|
st = described_class.fresh(request: req)
|
|
741
|
-
|
|
742
|
-
|
|
743
|
-
|
|
744
|
-
|
|
896
|
+
described_class.about_to(tools: [{ name: 'shell', args: { 'command' => 'man nmap' } }], state: st)
|
|
897
|
+
expect(st).not_to have_key(:request_kind)
|
|
898
|
+
end
|
|
899
|
+
end
|
|
900
|
+
|
|
901
|
+
describe 'original request visibility and PLAN: wrapper resistance' do
|
|
902
|
+
let(:original) do
|
|
903
|
+
'Fix TaskSummarizer so that it has visibility of the original request made by the user so tasks are relevant to the original request.'
|
|
904
|
+
end
|
|
905
|
+
let(:wrapped) do
|
|
906
|
+
<<~REQ
|
|
907
|
+
REQUEST:
|
|
908
|
+
GOAL: #{original}
|
|
909
|
+
|
|
910
|
+
PLAN:
|
|
911
|
+
1. shell command="rg -n --glob '!{vendor,tmp,pkg,.git,coverage}/**' -e 'TaskSummarizer|task_summarizer' /opt/pwn/lib /opt/pwn/spec /opt/pwn/bin"
|
|
912
|
+
2. shell command="sed -n '1,260p' /opt/pwn/lib/pwn/ai/agent/task_summarizer.rb"
|
|
913
|
+
3. shell command="rg -n -e original_request /opt/pwn/lib/pwn/ai"
|
|
914
|
+
REQ
|
|
915
|
+
end
|
|
916
|
+
|
|
917
|
+
it 'canonical_request strips GOAL/PLAN wrappers down to the operator ask' do
|
|
918
|
+
expect(described_class.canonical_request(request: wrapped)).to eq(original)
|
|
919
|
+
expect(described_class.canonical_request(request: original)).to eq(original)
|
|
920
|
+
collapsed = "GOAL: #{original} PLAN: 1. shell command=\"rg TaskSummarizer\" 2. shell command=\"sed -n 1,20p file\""
|
|
921
|
+
expect(described_class.canonical_request(request: collapsed)).to eq(original)
|
|
922
|
+
end
|
|
923
|
+
|
|
924
|
+
it 'fresh pins original_request to the canonical ask, not the PLAN: scaffold' do
|
|
925
|
+
st = described_class.fresh(request: wrapped)
|
|
926
|
+
expect(st[:original_request]).to eq(original)
|
|
927
|
+
expect(st[:request]).to eq(original)
|
|
928
|
+
expect(st[:request]).not_to match(/shell command=/i)
|
|
929
|
+
end
|
|
930
|
+
|
|
931
|
+
it 'plan does not treat PLAN: tool-call enumerations as English tasks' do
|
|
932
|
+
allow(described_class).to receive(:chat_for_plan).and_return(
|
|
933
|
+
[
|
|
934
|
+
'Locate the TaskSummarizer component and how it builds tasks',
|
|
935
|
+
'Identify where the original user request is available in the call chain',
|
|
936
|
+
'Update TaskSummarizer so it receives and uses the original user request',
|
|
937
|
+
'Verify generated tasks stay relevant to that original request'
|
|
938
|
+
].to_json
|
|
939
|
+
)
|
|
940
|
+
st = described_class.fresh(request: wrapped)
|
|
941
|
+
tasks = described_class.plan(request: wrapped, state: st)
|
|
942
|
+
expect(st[:plan_source]).not_to eq(:enumerated)
|
|
943
|
+
expect(tasks.join(' ')).not_to match(/shell command=/i)
|
|
944
|
+
expect(tasks.join(' ').downcase).to match(/original/)
|
|
945
|
+
expect(st[:original_request]).to eq(original)
|
|
946
|
+
end
|
|
947
|
+
|
|
948
|
+
it 'plan_context and compact focus restate the original request' do
|
|
949
|
+
st = described_class.fresh(request: wrapped)
|
|
950
|
+
st[:plan] = [
|
|
951
|
+
'Locate the TaskSummarizer component',
|
|
952
|
+
'Identify request visibility in the call chain',
|
|
953
|
+
'Update TaskSummarizer to use the original request'
|
|
954
|
+
]
|
|
955
|
+
st[:plan_idx] = 0
|
|
956
|
+
st[:plan_emitted] = true
|
|
957
|
+
|
|
958
|
+
ctx = described_class.plan_context(state: st)
|
|
959
|
+
expect(ctx).to include("Original request (immutable): #{original}")
|
|
960
|
+
expect(ctx).not_to include('shell command=')
|
|
961
|
+
|
|
962
|
+
first = described_class.active_task_prompt(state: st, force: true)
|
|
963
|
+
expect(first).to include("Original request (immutable): #{original}")
|
|
964
|
+
|
|
965
|
+
st[:plan_idx] = 1
|
|
966
|
+
nxt = described_class.active_task_prompt(state: st)
|
|
967
|
+
expect(nxt).to include("Original request (immutable): #{original}")
|
|
968
|
+
end
|
|
969
|
+
|
|
970
|
+
it 'still honors genuine English numbered steps from the user' do
|
|
971
|
+
req = 'Do three things. 1. locate the file 2. patch the truncation 3. add plan on submit'
|
|
972
|
+
tasks = described_class.plan(request: req)
|
|
973
|
+
expect(tasks.length).to be >= 3
|
|
974
|
+
expect(tasks[0].downcase).to include('locate')
|
|
975
|
+
expect(tasks[1].downcase).to match(/patch|truncat/)
|
|
745
976
|
end
|
|
746
977
|
end
|
|
747
978
|
end
|