pwn 0.5.679 → 0.5.682

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. checksums.yaml +4 -4
  2. data/.gitignore +1 -0
  3. data/documentation/AI-Integration.md +1 -1
  4. data/documentation/Agent-Tool-Registry.md +1 -1
  5. data/documentation/Configuration.md +3 -6
  6. data/documentation/How-PWN-Works.md +2 -2
  7. data/documentation/Reinforcement-Learning.md +1 -1
  8. data/documentation/diagrams/dot/task-summarizer.dot +3 -3
  9. data/documentation/pwn-ai-Agent.md +16 -35
  10. data/lib/pwn/ai/agent/loop.rb +272 -196
  11. data/lib/pwn/ai/agent/mistakes.rb +17 -0
  12. data/lib/pwn/ai/agent/policy.rb +55 -5
  13. data/lib/pwn/ai/agent/prompt_builder.rb +38 -12
  14. data/lib/pwn/ai/agent/registry.rb +17 -9
  15. data/lib/pwn/ai/agent/task_summarizer.rb +374 -503
  16. data/lib/pwn/ai/anthropic.rb +1 -0
  17. data/lib/pwn/ai/gemini.rb +1 -0
  18. data/lib/pwn/ai/grok.rb +1 -0
  19. data/lib/pwn/ai/ollama.rb +1 -0
  20. data/lib/pwn/ai/open_ai.rb +1 -0
  21. data/lib/pwn/ai/open_web_ui.rb +1 -0
  22. data/lib/pwn/config.rb +1 -1
  23. data/lib/pwn/cron.rb +1 -1
  24. data/lib/pwn/plugins/repl.rb +0 -6
  25. data/lib/pwn/plugins/tty_spinner.rb +7 -11
  26. data/lib/pwn/version.rb +1 -1
  27. data/spec/integration/prompt_builder_spec.rb +6 -4
  28. data/spec/lib/pwn/ai/agent/loop_spec.rb +251 -33
  29. data/spec/lib/pwn/ai/agent/mistakes_spec.rb +14 -0
  30. data/spec/lib/pwn/ai/agent/policy_spec.rb +52 -1
  31. data/spec/lib/pwn/ai/agent/prompt_builder_spec.rb +4 -9
  32. data/spec/lib/pwn/ai/agent/registry_spec.rb +22 -2
  33. data/spec/lib/pwn/ai/agent/signal_hygiene_spec.rb +4 -5
  34. data/spec/lib/pwn/ai/agent/task_summarizer_spec.rb +321 -90
  35. data/spec/lib/pwn/plugins/tty_spinner_spec.rb +0 -48
  36. data/third_party/pwn_rdoc.jsonl +24 -9
  37. metadata +1 -1
@@ -13,9 +13,29 @@ describe PWN::AI::Agent::Registry do
13
13
  expect(help_response).to respond_to :help
14
14
  end
15
15
 
16
- it 'DEFAULT_PREFERENCE is memory_recall, sessions_view, pwn_eval, shell, mistakes_record, mistakes_resolve, learning_note_outcome, memory_remember' do
16
+ it 'DEFAULT_PREFERENCE names only CORE_TOOLS in a recall-first fallback' do
17
17
  expect(described_class::DEFAULT_PREFERENCE).to eq(
18
- %w[memory_recall sessions_view pwn_eval shell mistakes_record mistakes_resolve learning_note_outcome memory_remember]
18
+ %w[memory_recall pwn_eval shell mistakes_record mistakes_resolve learning_note_outcome memory_remember]
19
19
  )
20
+ expect(described_class::DEFAULT_PREFERENCE - described_class::CORE_TOOLS).to be_empty
21
+ expect(described_class::DEFAULT_PREFERENCE).not_to include('sessions_view')
22
+ end
23
+
24
+ it 'preference_order defaults to shell / pwn_eval — there is no request type' do
25
+ expect(described_class.preference_order.first(2)).to eq(%w[shell pwn_eval])
26
+ expect(described_class.preference_order(kind: :question).first(2)).to eq(%w[shell pwn_eval])
27
+ end
28
+
29
+ it 'preference_order still honors explicit empty list and Env override' do
30
+ expect(described_class.preference_order(order: [])).to eq([])
31
+ expect(described_class.preference_order(preference: %w[shell])).to eq(%w[shell])
32
+ end
33
+
34
+ it 'definitions(core_only: true) ships CORE_TOOLS, not the full ~85 schema set' do
35
+ described_class.discover
36
+ names = described_class.definitions(core_only: true).map { |t| t.dig(:function, :name) }
37
+ expect(names).to match_array(described_class::CORE_TOOLS)
38
+ expect(names).not_to include('sessions_view')
39
+ expect(names.length).to be <= described_class::CORE_TOOLS.length
20
40
  end
21
41
  end
@@ -107,12 +107,11 @@ describe 'P0 signal hygiene (handler + inbox + policy-cold + calibrate)' do
107
107
  expect(one[:n]).to eq(1)
108
108
  end
109
109
 
110
- it 'classifies review/opinion asks as questions' do
110
+ it 'treats review/opinion asks as goals with no request type' do
111
111
  allow(PWN::Env).to receive(:dig).and_call_original
112
- allow(PWN::Env).to receive(:dig).with(:ai, :agent, :request_kind_llm).and_return(false)
113
112
  allow(PWN::Env).to receive(:dig).with(:ai, :agent, :task_summary_llm).and_return(false)
114
- expect(PWN::AI::Agent::TaskSummarizer.request_kind(request: 'suggest areas for improvement')).to eq(:question)
115
- expect(PWN::AI::Agent::TaskSummarizer.needs_task_breakdown?(request: 'suggest areas for improvement')).to eq(false)
116
- expect(PWN::AI::Agent::TaskSummarizer.request_kind(request: 'fix all the issues you just described')).to eq(:autonomous_goal)
113
+ expect(PWN::AI::Agent::TaskSummarizer).not_to respond_to(:request_kind)
114
+ expect(PWN::AI::Agent::TaskSummarizer.needs_task_breakdown?(request: 'suggest areas for improvement')).to eq(true)
115
+ expect(PWN::AI::Agent::TaskSummarizer.needs_task_breakdown?(request: 'fix all the issues you just described')).to eq(true)
117
116
  end
118
117
  end
@@ -51,6 +51,8 @@ describe PWN::AI::Agent::TaskSummarizer do
51
51
  expect(described_class).to respond_to :active_task_prompt
52
52
  expect(described_class).to respond_to :relevance_query
53
53
  expect(described_class).to respond_to :tool_jargon_task?
54
+ expect(described_class).to respond_to :plan_open?
55
+ expect(described_class).to respond_to :unfinished_tasks
54
56
  end
55
57
 
56
58
  it 'format_plan method should exist' do
@@ -330,7 +332,7 @@ describe PWN::AI::Agent::TaskSummarizer do
330
332
  expect(state[:plan_source]).to eq :llm
331
333
  end
332
334
 
333
- it 'about_to advances through task k/n without restating the full goal every batch' do
335
+ it 'about_to keeps task k/n without restating the full goal every batch' do
334
336
  expect(described_class).to receive(:chat_for_plan).with(request: subnet_req).and_return(llm_subnet_json)
335
337
  state = described_class.fresh(request: subnet_req)
336
338
  plan_text = described_class.emit_plan!(state: state)
@@ -346,13 +348,15 @@ describe PWN::AI::Agent::TaskSummarizer do
346
348
  expect(first).to match(/via |shell/i)
347
349
  expect(first).not_to include(subnet_req)
348
350
 
349
- # Simulate tools completing so plan_idx can advance
351
+ # Search-only tools must NOT consume the English plan. A later batch
352
+ # with a different tool still restates task k/n, never the full goal.
350
353
  3.times do |i|
351
354
  described_class.record!(state: state, name: 'shell', args: "probe #{i}", result: '{success:true}')
352
355
  end
356
+ expect(state[:plan_idx]).to eq 0
353
357
 
354
358
  second = described_class.about_to(
355
- tools: [{ name: 'shell', args: { 'command' => 'nmap -sn 192.168.1.0/24' } }],
359
+ tools: [{ name: 'pwn_eval', args: { 'code' => '1+1' } }],
356
360
  state: state
357
361
  )
358
362
  expect(second).to be_a(String)
@@ -360,17 +364,13 @@ describe PWN::AI::Agent::TaskSummarizer do
360
364
  expect(second).not_to include(subnet_req)
361
365
  end
362
366
 
363
- it 'how-to questions get no multi-step task breakdown' do
367
+ it 'how-to questions still get a task compass — there is no request type' do
364
368
  allow(PWN::Env).to receive(:dig).and_call_original
365
369
  allow(PWN::Env).to receive(:dig).with(:ai, :agent, :task_summary_llm).and_return(false)
366
370
  req = 'how to do a ping sweep of a subnet using hping3?'
367
- expect(described_class.request_kind(request: req)).to eq(:question)
368
- expect(described_class.needs_task_breakdown?(request: req)).to eq(false)
369
- tasks = described_class.plan(request: req)
370
- expect(tasks).to eq([])
371
- banner = described_class.format_plan(tasks: tasks, request: req)
372
- expect(banner).to match(/Request type: question/i)
373
- expect(banner).not_to match(/Tangible tasks \(\d+\)/)
371
+ expect(described_class.needs_task_breakdown?(request: req)).to eq(true)
372
+ expect(described_class).not_to respond_to :request_kind
373
+ expect(described_class.plan(request: 'what color is a cherry')).to eq([])
374
374
  end
375
375
 
376
376
  it 'falls back generically when LLM is disabled (no static domain scripts)' do
@@ -462,23 +462,26 @@ describe PWN::AI::Agent::TaskSummarizer do
462
462
  expect(info[:item]).to include('locate')
463
463
 
464
464
  ctx = described_class.plan_context(state: state)
465
- expect(ctx).to include('English tangible tasks are the SOLE driver')
465
+ expect(ctx).to include('advisory compass')
466
+ expect(ctx).not_to include('SOLE driver')
466
467
  expect(ctx).to include('▶ task 1/3:')
467
468
  expect(ctx).to include('task 2/3:')
469
+ expect(ctx).to include("Original request (immutable): #{goal}")
468
470
 
469
471
  first = described_class.active_task_prompt(state: state, force: true)
470
472
  expect(first).to include('Active:')
471
473
  expect(first).to include('task 1/3:')
474
+ expect(first).to include("Original request (immutable): #{goal}")
472
475
  # second call same idx → nil (no spam)
473
476
  expect(described_class.active_task_prompt(state: state)).to be_nil
474
477
 
475
478
  state[:plan_idx] = 1
476
479
  nxt = described_class.active_task_prompt(state: state)
477
- expect(nxt).to match(%r{focus on task 2/3:}i)
480
+ expect(nxt).to match(%r{Compass: task 2/3:}i)
481
+ expect(nxt).to include("Original request (immutable): #{goal}")
478
482
  end
479
483
 
480
- it 'record! emits advancement brief when plan_idx moves' do
481
- # Drive PRM streak on locate task with search intents
484
+ it 'record! does not treat two successful searches as finishing the whole plan' do
482
485
  line = nil
483
486
  2.times do
484
487
  line = described_class.record!(
@@ -488,7 +491,70 @@ describe PWN::AI::Agent::TaskSummarizer do
488
491
  result: '{"success":true,"result":{"stdout":"hit","exit":0}}'
489
492
  )
490
493
  end
491
- expect(state[:plan_idx]).to be >= 1
494
+ expect(state[:plan_idx]).to be < (state[:plan].length - 1)
495
+ expect(described_class.plan_open?(state: state)).to eq true
496
+ leftover = described_class.unfinished_tasks(state: state).map { |t| t[:idx] }
497
+ expect(leftover).to include(2)
498
+ end
499
+
500
+ it 'two ls/find calls do not skip to the last English task' do
501
+ described_class.active_task_prompt(state: state, force: true)
502
+ briefs = [
503
+ 'ls /opt/pwn/lib/pwn/ai/agent',
504
+ 'find /opt/pwn/lib -name task_summarizer.rb'
505
+ ].map do |cmd|
506
+ described_class.record!(
507
+ state: state,
508
+ name: 'shell',
509
+ args: { 'command' => cmd },
510
+ result: '{"success":true,"result":{"stdout":"task_summarizer.rb","exit":0}}'
511
+ )
512
+ end
513
+
514
+ expect(state[:plan_idx]).to be < (state[:plan].length - 1)
515
+ expect(briefs.join).not_to match(/Completed task/i)
516
+
517
+ info = described_class.active_task(state: state)
518
+ leftover = described_class.unfinished_tasks(state: state).map { |t| t[:idx] }
519
+ expect(described_class.plan_open?(state: state)).to eq true
520
+ expect(leftover).to include(2)
521
+ # Covered locate must not keep focus on task 1 while later work is open.
522
+ expect(info[:idx]).to eq(leftover.min)
523
+ expect(info[:idx]).not_to eq(state[:plan].length - 1)
524
+ end
525
+
526
+ it 'active_task_prompt stays quiet once every English task is covered' do
527
+ st = described_class.fresh(request: 'what is my hostname?')
528
+ st[:plan] = [
529
+ 'Determine the local hostname',
530
+ 'Present the result and report completion'
531
+ ]
532
+ st[:plan_idx] = 0
533
+ st[:plan_emitted] = true
534
+ described_class.record!(
535
+ state: st,
536
+ name: 'shell',
537
+ args: { 'command' => 'hostname' },
538
+ result: '{"success":true,"result":{"stdout":"kali-box","exit":0}}'
539
+ )
540
+ expect(described_class.plan_open?(state: st)).to eq false
541
+ expect(described_class.active_task_prompt(state: st, force: true)).to be_nil
542
+ end
543
+
544
+ it 'record! emits advancement brief only on real handoff to the next English task' do
545
+ described_class.record!(
546
+ state: state,
547
+ name: 'shell',
548
+ args: { 'command' => 'rg TaskSummarizer lib' },
549
+ result: '{"success":true,"result":{"stdout":"lib/pwn/ai/agent/task_summarizer.rb","exit":0}}'
550
+ )
551
+ line = described_class.record!(
552
+ state: state,
553
+ name: 'shell',
554
+ args: { 'command' => 'sed -i s/truncation/fixed/ lib/pwn/ai/agent/task_summarizer.rb' },
555
+ result: '{"success":true,"result":{"stdout":"patched task_summarizer.rb","exit":0}}'
556
+ )
557
+ expect(state[:plan_idx]).to eq 1
492
558
  expect(line).to be_a(String)
493
559
  expect(line).to match(%r{Advanced past task 1/3}i)
494
560
  expect(line).to match(%r{now task 2/3:}i)
@@ -574,8 +640,7 @@ describe PWN::AI::Agent::TaskSummarizer do
574
640
  expect(q).to include('run rspec to verify')
575
641
  end
576
642
 
577
- it 'apply_prm_advancement! advances on +1 intent-matched streak and holds on -1' do
578
- # first +1 alone is only a streak, not yet advance
643
+ it 'apply_prm_advancement! holds on +1 search streak and on -1; advances on next-task handoff' do
579
644
  described_class.apply_prm_advancement!(
580
645
  state: state,
581
646
  rewards: [1],
@@ -583,7 +648,7 @@ describe PWN::AI::Agent::TaskSummarizer do
583
648
  names: ['shell']
584
649
  )
585
650
  expect(state[:plan_idx]).to eq 0
586
- expect(%i[streak advance]).to include(state[:last_prm_signal])
651
+ expect(state[:last_prm_signal].to_s).to match(/hold|streak/)
587
652
 
588
653
  described_class.apply_prm_advancement!(
589
654
  state: state,
@@ -591,8 +656,19 @@ describe PWN::AI::Agent::TaskSummarizer do
591
656
  intents: %w[search read],
592
657
  names: ['shell']
593
658
  )
659
+ # Two successful searches must NOT complete "locate the source".
660
+ expect(state[:plan_idx]).to eq 0
661
+
662
+ state[:tools_on_task] = 2
663
+ described_class.apply_prm_advancement!(
664
+ state: state,
665
+ rewards: [1],
666
+ intents: ['edit'],
667
+ names: ['shell'],
668
+ result: 'patched task_summarizer.rb'
669
+ )
594
670
  expect(state[:plan_idx]).to eq 1
595
- expect(state[:last_prm_signal]).to eq :advance
671
+ expect(state[:last_prm_signal].to_s).to match(/advance/)
596
672
 
597
673
  hold_idx = state[:plan_idx]
598
674
  described_class.apply_prm_advancement!(
@@ -643,105 +719,260 @@ describe PWN::AI::Agent::TaskSummarizer do
643
719
  end
644
720
  end
645
721
 
646
- describe 'request_kind: statement | question | autonomous_goal' do
722
+ describe 'task compass (no request type)' do
647
723
  before do
648
724
  allow(PWN::Env).to receive(:dig).and_call_original
649
725
  allow(PWN::Env).to receive(:dig).with(:ai, :agent, :task_summary_llm).and_return(false)
650
- allow(PWN::Env).to receive(:dig).with(:ai, :agent, :request_kind_llm).and_return(false)
651
726
  end
652
727
 
653
- it 'exposes request_kind and needs_task_breakdown?' do
654
- expect(described_class).to respond_to :request_kind
728
+ it 'has no request_kind classifier' do
729
+ expect(described_class).not_to respond_to :request_kind
730
+ expect(described_class).not_to respond_to :heuristic_request_kind
731
+ expect(described_class).not_to respond_to :parse_kind_label
732
+ expect(described_class).not_to respond_to :chat_for_kind
655
733
  expect(described_class).to respond_to :needs_task_breakdown?
656
- expect(described_class).to respond_to :llm_kind_enabled?
657
- expect(described_class).to respond_to :heuristic_request_kind
658
- expect(described_class).to respond_to :llm_classify_kind
659
- expect(described_class).to respond_to :chat_for_kind
660
- expect(described_class).to respond_to :parse_kind_label
661
- end
662
-
663
- it 'classifies general statements without multi-step plans' do
664
- req = 'FYI the staging build is green.'
665
- expect(described_class.request_kind(request: req)).to eq(:statement)
666
- expect(described_class.needs_task_breakdown?(request: req)).to eq(false)
667
- st = described_class.fresh(request: req)
668
- expect(st[:request_kind]).to eq(:statement)
669
- expect(described_class.plan(request: req, state: st)).to eq([])
670
- expect(st[:plan_source].to_s).to match(/no_breakdown/)
671
- text = described_class.emit_plan!(state: st)
672
- expect(text.to_s).to match(/Request type: statement/i)
673
- expect(text.to_s).not_to match(%r{task 1/\d+:})
674
- end
675
-
676
- it 'classifies questions without multi-step plans' do
677
- req = 'what is the default GQRX remote-control port?'
678
- expect(described_class.request_kind(request: req)).to eq(:question)
679
- expect(described_class.needs_task_breakdown?(kind: :question)).to eq(false)
680
- expect(described_class.plan(request: req)).to eq([])
681
- end
682
-
683
- it 'classifies host-evidence questions as autonomous goals' do
684
- req = 'what is my hostname?'
685
- expect(described_class.request_kind(request: req)).to eq(:autonomous_goal)
686
- expect(described_class.needs_task_breakdown?(request: req)).to eq(true)
687
- expect(described_class.request_kind(request: 'excellent - what is my hostname?')).to eq(:autonomous_goal)
734
+ expect(described_class.needs_task_breakdown?(request: 'anything')).to eq(true)
688
735
  end
689
736
 
690
- it 'classifies autonomous goals and decomposes into ordered work units' do
737
+ it 'fresh state has no request_kind and format_plan has no Request type line' do
691
738
  req = 'refactor the authentication middleware and add unit tests'
692
- expect(described_class.request_kind(request: req)).to eq(:autonomous_goal)
693
- expect(described_class.needs_task_breakdown?(request: req)).to eq(true)
739
+ st = described_class.fresh(request: req)
740
+ expect(st).not_to have_key(:request_kind)
694
741
  steps = [
695
742
  'locate the authentication middleware source',
696
743
  'refactor the middleware for clarity and safety',
697
744
  'add unit tests covering the new behavior',
698
745
  'run the test suite and report completion'
699
746
  ]
700
- st = described_class.fresh(request: req)
701
747
  tasks = described_class.plan(request: req, state: st, tasks: steps)
702
748
  expect(tasks.length).to be >= 3
703
- expect(st[:request_kind]).to eq(:autonomous_goal)
704
- text = described_class.format_plan(tasks: tasks, request: req, request_kind: :autonomous_goal)
705
- expect(text).to match(/Request type: autonomous_goal/)
749
+ text = described_class.format_plan(tasks: tasks, request: req)
750
+ expect(text).not_to match(/Request type/)
706
751
  expect(text).to match(/Tangible tasks/)
707
752
  expect(text).to match(%r{task 1/\d+:})
708
- # each work unit may use many tools — banner still says so
709
- expect(text).to match(/one or more tools/i)
710
753
  end
711
754
 
712
- it 'treats polite agent-do questions as autonomous goals' do
713
- req = 'can you fix the truncation bug in TaskSummarizer?'
714
- expect(described_class.request_kind(request: req)).to eq(:autonomous_goal)
715
- expect(described_class.needs_task_breakdown?(request: req)).to eq(true)
755
+ it 'plan_open? stays true until mutate and verify English tasks have evidence' do
756
+ st = described_class.fresh(request: 'locate, fix, verify')
757
+ st[:plan] = [
758
+ 'locate the TaskSummarizer source',
759
+ 'fix the truncation bug',
760
+ 'run rspec to verify'
761
+ ]
762
+ st[:plan_idx] = 0
763
+ expect(described_class.plan_open?(state: st)).to eq true
764
+
765
+ described_class.record!(
766
+ state: st,
767
+ name: 'shell',
768
+ args: { 'command' => 'rg TaskSummarizer lib' },
769
+ result: '{"success":true,"result":{"stdout":"lib/pwn/ai/agent/task_summarizer.rb","exit":0}}'
770
+ )
771
+ expect(described_class.plan_open?(state: st)).to eq true
772
+ expect(described_class.unfinished_tasks(state: st).map { |t| t[:idx] }).to include(1, 2)
773
+
774
+ described_class.record!(
775
+ state: st,
776
+ name: 'shell',
777
+ args: { 'command' => 'sed -i s/bug/fix/ lib/pwn/ai/agent/task_summarizer.rb' },
778
+ result: '{"success":true,"result":{"stdout":"patched task_summarizer.rb","exit":0}}'
779
+ )
780
+ open_idx = described_class.unfinished_tasks(state: st).map { |t| t[:idx] }
781
+ expect(open_idx).to include(2)
782
+ expect(open_idx).not_to include(1)
783
+ expect(described_class.plan_open?(state: st)).to eq true
784
+
785
+ described_class.record!(
786
+ state: st,
787
+ name: 'shell',
788
+ args: { 'command' => 'bundle exec rspec spec/lib/pwn/ai/agent/task_summarizer_spec.rb' },
789
+ result: '{"success":true,"result":{"stdout":"12 examples, 0 failures","exit":0}}'
790
+ )
791
+ expect(described_class.plan_open?(state: st)).to eq false
792
+ expect(described_class.unfinished_tasks(state: st)).to eq([])
793
+ end
794
+
795
+ it 'task_phase classifies English stems (locate/verify/change), not only exact \b-words' do
796
+ expect(described_class.send(:task_phase, item: 'locate the TaskSummarizer source')).to eq(:discover)
797
+ expect(described_class.send(:task_phase, item: 'identify where tasks are skipped')).to eq(:discover)
798
+ expect(described_class.send(:task_phase, item: 'Determine the local hostname')).to eq(:discover)
799
+ expect(described_class.send(:task_phase, item: 'change the advancement logic')).to eq(:mutate)
800
+ expect(described_class.send(:task_phase, item: 'improve completion handling')).to eq(:mutate)
801
+ expect(described_class.send(:task_phase, item: 'Refactor Loop.run')).to eq(:mutate)
802
+ expect(described_class.send(:task_phase, item: 'Verify the result and report completion')).to eq(:present)
803
+ expect(described_class.send(:task_phase, item: 'Present the result and report completion')).to eq(:present)
804
+ expect(described_class.send(:task_phase, item: 'Verify full task completion end-to-end')).to eq(:verify)
805
+ expect(described_class.send(:task_phase, item: 'run rspec to verify')).to eq(:verify)
806
+ end
807
+
808
+ it 'host-evidence plans close after a live lookup (forced closer is not a test-runner verify)' do
809
+ st = described_class.fresh(request: 'what is my hostname?')
810
+ closer = described_class.send(
811
+ :normalize_task_list,
812
+ tasks: ['Determine the local hostname'],
813
+ goal: 'what is my hostname?'
814
+ )
815
+ expect(closer.join(' ')).not_to match(/\brspec\b|\brubocop\b/)
816
+ expect(described_class.send(:task_phase, item: closer.last)).not_to eq(:verify)
817
+ st[:plan] = closer
818
+ st[:plan_idx] = 0
819
+ described_class.record!(
820
+ state: st,
821
+ name: 'shell',
822
+ args: { 'command' => 'hostname' },
823
+ result: '{"success":true,"result":{"stdout":"kali-box","exit":0}}'
824
+ )
825
+ expect(described_class.plan_open?(state: st)).to eq false
826
+ expect(described_class.unfinished_tasks(state: st)).to eq([])
716
827
  end
717
828
 
718
- it 'honors injected llm_kind and parse_kind_label' do
719
- expect(described_class.parse_kind_label(raw: 'STATEMENT')).to eq(:statement)
720
- expect(described_class.parse_kind_label(raw: 'question')).to eq(:question)
721
- expect(described_class.parse_kind_label(raw: 'autonomous_goal')).to eq(:autonomous_goal)
722
- expect(described_class.parse_kind_label(raw: 'Label: goal')).to eq(:autonomous_goal)
723
- expect(
724
- described_class.request_kind(request: 'ambiguous residual text here', llm_kind: 'question')
725
- ).to eq(:question)
829
+ it 'three on-task discover tools advance one English task and do not jump to the last' do
830
+ st = described_class.fresh(request: 'map then fix then verify')
831
+ st[:plan] = [
832
+ 'locate the TaskSummarizer source',
833
+ 'identify where planned tasks are skipped',
834
+ 'Implement the completion fix',
835
+ 'Verify full task completion end-to-end'
836
+ ]
837
+ st[:plan_idx] = 0
838
+ 3.times do |i|
839
+ described_class.record!(
840
+ state: st,
841
+ name: 'shell',
842
+ args: { 'command' => "rg TaskSummarizer lib #{i}" },
843
+ result: '{"success":true,"result":{"stdout":"lib/pwn/ai/agent/task_summarizer.rb","exit":0}}'
844
+ )
845
+ end
846
+ expect(st[:plan_idx]).to eq 1
847
+ expect(st[:plan_idx]).to be < (st[:plan].length - 1)
848
+ expect(described_class.plan_open?(state: st)).to eq true
726
849
  end
727
850
 
728
- it 'uses chat_for_kind when request_kind_llm is enabled' do
729
- allow(PWN::Env).to receive(:dig).with(:ai, :agent, :request_kind_llm).and_return(true)
730
- allow(described_class).to receive(:chat_for_kind).and_return('statement')
731
- # Strong goal regex still wins before LLM for known agent-do.
732
- expect(described_class.request_kind(request: 'implement the feature')).to eq(:autonomous_goal)
733
- # Ambiguous residual goes to LLM.
734
- expect(described_class.request_kind(request: 'the purple widgets arrived yesterday afternoon')).to eq(:statement)
735
- expect(described_class).to have_received(:chat_for_kind).at_least(:once)
851
+ it 'a verify task is covered after the verifier ran, even with remaining offenses' do
852
+ st = described_class.fresh(request: 'run rubocop')
853
+ st[:plan] = [
854
+ 'Refactor Loop.run',
855
+ 'run rubocop to verify'
856
+ ]
857
+ st[:plan_idx] = 1
858
+ described_class.record!(
859
+ state: st,
860
+ name: 'shell',
861
+ args: { 'command' => 'bundle exec rubocop lib/pwn/ai/agent/loop.rb' },
862
+ result: '{"success":true,"result":{"stdout":"4 offenses detected","exit":1}}'
863
+ )
864
+ leftover = described_class.unfinished_tasks(state: st).map { |t| t[:idx] }
865
+ expect(leftover).not_to include(1)
866
+ end
867
+
868
+ it 'five successful ls/find shells never skip a 5-step mapping plan to the last task' do
869
+ st = described_class.fresh(request: 'map then fix autonomous_goal completion')
870
+ st[:plan] = [
871
+ 'Map how autonomous_goal plans, tracks, and completes work units',
872
+ 'Identify where and why planned tasks are skipped',
873
+ 'Determine the root cause of premature finalized responses',
874
+ 'Implement fixes so every planned task is finished',
875
+ 'Verify full task completion end-to-end'
876
+ ]
877
+ st[:plan_idx] = 0
878
+ st[:plan_emitted] = true
879
+ 5.times do |i|
880
+ described_class.record!(
881
+ state: st,
882
+ name: 'shell',
883
+ args: { 'command' => "ls /opt/pwn && find lib -name '*#{i}*'" },
884
+ result: '{"success":true,"result":{"stdout":"/opt/pwn","exit":0}}'
885
+ )
886
+ end
887
+ expect(st[:plan_idx]).to be < (st[:plan].length - 1)
888
+ expect(described_class.plan_open?(state: st)).to eq true
889
+ unfinished = described_class.unfinished_tasks(state: st).map { |t| t[:item] }
890
+ expect(unfinished.join(' ')).to match(/Implement|Verify/i)
736
891
  end
737
892
 
738
- it 'about_to does not invent a multi-step plan for questions' do
893
+ it 'about_to plans a question-shaped request with no request type' do
739
894
  req = 'what flags does nmap use for a ping scan?'
740
895
  st = described_class.fresh(request: req)
741
- line = described_class.about_to(tools: [{ name: 'shell', args: { 'command' => 'man nmap' } }], state: st)
742
- # May brief tools, but plan stays empty
743
- expect(Array(st[:plan])).to eq([])
744
- expect(st[:request_kind]).to eq(:question)
896
+ described_class.about_to(tools: [{ name: 'shell', args: { 'command' => 'man nmap' } }], state: st)
897
+ expect(st).not_to have_key(:request_kind)
898
+ end
899
+ end
900
+
901
+ describe 'original request visibility and PLAN: wrapper resistance' do
902
+ let(:original) do
903
+ 'Fix TaskSummarizer so that it has visibility of the original request made by the user so tasks are relevant to the original request.'
904
+ end
905
+ let(:wrapped) do
906
+ <<~REQ
907
+ REQUEST:
908
+ GOAL: #{original}
909
+
910
+ PLAN:
911
+ 1. shell command="rg -n --glob '!{vendor,tmp,pkg,.git,coverage}/**' -e 'TaskSummarizer|task_summarizer' /opt/pwn/lib /opt/pwn/spec /opt/pwn/bin"
912
+ 2. shell command="sed -n '1,260p' /opt/pwn/lib/pwn/ai/agent/task_summarizer.rb"
913
+ 3. shell command="rg -n -e original_request /opt/pwn/lib/pwn/ai"
914
+ REQ
915
+ end
916
+
917
+ it 'canonical_request strips GOAL/PLAN wrappers down to the operator ask' do
918
+ expect(described_class.canonical_request(request: wrapped)).to eq(original)
919
+ expect(described_class.canonical_request(request: original)).to eq(original)
920
+ collapsed = "GOAL: #{original} PLAN: 1. shell command=\"rg TaskSummarizer\" 2. shell command=\"sed -n 1,20p file\""
921
+ expect(described_class.canonical_request(request: collapsed)).to eq(original)
922
+ end
923
+
924
+ it 'fresh pins original_request to the canonical ask, not the PLAN: scaffold' do
925
+ st = described_class.fresh(request: wrapped)
926
+ expect(st[:original_request]).to eq(original)
927
+ expect(st[:request]).to eq(original)
928
+ expect(st[:request]).not_to match(/shell command=/i)
929
+ end
930
+
931
+ it 'plan does not treat PLAN: tool-call enumerations as English tasks' do
932
+ allow(described_class).to receive(:chat_for_plan).and_return(
933
+ [
934
+ 'Locate the TaskSummarizer component and how it builds tasks',
935
+ 'Identify where the original user request is available in the call chain',
936
+ 'Update TaskSummarizer so it receives and uses the original user request',
937
+ 'Verify generated tasks stay relevant to that original request'
938
+ ].to_json
939
+ )
940
+ st = described_class.fresh(request: wrapped)
941
+ tasks = described_class.plan(request: wrapped, state: st)
942
+ expect(st[:plan_source]).not_to eq(:enumerated)
943
+ expect(tasks.join(' ')).not_to match(/shell command=/i)
944
+ expect(tasks.join(' ').downcase).to match(/original/)
945
+ expect(st[:original_request]).to eq(original)
946
+ end
947
+
948
+ it 'plan_context and compact focus restate the original request' do
949
+ st = described_class.fresh(request: wrapped)
950
+ st[:plan] = [
951
+ 'Locate the TaskSummarizer component',
952
+ 'Identify request visibility in the call chain',
953
+ 'Update TaskSummarizer to use the original request'
954
+ ]
955
+ st[:plan_idx] = 0
956
+ st[:plan_emitted] = true
957
+
958
+ ctx = described_class.plan_context(state: st)
959
+ expect(ctx).to include("Original request (immutable): #{original}")
960
+ expect(ctx).not_to include('shell command=')
961
+
962
+ first = described_class.active_task_prompt(state: st, force: true)
963
+ expect(first).to include("Original request (immutable): #{original}")
964
+
965
+ st[:plan_idx] = 1
966
+ nxt = described_class.active_task_prompt(state: st)
967
+ expect(nxt).to include("Original request (immutable): #{original}")
968
+ end
969
+
970
+ it 'still honors genuine English numbered steps from the user' do
971
+ req = 'Do three things. 1. locate the file 2. patch the truncation 3. add plan on submit'
972
+ tasks = described_class.plan(request: req)
973
+ expect(tasks.length).to be >= 3
974
+ expect(tasks[0].downcase).to include('locate')
975
+ expect(tasks[1].downcase).to match(/patch|truncat/)
745
976
  end
746
977
  end
747
978
  end
@@ -10,52 +10,4 @@ describe PWN::Plugins::TTYSpinner do
10
10
  it 'should display information for existing help method' do
11
11
  expect(described_class).to respond_to :help
12
12
  end
13
-
14
- # StringIO is never a TTY so TTY::Spinner#write is a no-op, but the
15
- # auto_spin worker thread still starts. That is enough to prove #stop
16
- # actually joins the worker (the bug: ensure spin.stop left it sleeping
17
- # so a following response could still be overwritten).
18
- def run_rest_like(opts = {})
19
- provide_response = opts.fetch(:provide_response, true)
20
- spinner = true
21
- spin = described_class.start(output: StringIO.new) if spinner
22
- response = nil
23
- begin
24
- sleep 0.15
25
- response = provide_response ? 'HTTP_RESPONSE' : nil
26
- response
27
- ensure
28
- described_class.stop(spin: spin)
29
- end
30
- end
31
-
32
- it 'joins the auto_spin thread when a response is provided' do
33
- before = Thread.list.size
34
- result = run_rest_like(provide_response: true)
35
- expect(result).to eq('HTTP_RESPONSE')
36
- sleep 0.05
37
- expect(Thread.list.size).to eq(before)
38
- end
39
-
40
- it 'joins the auto_spin thread when no response is provided' do
41
- before = Thread.list.size
42
- result = run_rest_like(provide_response: false)
43
- expect(result).to be_nil
44
- sleep 0.05
45
- expect(Thread.list.size).to eq(before)
46
- end
47
-
48
- it 'is a no-op when spin is nil' do
49
- expect { described_class.stop(spin: nil) }.not_to raise_error
50
- end
51
-
52
- it 'marks the spinner done and dead after stop' do
53
- spin = described_class.start(output: StringIO.new)
54
- thread = spin.instance_variable_get(:@thread)
55
- expect(thread).not_to be_nil
56
- described_class.stop(spin: spin)
57
- expect(spin.done?).to be true
58
- expect(spin.spinning?).to be false
59
- expect(thread.status).to eq(false).or(be_nil)
60
- end
61
13
  end