pwn 0.5.679 → 0.5.682

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. checksums.yaml +4 -4
  2. data/.gitignore +1 -0
  3. data/documentation/AI-Integration.md +1 -1
  4. data/documentation/Agent-Tool-Registry.md +1 -1
  5. data/documentation/Configuration.md +3 -6
  6. data/documentation/How-PWN-Works.md +2 -2
  7. data/documentation/Reinforcement-Learning.md +1 -1
  8. data/documentation/diagrams/dot/task-summarizer.dot +3 -3
  9. data/documentation/pwn-ai-Agent.md +16 -35
  10. data/lib/pwn/ai/agent/loop.rb +272 -196
  11. data/lib/pwn/ai/agent/mistakes.rb +17 -0
  12. data/lib/pwn/ai/agent/policy.rb +55 -5
  13. data/lib/pwn/ai/agent/prompt_builder.rb +38 -12
  14. data/lib/pwn/ai/agent/registry.rb +17 -9
  15. data/lib/pwn/ai/agent/task_summarizer.rb +374 -503
  16. data/lib/pwn/ai/anthropic.rb +1 -0
  17. data/lib/pwn/ai/gemini.rb +1 -0
  18. data/lib/pwn/ai/grok.rb +1 -0
  19. data/lib/pwn/ai/ollama.rb +1 -0
  20. data/lib/pwn/ai/open_ai.rb +1 -0
  21. data/lib/pwn/ai/open_web_ui.rb +1 -0
  22. data/lib/pwn/config.rb +1 -1
  23. data/lib/pwn/cron.rb +1 -1
  24. data/lib/pwn/plugins/repl.rb +0 -6
  25. data/lib/pwn/plugins/tty_spinner.rb +7 -11
  26. data/lib/pwn/version.rb +1 -1
  27. data/spec/integration/prompt_builder_spec.rb +6 -4
  28. data/spec/lib/pwn/ai/agent/loop_spec.rb +251 -33
  29. data/spec/lib/pwn/ai/agent/mistakes_spec.rb +14 -0
  30. data/spec/lib/pwn/ai/agent/policy_spec.rb +52 -1
  31. data/spec/lib/pwn/ai/agent/prompt_builder_spec.rb +4 -9
  32. data/spec/lib/pwn/ai/agent/registry_spec.rb +22 -2
  33. data/spec/lib/pwn/ai/agent/signal_hygiene_spec.rb +4 -5
  34. data/spec/lib/pwn/ai/agent/task_summarizer_spec.rb +321 -90
  35. data/spec/lib/pwn/plugins/tty_spinner_spec.rb +0 -48
  36. data/third_party/pwn_rdoc.jsonl +24 -9
  37. metadata +1 -1
@@ -34,6 +34,12 @@ module PWN
34
34
  # PromptBuilder.mistakes_block re-injects the top open mistakes and
35
35
  # top known fixes into the system prompt of every future turn.
36
36
  #
37
+ # COMPLETION
38
+ # ----------
39
+ # The original request is the completion signal. TaskSummarizer and
40
+ # Policy are advisory (compass / rank). Loop keeps calling CORE_TOOLS
41
+ # until that request is done or truly blocked, then stops.
42
+ #
37
43
  # LOCAL-MODEL SCAFFOLDING
38
44
  # -----------------------
39
45
  # When the active engine is :ollama (or the corresponding :agent flags
@@ -160,22 +166,31 @@ module PWN
160
166
 
161
167
  private_class_method def self.maybe_park_budget_scars!
162
168
  return unless defined?(Mistakes)
163
- return if budget_exhaustion_hot?
164
169
  return unless Mistakes.respond_to?(:park)
165
170
 
166
- top = Mistakes.top(limit: 12, unresolved_only: true)
171
+ top = Mistakes.top(limit: 24, unresolved_only: true)
167
172
  now = Time.now
168
- top.each do |mistake|
169
- next unless budget_hit?(mistake: mistake)
170
- next if mistake[:parked]
171
-
173
+ budget = top.select { |mistake| budget_hit?(mistake: mistake) && !mistake[:parked] }
174
+ budget.each do |mistake|
172
175
  stamp = mistake_ts(mistake: mistake)
173
- # cool detector + scar older than PARK_COOL_SECS → park
174
176
  next if stamp && (now - stamp) <= PARK_COOL_SECS
175
177
 
176
178
  Mistakes.park(
177
179
  signature: mistake[:signature].to_s,
178
- reason: 'p17 rate-cool: outside PARK_COOL_SECS while hot?=false'
180
+ reason: 'p17 rate-cool: outside PARK_COOL_SECS'
181
+ )
182
+ end
183
+ live = budget.reject do |mistake|
184
+ row = Mistakes.find(signature: mistake[:signature].to_s)
185
+ row.nil? || row[:parked]
186
+ end
187
+ return if live.length <= 1
188
+
189
+ # Never let 2+ budget scars latch hot forever. Keep only the newest.
190
+ live.sort_by { |mistake| mistake_ts(mistake: mistake) || Time.at(0) }[0...-1].each do |mistake|
191
+ Mistakes.park(
192
+ signature: mistake[:signature].to_s,
193
+ reason: 'p17 keep-newest budget scar; extras parked so tomorrow is not hot'
179
194
  )
180
195
  end
181
196
  rescue StandardError
@@ -215,10 +230,9 @@ module PWN
215
230
  end
216
231
 
217
232
  # P17 — evidence-enough early final: latest tool rounds already answer
218
- # the ask → force synthesis instead of burning iters into text-only tail.
219
- # Must NOT fire on routine tool JSON {"success":true} while a multi-step
220
- # English plan still has open tasks — that blocks legitimate completion
221
- # (mid-fix "write the complete final answer now" thrash).
233
+ # the original request → force synthesis. English tasks are an advisory
234
+ # compass only — an open verify tail must not block a finished ask.
235
+ # The original request is the completion signal.
222
236
  private_class_method def self.evidence_enough_to_finalize?(opts = {})
223
237
  messages = Array(opts[:messages])
224
238
  turn_fails = opts[:turn_fails] || {}
@@ -226,30 +240,15 @@ module PWN
226
240
  max_i = opts[:max_iters].to_i
227
241
  request = opts[:request].to_s
228
242
  return false if max_i <= 0 || iter < 2
229
- # Need runway before the text-only strip, and no thrash.
230
243
  return false if turn_fails['empty_final'].to_i.positive?
231
244
  return false if turn_fails['incomplete_final'].to_i > 1
232
245
 
233
246
  fail_n = turn_fails.values.sum
234
247
  return false if fail_n >= 3
235
248
 
236
- # English-task gate: multi-step plans only early-final on/after the
237
- # last tangible task. plan_idx is 0-based; open work => not enough.
238
- ts_state = opts[:ts_state]
239
- if ts_state.is_a?(Hash)
240
- plan = Array(ts_state[:plan])
241
- if plan.length >= 2
242
- idx = ts_state[:plan_idx].to_i
243
- return false if idx < (plan.length - 1)
244
- end
245
- end
246
-
247
249
  tools_ok = messages.select { |msg| msg[:role].to_s == 'tool' }
248
250
  return false if tools_ok.size < 2
249
251
 
250
- # Last two tool payloads should look like successful evidence, not errors.
251
- # Agent tool wrappers always emit "success":true on ok — that alone is
252
- # NOT proof the user goal is done (do not match bare success JSON).
253
252
  last2 = tools_ok.last(2)
254
253
  return false if last2.any? do |msg|
255
254
  content = msg[:content].to_s
@@ -257,20 +256,19 @@ module PWN
257
256
  !content.match?(/"success"\s*:\s*true/i)
258
257
  end
259
258
 
260
- # Prefer when plan was short / we already spent half the budget usefully.
261
259
  plan_steps = opts[:plan_steps].to_i
262
260
  short_plan = plan_steps.positive? && plan_steps <= 3
263
261
  deep_enough = tools_ok.size >= 3 || (short_plan && tools_ok.size >= plan_steps)
264
262
  return false unless deep_enough
265
263
 
266
264
  recent_txt = last2.map { |msg| msg[:content].to_s[0, 500] }.join(' ')
267
- # Goal-shaped completion only — write/patch/verify, not shell success wrappers.
268
265
  mutation_done = recent_txt.match?(
269
- /syntax ok|wrote |patched|resolved|File\.write|ruby -c|0 offenses|examples?,\s*0 failures/i
266
+ /syntax ok|wrote |patched|File\.write|ruby -c|0 offenses|examples?,\s*0 failures/i
270
267
  )
271
268
  return true if mutation_done
269
+ return true if request_path_evidenced?(request: request, blob: recent_txt)
272
270
  return true if short_plan && tools_ok.size >= plan_steps && fail_n.zero? &&
273
- request.match?(/\b(what|who|when|where|which|how many|status|list|show|print|uname|cwd|version)\b/i)
271
+ request.match?(/\b(what|who|when|where|which|how many|status|list|show|print|uname|cwd|version|hostname)\b/i)
274
272
 
275
273
  false
276
274
  rescue StandardError
@@ -312,6 +310,94 @@ module PWN
312
310
  )
313
311
  /ix
314
312
 
313
+ ACT_REQUEST_RX = /
314
+ \b(write|create|implement|fix|patch|replace|refactor|overwrite|
315
+ add (?:a |the )?|update|install|delete|remove|rename)\b
316
+ /ix
317
+ MUTATION_EVIDENCE_RX = /
318
+ printf\s|tee\s|sed\s+-i|ruby\s+-i|>\s|>>\s|file\.write|binwrite|
319
+ patched|wrote\s|syntax\sok|0\s+offenses|examples?,\s*0\s+failures
320
+ /ix
321
+ LOOKUP_REQUEST_RX = /
322
+ \b(what\s+is\s+my|hostname|uname|cwd|whoami|status|version|how\s+many)\b
323
+ /ix
324
+ HOST_PATH_RX = %r{(?:/|\./)[\w./-]+\.\w+}
325
+
326
+ # True only when the ask needs a live host/file effect. World-knowledge
327
+ # questions ("what color is a cherry") do not.
328
+ public_class_method def self.needs_host_work?(opts = {})
329
+ request = opts[:request].to_s
330
+ return false if request.strip.empty?
331
+ return true if request.match?(ACT_REQUEST_RX)
332
+ return true if request.match?(LOOKUP_REQUEST_RX)
333
+ return true if request.match?(HOST_PATH_RX)
334
+
335
+ false
336
+ rescue StandardError
337
+ false
338
+ end
339
+
340
+ # Short world-knowledge asks (no host/file work). Skip the planner LLM
341
+ # and do not bounce a text-only answer.
342
+ public_class_method def self.world_knowledge?(opts = {})
343
+ request = opts[:request].to_s.strip
344
+ return false if request.empty?
345
+ return false if needs_host_work?(request: request)
346
+ return false if request.length > 120
347
+ return false if request.match?(%r{\b(this\s+(?:host|machine|box|system|subnet|file|repo)|/opt/|implement|scan|hosts?)\b}i)
348
+
349
+ request.match?(/\A(?:what|why|who|when|where|which|how)\b/i)
350
+ rescue StandardError
351
+ false
352
+ end
353
+
354
+ # True when a text-only reply cannot yet be the original request.
355
+ # Distinct from English-task leftovers (advisory) and polite handoffs.
356
+ # Never bounce world-knowledge / no-host-work asks.
357
+ private_class_method def self.request_unsatisfied?(opts = {})
358
+ return false if opts[:last_iter]
359
+ return false if world_knowledge?(request: opts[:request])
360
+ return false unless needs_host_work?(request: opts[:request])
361
+
362
+ request = opts[:request].to_s
363
+ messages = Array(opts[:messages])
364
+ tools = messages.select { |msg| msg.is_a?(Hash) && msg[:role].to_s == 'tool' }
365
+ blob = +''
366
+ tools.each do |msg|
367
+ blob << msg[:name].to_s << ' ' << msg[:content].to_s << "\n"
368
+ end
369
+ messages.each do |msg|
370
+ next unless msg.is_a?(Hash) && msg[:role].to_s == 'assistant'
371
+
372
+ Array(msg[:tool_calls]).each do |tc|
373
+ blob << tc.dig(:function, :name).to_s << ' '
374
+ blob << tc.dig(:function, :arguments).to_s << "\n"
375
+ end
376
+ end
377
+
378
+ return false if request.match?(LOOKUP_REQUEST_RX) && tools.any? && blob.length >= 20
379
+ return true if tools.empty?
380
+ return false if blob.match?(MUTATION_EVIDENCE_RX)
381
+ return false if request_path_evidenced?(request: request, blob: blob)
382
+
383
+ true
384
+ rescue StandardError
385
+ false
386
+ end
387
+
388
+ private_class_method def self.request_path_evidenced?(opts = {})
389
+ request = opts[:request].to_s
390
+ blob = opts[:blob].to_s
391
+ paths = request.scan(%r{(?:/|\./)[\w./-]+\.\w+})
392
+ return false if paths.empty?
393
+
394
+ paths.any? do |path|
395
+ blob.include?(path) && blob.match?(
396
+ /open\(|File\.(?:write|open|binwrite)|write\(|puts\s|print\s|>\s|>>\s|tee\s|sed\s+-i|ruby\s+-i|patched|wrote/i
397
+ )
398
+ end
399
+ end
400
+
315
401
  private_class_method def self.incomplete_final?(opts = {})
316
402
  text = opts[:text].to_s
317
403
  return false if text.strip.empty?
@@ -999,21 +1085,14 @@ module PWN
999
1085
  if env_tc && !env_tc.to_s.empty?
1000
1086
  cwt_opts[:tool_choice] = env_tc
1001
1087
  else
1002
- # Weak chat templates (TEMPLATE {{ .Prompt }} on abliterated
1003
- # Gemma etc.) ignore tools: under tool_choice=auto and dump
1004
- # monologue as content. Stay on required until a tool result
1005
- # exists AND the last assistant turn already looks like a
1006
- # genuine final (no monologue / handoff markers). That keeps
1007
- # pressure on native tool_calls through the mid-loop thrash
1008
- # that previously returned "Wait, let's try hping3…" as FINAL.
1088
+ # After the first tool result, auto so the model can emit a
1089
+ # real final. Leftover English tasks do not keep required.
1009
1090
  has_tool_result = Array(messages).any? { |m| m[:role].to_s == 'tool' }
1010
- last_asst = Array(messages).reverse.find { |m| m[:role].to_s == 'assistant' }
1011
- last_txt = last_asst.is_a?(Hash) ? last_asst[:content].to_s : ''
1012
- still_acting = last_txt.strip.empty? || incomplete_final?(text: last_txt, last_iter: false)
1013
- cwt_opts[:tool_choice] = has_tool_result && !still_acting ? 'auto' : 'required'
1091
+ need_tools = needs_host_work?(request: Array(messages).find { |m| m[:role].to_s == 'user' }&.[](:content))
1092
+ cwt_opts[:tool_choice] = has_tool_result || !need_tools ? 'auto' : 'required'
1014
1093
  end
1015
1094
  end
1016
- response = mod.chat_with_tools(**cwt_opts)
1095
+ response = mod.chat_with_tools(cwt_opts)
1017
1096
  publish_usage(response: response, engine: engine)
1018
1097
  normalize_llm(response: response)
1019
1098
  else
@@ -1063,15 +1142,9 @@ module PWN
1063
1142
  # 3.2 — local models cannot afford auto_introspect (judge+prm+critic+
1064
1143
  # sentinel+extro) on every success. Default :failure_only when local.
1065
1144
  private_class_method def self.should_auto_introspect?(opts = {})
1066
- kind = (opts[:kind] || Thread.current[:pwn_request_kind]).to_s.to_sym
1067
1145
  intent = (opts[:intent] || Thread.current[:pwn_request_intent]).to_s.to_sym
1068
- fails = opts[:turn_fails].is_a?(Hash) ? opts[:turn_fails].values.sum : 0
1069
- # Cheap answers already returned user-visible text. The post-answer
1070
- # critic + 12s ORM printed ERROR: Timed out reading data from server
1071
- # after greetings / takes / questions.
1146
+ # Cheap answers already returned user-visible text.
1072
1147
  return false if %i[greeting howto recall].include?(intent)
1073
- return false if kind == :statement
1074
- return false if kind == :question && fails.zero?
1075
1148
 
1076
1149
  return true unless opts[:local]
1077
1150
 
@@ -1116,9 +1189,13 @@ module PWN
1116
1189
  messages = opts[:messages]
1117
1190
  return nil unless state.is_a?(Hash) && messages.is_a?(Array)
1118
1191
  return nil unless defined?(TaskSummarizer) && TaskSummarizer.enabled?
1192
+ return nil if respond_to?(:needs_host_work?) && !needs_host_work?(request: opts[:request] || state[:original_request] || state[:request])
1193
+ return nil unless TaskSummarizer.plan_open?(state: state, messages: messages)
1119
1194
 
1195
+ req = opts[:request]
1196
+ req = state[:original_request] || state[:request] if req.to_s.strip.empty? && state.is_a?(Hash)
1120
1197
  text =
1121
- (TaskSummarizer.active_task_prompt(state: state, force: opts[:force]) if TaskSummarizer.respond_to?(:active_task_prompt))
1198
+ (TaskSummarizer.active_task_prompt(state: state, force: opts[:force], request: req) if TaskSummarizer.respond_to?(:active_task_prompt))
1122
1199
  return nil if text.to_s.strip.empty?
1123
1200
 
1124
1201
  messages << { role: 'user', content: text }
@@ -1335,55 +1412,6 @@ module PWN
1335
1412
  :act
1336
1413
  end
1337
1414
 
1338
- # Top-level request kind for task planning (statement | question | autonomous_goal).
1339
- # Single source of truth: TaskSummarizer.request_kind (LLM + heuristics).
1340
- # Mirrors intent/heuristics only when TaskSummarizer is unavailable.
1341
- #
1342
- # Supported Method Parameters::
1343
- # kind = PWN::AI::Agent::Loop.request_kind(
1344
- # request: 'required - user text',
1345
- # kind: 'optional - precomputed',
1346
- # llm_kind: 'optional - injected LLM label',
1347
- # heuristic_only: 'optional - skip LLM'
1348
- # )
1349
- public_class_method def self.request_kind(opts = {})
1350
- req = opts[:request].to_s
1351
- if defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:request_kind)
1352
- return TaskSummarizer.request_kind(
1353
- request: req,
1354
- kind: opts[:kind],
1355
- llm_kind: opts[:llm_kind],
1356
- heuristic_only: opts[:heuristic_only]
1357
- )
1358
- end
1359
-
1360
- case request_intent(request: req)
1361
- when :greeting, :empty
1362
- :statement
1363
- when :howto, :recall
1364
- :question
1365
- when :recon_act
1366
- :autonomous_goal
1367
- else
1368
- # :act — distinguish bare questions from work the agent must do.
1369
- # Host-local facts need tools → autonomous_goal.
1370
- if defined?(TaskSummarizer) && TaskSummarizer.const_defined?(:NEEDS_LOCAL_EVIDENCE_RX)
1371
- return :autonomous_goal if req.match?(TaskSummarizer::NEEDS_LOCAL_EVIDENCE_RX)
1372
- elsif req.match?(/\b(?:hostname|whoami|\bcwd\b|\bpwd\b|my\s+ip)\b/i)
1373
- return :autonomous_goal
1374
- end
1375
- return :question if req.match?(/\?\s*\z/) && !req.match?(
1376
- /\b(please|implement|fix|patch|refactor|run|scan|find|write|change)\b/i
1377
- )
1378
- return :question if req.match?(/\A\s*(?:what|why|when|where|who|which|how)\b/i) &&
1379
- !req.match?(/\b(please|implement|fix|patch|run|scan)\b/i)
1380
-
1381
- :autonomous_goal
1382
- end
1383
- rescue StandardError
1384
- :autonomous_goal
1385
- end
1386
-
1387
1415
  public_class_method def self.recon_authorized?(opts = {})
1388
1416
  req = opts[:request].to_s
1389
1417
  return true if req.match?(AUTH_SCOPE_RX)
@@ -1404,9 +1432,46 @@ module PWN
1404
1432
  private_class_method def self.answer_statement(opts = {})
1405
1433
  request = opts[:request].to_s
1406
1434
  session_id = opts[:session_id]
1407
- txt = <<~ACK.strip
1408
- Noted. No multi-step task breakdown for a general statement — ready when you have a question or a goal to accomplish.
1409
- ACK
1435
+ system_role_content = opts[:system_role_content].to_s
1436
+ engine = active_engine
1437
+ mod_name = ENGINE_MODS[engine]
1438
+ raise "ERROR: Unsupported AI engine for agent loop: #{engine}" unless mod_name
1439
+
1440
+ mod = Object.const_get(mod_name)
1441
+ q_sys = <<~SYS
1442
+ #{system_role_content}
1443
+
1444
+ INTENT: STATEMENT (this turn only)
1445
+ The user is making a statement — not an autonomous multi-step goal.
1446
+ Respond concisely in plain US English. Do NOT call tools unless a
1447
+ single factual lookup is strictly required and already present in
1448
+ context. Do NOT plan multi-step work. Do NOT invent task traces,
1449
+ planner monologue, rubocop, rake, or live recon.
1450
+ SYS
1451
+
1452
+ txt =
1453
+ if mod.respond_to?(:chat)
1454
+ r = mod.chat(
1455
+ request: request,
1456
+ system_role_content: q_sys,
1457
+ spinner: true
1458
+ )
1459
+ if r.is_a?(Hash)
1460
+ (r.dig(:choices, -1, :content) || r.dig(:choices, -1, :text) || r[:content]).to_s
1461
+ else
1462
+ r.to_s
1463
+ end
1464
+ else
1465
+ messages = [
1466
+ { role: 'system', content: q_sys },
1467
+ { role: 'user', content: request }
1468
+ ]
1469
+ msg = call_engine(messages: messages, tools: nil)
1470
+ msg.is_a?(Hash) ? msg[:content].to_s : msg.to_s
1471
+ end
1472
+
1473
+ txt = txt.to_s.strip
1474
+ txt = 'I do not have enough context to respond to your statement.' if txt.empty?
1410
1475
 
1411
1476
  append_session(session_id: session_id, role: 'user', content: request)
1412
1477
  append_session(session_id: session_id, role: 'assistant', content: txt)
@@ -1415,7 +1480,7 @@ module PWN
1415
1480
  session_id: session_id,
1416
1481
  request: request,
1417
1482
  final: txt,
1418
- predicted: 0.9,
1483
+ predicted: 0.85,
1419
1484
  plan: [],
1420
1485
  ts_state: nil
1421
1486
  )
@@ -1876,73 +1941,76 @@ module PWN
1876
1941
  session_id = opts[:session_id]
1877
1942
  on_tool = opts[:on_tool]
1878
1943
  TurnFinalizer.enter_user_path! if defined?(TurnFinalizer)
1879
- # Live coalesced "what am I doing" lines for the TUI (not a model tool).
1880
- ts_state = (TaskSummarizer.fresh(request: request) if defined?(TaskSummarizer) && TaskSummarizer.enabled? && Thread.current[:pwn_reflect_depth].to_i.zero?)
1881
1944
  engine = active_engine
1882
1945
  local = local_engine?(engine: engine)
1883
- system_role_content = opts[:system_role_content] ||= PWN::AI::Agent::PromptBuilder.build(session_id: session_id, request: request)
1884
-
1885
- Registry.discover
1886
- maybe_refresh_extro_snapshot!
1887
- opts[:enabled_toolsets] = default_interactive_toolsets(request: request) unless opts.key?(:enabled_toolsets)
1888
- expose_current_session(session_id: session_id)
1889
- Mistakes.check_user_correction(request: request, session_id: session_id) if defined?(Mistakes)
1890
1946
 
1947
+ # Cheap intent/kind FIRST - before PromptBuilder / Registry / TaskSummarizer
1948
+ # so greetings, FYIs, how-tos, recall, and simple Qs never pay the fat path.
1891
1949
  intent = request_intent(request: request)
1892
- kind = request_kind(request: request)
1893
1950
  Thread.current[:pwn_request_intent] = intent
1894
- Thread.current[:pwn_request_kind] = kind
1895
1951
  Thread.current[:pwn_recon_authorized] = recon_authorized?(request: request)
1896
1952
  Thread.current[:pwn_extinguished] = {}
1897
- # Greeting / light smalltalk: deterministic ack — no weather echo, no tools.
1953
+ expose_current_session(session_id: session_id)
1954
+ Mistakes.check_user_correction(request: request, session_id: session_id) if defined?(Mistakes)
1955
+
1956
+ cheap = opts[:force_tools] != true && %i[greeting howto recall].include?(intent)
1957
+
1958
+ # Greeting / light smalltalk: deterministic ack - no weather echo, no tools,
1959
+ # no PromptBuilder, no Registry.
1898
1960
  if intent == :greeting && opts[:force_tools] != true
1899
1961
  return answer_greeting(
1900
1962
  request: request,
1901
1963
  session_id: session_id
1902
1964
  )
1903
1965
  end
1904
- # How-to: never enter plan_first / task recon / tool thrash (ollama/openwebui).
1905
- if intent == :howto && opts[:force_tools] != true
1906
- return answer_howto(
1907
- request: request,
1908
- session_id: session_id,
1909
- system_role_content: system_role_content
1910
- )
1911
- end
1912
- # Pure prior-turn / vague memory recall: one cheap path, no plan_first.
1913
- if intent == :recall && opts[:force_tools] != true
1914
- return answer_recall(
1915
- request: request,
1916
- session_id: session_id,
1917
- system_role_content: system_role_content
1918
- )
1919
- end
1920
- # General statements: acknowledge briefly — no multi-step task plan.
1921
- # Kind is source of truth (LLM+heuristic). Never short-circuit goals.
1922
- if kind.to_sym == :statement && intent != :recon_act && opts[:force_tools] != true
1923
- return answer_statement(
1924
- request: request,
1925
- session_id: session_id
1926
- )
1927
- end
1928
- # Pure questions that are not how-to/recall: concise answer, no multi-step plan.
1929
- # Host-evidence interrogatives classify as autonomous_goal above so they
1930
- # keep tools (e.g. "what is my hostname?"). force_tools bypasses for tests.
1931
- if kind.to_sym == :question && !%i[recon_act].include?(intent) && opts[:force_tools] != true
1932
- return answer_question(
1933
- request: request,
1934
- session_id: session_id,
1935
- system_role_content: system_role_content
1936
- )
1966
+
1967
+ # Thin system prompt only for remaining cheap paths (howto/recall/statement/question).
1968
+ if cheap
1969
+ system_role_content = opts[:system_role_content]
1970
+ if system_role_content.nil? || system_role_content.to_s.empty?
1971
+ system_role_content = PWN::AI::Agent::PromptBuilder.build(
1972
+ session_id: session_id,
1973
+ request: request,
1974
+ thin: true
1975
+ )
1976
+ opts[:system_role_content] = system_role_content
1977
+ end
1978
+ # How-to: never enter plan_first / task recon / tool thrash.
1979
+ if intent == :howto
1980
+ return answer_howto(
1981
+ request: request,
1982
+ session_id: session_id,
1983
+ system_role_content: system_role_content
1984
+ )
1985
+ end
1986
+ # Pure prior-turn / vague memory recall: one cheap path, no plan_first.
1987
+ if intent == :recall
1988
+ return answer_recall(
1989
+ request: request,
1990
+ session_id: session_id,
1991
+ system_role_content: system_role_content
1992
+ )
1993
+ end
1937
1994
  end
1938
1995
 
1996
+ # --- act / recon / autonomous_goal: full context + tools ---
1997
+ # Reuse precomputed kind so TaskSummarizer.fresh does not classify twice.
1998
+ ts_state = (TaskSummarizer.fresh(request: request) if defined?(TaskSummarizer) && TaskSummarizer.enabled? && Thread.current[:pwn_reflect_depth].to_i.zero?)
1999
+ system_role_content = opts[:system_role_content] ||= PWN::AI::Agent::PromptBuilder.build(
2000
+ session_id: session_id,
2001
+ request: request
2002
+ )
2003
+
2004
+ Registry.discover
2005
+ maybe_refresh_extro_snapshot!
2006
+ opts[:enabled_toolsets] = default_interactive_toolsets(request: request) unless opts.key?(:enabled_toolsets)
2007
+
1939
2008
  # R5 — open the live MDP episode BEFORE the first Registry.rank so
1940
2009
  # Q(s,a) can advise this turn. Planning still owns the task list.
1941
2010
  if defined?(PWN::AI::Agent::Policy) && Policy.respond_to?(:begin_episode)
1942
2011
  Policy.begin_episode(
1943
2012
  session_id: session_id,
1944
2013
  request: request,
1945
- kind: kind,
1946
2014
  intent: intent,
1947
2015
  engine: engine,
1948
2016
  ts_state: ts_state
@@ -1953,53 +2021,43 @@ module PWN
1953
2021
  # TaskSummarizer.emit_plan! we re-rank using English tangible tasks
1954
2022
  # so generated tasks — not the bare request — drive which tools
1955
2023
  # the model may call.
1956
- tools = Registry.definitions(enabled: opts[:enabled_toolsets], relevance: request)
2024
+ # CORE_TOOLS is the default action space. Extra schemas are
2025
+ # opt-in via enabled_toolsets + core_only: false.
2026
+ core_only = opts.fetch(:core_only, true)
2027
+ tools = Registry.definitions(
2028
+ enabled: opts[:enabled_toolsets],
2029
+ relevance: request,
2030
+ core_only: core_only,
2031
+ intent: intent
2032
+ )
1957
2033
  messages = [{ role: 'system', content: system_role_content }]
1958
2034
  messages.concat(Learning.exemplars_for(request: request)) if local && defined?(Learning) && Learning.respond_to?(:exemplars_for)
1959
2035
  messages << { role: 'user', content: request }
1960
2036
  append_session(session_id: session_id, role: 'user', content: request)
1961
2037
 
1962
- # Tangible-task breakdown ONLY for autonomous goals.
1963
- # General statements and questions stay without multi-step plans.
1964
- needs_breakdown =
1965
- if defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:needs_task_breakdown?)
1966
- TaskSummarizer.needs_task_breakdown?(kind: kind, request: request)
1967
- else
1968
- kind.to_sym == :autonomous_goal
1969
- end
1970
- ts_state[:request_kind] = kind if ts_state.is_a?(Hash)
1971
- if needs_breakdown
1972
- task_summary_plan!(state: ts_state, request: request, on_tool: on_tool)
1973
- elsif ts_state.is_a?(Hash) && defined?(TaskSummarizer)
1974
- # Record kind on state; optional one-line kind banner (no task list).
1975
- ts_state[:plan] = []
1976
- ts_state[:request_kind] = kind
1977
- if TaskSummarizer.respond_to?(:format_plan)
1978
- banner = TaskSummarizer.format_plan(tasks: [], request: request, request_kind: kind)
1979
- if banner && !banner.to_s.empty?
1980
- ts_state[:plan_text] = banner
1981
- ts_state[:plan_emitted] = true
1982
- emit_task_summary(line: banner, on_tool: on_tool)
1983
- end
1984
- end
1985
- end
2038
+ # Every request gets a task compass.
2039
+ task_summary_plan!(state: ts_state, request: request, on_tool: on_tool) if defined?(TaskSummarizer)
1986
2040
  # Re-bind tools from English plan so task list is the sole driver of
1987
2041
  # tool exposure/ranking (Registry keyword router + CORE).
1988
2042
  if ts_state.is_a?(Hash) && defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:relevance_query)
1989
2043
  rq = TaskSummarizer.relevance_query(state: ts_state, request: request)
1990
- tools = Registry.definitions(enabled: opts[:enabled_toolsets], relevance: rq) unless rq.to_s.strip.empty?
2044
+ unless rq.to_s.strip.empty?
2045
+ tools = Registry.definitions(
2046
+ enabled: opts[:enabled_toolsets],
2047
+ relevance: rq,
2048
+ core_only: core_only,
2049
+ intent: intent
2050
+ )
2051
+ end
1991
2052
  end
1992
2053
  # English-task-as-primary: inject tangible tasks only for autonomous goals.
1993
- inject_task_focus!(messages: messages, state: ts_state, force: true) if needs_breakdown
1994
-
2054
+ inject_task_focus!(messages: messages, state: ts_state, force: true, request: request)
1995
2055
  predicted = nil
1996
2056
  Thread.current[:pwn_plan_predicted] = nil
1997
2057
  cal_state = calibration_state
1998
2058
  force_plan = cal_state[:force_plan]
1999
- # Skip plan_first for non-goals (statements/questions) and cheap intents.
2000
- skip_plan = %i[howto recall greeting].include?(intent) ||
2001
- %i[statement question].include?(kind.to_sym) ||
2002
- !needs_breakdown
2059
+ # Skip plan_first only for remaining cheap intents.
2060
+ skip_plan = %i[howto recall greeting].include?(intent)
2003
2061
  if !skip_plan && (force_plan || agent_flag(key: :plan_first, default: local) || budget_exhaustion_hot?) && !Array(tools).empty?
2004
2062
  predicted = plan_first(messages: messages, request: request, ts_state: ts_state)
2005
2063
  # P22 — prefer explicit return; fall back to thread stash
@@ -2009,12 +2067,21 @@ module PWN
2009
2067
  # PLAN: tool-call scaffold jargon (unify_plan! refuses that).
2010
2068
  if ts_state.is_a?(Hash) && defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:relevance_query)
2011
2069
  rq = TaskSummarizer.relevance_query(state: ts_state, request: request)
2012
- tools = Registry.definitions(enabled: opts[:enabled_toolsets], relevance: rq) unless rq.to_s.strip.empty?
2070
+ unless rq.to_s.strip.empty?
2071
+ tools = Registry.definitions(
2072
+ enabled: opts[:enabled_toolsets],
2073
+ relevance: rq,
2074
+ core_only: core_only,
2075
+ intent: intent
2076
+ )
2077
+ end
2013
2078
  end
2014
- inject_task_focus!(messages: messages, state: ts_state, force: true)
2079
+ inject_task_focus!(messages: messages, state: ts_state, force: true, request: request)
2015
2080
  end
2016
2081
  if budget_exhaustion_hot?
2017
- hot_hint = if local_engine?
2082
+ english_open = defined?(TaskSummarizer) && TaskSummarizer.respond_to?(:plan_open?) &&
2083
+ TaskSummarizer.plan_open?(state: ts_state, messages: messages)
2084
+ hot_hint = if local_engine? && !english_open
2018
2085
  '[pwn-ai/p17] Budget-exhaustion is the top open failure on this host. ' \
2019
2086
  'Prefer the SHORTEST plan that finishes the ask (≤3 tool calls). ' \
2020
2087
  'Emit a final answer as soon as you have evidence — do not explore.'
@@ -2046,7 +2113,7 @@ module PWN
2046
2113
  compact_history!(messages: messages) if local
2047
2114
  # English-task-as-primary: when plan_idx advanced, tell the model
2048
2115
  # which plain-English task is active before the next tool batch.
2049
- inject_task_focus!(messages: messages, state: ts_state)
2116
+ inject_task_focus!(messages: messages, state: ts_state, request: request)
2050
2117
 
2051
2118
  # P17 — on the final iteration, strip tools and demand a plain-text
2052
2119
  # answer. Without this the model happily emits one more tool_calls
@@ -2079,15 +2146,9 @@ module PWN
2079
2146
  plan_faithful = hot && plan_steps.positive? && plan_steps <= plan_step_limit &&
2080
2147
  turn_fails['empty_final'].to_i.zero? &&
2081
2148
  turn_fails.values.sum < 2
2082
- text_only_iters = if hot
2083
- if plan_faithful
2084
- 1
2085
- else
2086
- (local_engine? ? 3 : 2)
2087
- end
2088
- else
2089
- 1
2090
- end
2149
+ # Last-iter strips tools only on the true last slot. English
2150
+ # leftovers and budget-hot must not steal runway from a live goal.
2151
+ text_only_iters = 1
2091
2152
  last_iter = (i >= max_iters - text_only_iters)
2092
2153
  if last_iter
2093
2154
  tag = i >= max_iters - 1 ? 'FINAL ITERATION' : 'PENULTIMATE — wrap up'
@@ -2101,7 +2162,7 @@ module PWN
2101
2162
  }
2102
2163
  end
2103
2164
 
2104
- msg = call_engine(messages: messages, tools: last_iter ? nil : tools)
2165
+ msg = call_engine(messages: messages, tools: last_iter ? nil : tools, ts_state: ts_state)
2105
2166
  if msg.nil?
2106
2167
  task_summary_flush!(state: ts_state, on_tool: on_tool)
2107
2168
  return '[pwn-ai] engine returned no message'
@@ -2158,6 +2219,21 @@ module PWN
2158
2219
  }
2159
2220
  next
2160
2221
  end
2222
+ if request_unsatisfied?(
2223
+ request: request,
2224
+ messages: messages,
2225
+ last_iter: last_iter
2226
+ ) && turn_fails['unsatisfied'].to_i < 4
2227
+ turn_fails['unsatisfied'] += 1
2228
+ warn "[pwn-ai/loop] original request not evidenced on iter=#{i}; continuing"
2229
+ messages << {
2230
+ role: 'user',
2231
+ content: '[pwn-ai] The original request is not evidenced yet. ' \
2232
+ 'Keep calling CORE_TOOLS (shell, pwn_eval) until that request is ' \
2233
+ 'done or truly blocked. Do not declare completion from a listing alone.'
2234
+ }
2235
+ next
2236
+ end
2161
2237
  append_session(session_id: session_id, role: 'assistant', content: text)
2162
2238
  Learning.auto_introspect(session_id: session_id, request: request, final: text, predicted: predicted, plan: ts_state && ts_state[:plan], ts_state: ts_state) if defined?(Learning) && should_auto_introspect?(local: local, turn_fails: turn_fails, iter: i)
2163
2239
  maybe_finish_policy(session_id: session_id, proxy_ok: true, ts_state: ts_state)