pwn 0.5.743 → 0.5.745

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. checksums.yaml +4 -4
  2. data/Gemfile +1 -1
  3. data/Rakefile +4 -0
  4. data/documentation/Agent-Tool-Registry.md +2 -2
  5. data/documentation/CLI-Drivers.md +10 -0
  6. data/documentation/How-PWN-Works.md +3 -1
  7. data/documentation/Installation.md +2 -0
  8. data/documentation/Policy-Benchmark.md +76 -0
  9. data/documentation/Recon-Findings-API.md +4 -2
  10. data/documentation/Reinforcement-Learning.md +10 -2
  11. data/documentation/Reporting.md +8 -0
  12. data/documentation/Skills-Memory-Learning.md +15 -4
  13. data/documentation/diagrams/dot/pwn-ai-feedback-learning-loop.dot +2 -1
  14. data/documentation/diagrams/dot/reinforcement-learning.dot +3 -1
  15. data/documentation/diagrams/dot/reporting-pipeline.dot +6 -0
  16. data/documentation/diagrams/dot/zero-day-research-flow.dot +7 -2
  17. data/documentation/diagrams/pwn-ai-feedback-learning-loop.svg +406 -398
  18. data/documentation/diagrams/reinforcement-learning.svg +240 -225
  19. data/documentation/diagrams/reporting-pipeline.svg +109 -65
  20. data/documentation/diagrams/zero-day-research-flow.svg +115 -80
  21. data/documentation/pwn-ai-Agent.md +2 -0
  22. data/etc/default_skills/humanizer/LICENSE +21 -0
  23. data/etc/default_skills/humanizer/SKILL.md +650 -0
  24. data/etc/default_skills/pwn/ai/agent/learning/SKILL.md +1 -0
  25. data/etc/default_skills/pwn/ai/agent/metrics/SKILL.md +5 -0
  26. data/etc/default_skills/pwn/ai/agent/mission/SKILL.md +77 -0
  27. data/etc/default_skills/pwn/ai/agent/skill_review/SKILL.md +49 -0
  28. data/etc/default_skills/pwn/plugins/recon/SKILL.md +4 -0
  29. data/lib/pwn/ai/agent/confirmation.rb +34 -1
  30. data/lib/pwn/ai/agent/dispatch.rb +4 -1
  31. data/lib/pwn/ai/agent/learning.rb +40 -1
  32. data/lib/pwn/ai/agent/loop.rb +57 -12
  33. data/lib/pwn/ai/agent/metrics.rb +126 -1
  34. data/lib/pwn/ai/agent/mission.rb +455 -0
  35. data/lib/pwn/ai/agent/open_goal.rb +15 -1
  36. data/lib/pwn/ai/agent/prompt_builder.rb +12 -1
  37. data/lib/pwn/ai/agent/skill_review.rb +223 -0
  38. data/lib/pwn/ai/agent/task_dag.rb +38 -1
  39. data/lib/pwn/ai/agent/tools/fuzz_campaign.rb +5 -1
  40. data/lib/pwn/ai/agent/tools/job_run.rb +5 -1
  41. data/lib/pwn/ai/agent/tools/nuclei_scan.rb +1 -1
  42. data/lib/pwn/ai/agent.rb +2 -0
  43. data/lib/pwn/ai/cli.rb +83 -7
  44. data/lib/pwn/config.rb +2 -0
  45. data/lib/pwn/migrate.rb +5 -1
  46. data/lib/pwn/plugins/exploit_dev.rb +9 -2
  47. data/lib/pwn/plugins/findings.rb +113 -33
  48. data/lib/pwn/plugins/fuzz.rb +7 -2
  49. data/lib/pwn/plugins/nmap_it.rb +25 -1
  50. data/lib/pwn/plugins/nuclei.rb +35 -15
  51. data/lib/pwn/plugins/recon.rb +88 -0
  52. data/lib/pwn/plugins/repl/mesh.rb +1 -1
  53. data/lib/pwn/plugins/sbom.rb +14 -5
  54. data/lib/pwn/plugins/vault.rb +12 -0
  55. data/lib/pwn/reports.rb +15 -0
  56. data/lib/pwn/version.rb +1 -1
  57. data/spec/integration/reports_spec.rb +3 -3
  58. data/spec/lib/pwn/ai/agent/confirmation_spec.rb +20 -0
  59. data/spec/lib/pwn/ai/agent/dispatch_spec.rb +11 -0
  60. data/spec/lib/pwn/ai/agent/mission_spec.rb +119 -0
  61. data/spec/lib/pwn/ai/agent/open_goal_spec.rb +7 -0
  62. data/spec/lib/pwn/ai/agent/rsi_metrics_spec.rb +49 -0
  63. data/spec/lib/pwn/ai/agent/skill_review_spec.rb +127 -0
  64. data/spec/lib/pwn/ai/agent/tools/nuclei_scan_spec.rb +7 -4
  65. data/spec/lib/pwn/ai/cli_spec.rb +143 -0
  66. data/spec/lib/pwn/migrate_spec.rb +29 -2
  67. data/spec/lib/pwn/plugins/findings_spec.rb +3 -3
  68. data/spec/lib/pwn/plugins/findings_structured_spec.rb +24 -28
  69. data/spec/lib/pwn/plugins/nuclei_spec.rb +5 -3
  70. data/spec/lib/pwn/plugins/research_evidence_spec.rb +88 -0
  71. data/spec/lib/pwn/reports_spec.rb +3 -3
  72. data/spec/spec_helper.rb +11 -0
  73. data/third_party/pwn_rdoc.jsonl +77 -1
  74. metadata +13 -3
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 8089c1c616fe3d19f70170b55204b0dd762a14faa40432e6acf421bd6d55197d
4
- data.tar.gz: 6633cdc11ab8b811913d6b6e30608a79fa85c06c5bde599d4155c9c63d5a3884
3
+ metadata.gz: f3d60fd604a668b531459517b97dfee6448f9485679bd4c25bfe3ec12090cee9
4
+ data.tar.gz: 64a4dd65248017933e877c03db222ce61741e60044a03a1c9c9a1297ca7a4fde
5
5
  SHA512:
6
- metadata.gz: 83c1b09ae27724b7e5df7972f018f11b7bee7f77e38d295c45395bf4c828f6c57e9736306687ae243e1fc05dc4dd63e4b33407d4101b916c2e32a8b241f39958
7
- data.tar.gz: 19ad087ec133a1194b2fd60d7348edff887b94940647b5a0e34792fa5084742dcb7a04f213ee17b404e4fffda5fc1b8db4821fd1bc6c047c0867c96a2752d8af
6
+ metadata.gz: 1792ff5aea6a7c406248ab0fd6647732c490e56450afdf768840f340b4d7b4847a2d2cd996ff1349b3ca1a1bbfc05be40310225345e675937cc16022aaad5ec3
7
+ data.tar.gz: 7155e5d3cfca02bda195d0b0f926b85d6cd95b306b570d5956484ae237d360b36628a979e2d98a2c5ce3008343c7cecc1d14f12d0a986dbc577923a9f7905930
data/Gemfile CHANGED
@@ -49,7 +49,7 @@ gem 'jwt', '3.3.0'
49
49
  gem 'libusb', '0.8.0'
50
50
  gem 'luhn', '3.0.0'
51
51
  gem 'mail', '2.9.1'
52
- gem 'mcp', '1.5.1'
52
+ gem 'mcp', '1.6.0'
53
53
  gem 'meshtastic', '0.0.184'
54
54
  gem 'metasm', '1.0.6'
55
55
  gem 'mongo', '2.26.0'
data/Rakefile CHANGED
@@ -11,6 +11,10 @@ require 'rdoc/task'
11
11
  require 'rubocop/rake_task'
12
12
 
13
13
  RSpec::Core::RakeTask.new(:spec)
14
+ task :silence_rspec_filter_banner do
15
+ ENV['PWN_RAKE_SPEC'] = '1'
16
+ end
17
+ Rake::Task[:spec].enhance([:silence_rspec_filter_banner])
14
18
 
15
19
  RuboCop::RakeTask.new do |rubocop|
16
20
  config_file = '.rubocop.yml'
@@ -12,7 +12,7 @@ toolsets; the JSON-Schema for each tool is what the model actually sees.
12
12
  |---|---|---|
13
13
  | `http` | `http_proxy_start` · `http_proxy_stop` · `http_proxy_entries` · `http_proxy_rules` · `http_replay` | `PWN::Plugins::MitmProxy`: native HTTP HAR capture/replay; opaque CONNECT tunnels |
14
14
  | `terminal` | `shell` · `job_run` · `job_status` · `job_tail` · `job_result` · `job_kill` | Foreground host commands plus `PWN::Plugins::Jobs` durable supervisors |
15
- | `pwn` | `pwn_eval` · `sbom_scan` · `r2_functions` · `r2_disasm` · `gdb_run_to_crash` · `nmap_scan` · `nuclei_scan` · `browser_goto` · `loot_query` | `TOPLEVEL_BINDING.eval` in the live REPL process, after `ToolGuard`; SBOM engines via `PWN::Plugins::SBOM`; r2pipe JSON via `PWN::Plugins::Radare2`; gdb MI3 crash reports via `PWN::Plugins::GDBMI`; nmap XML ingest/diff via `PWN::Plugins::NmapIt`; nuclei/httpx JSONL findings via `PWN::Plugins::Nuclei`; TransparentBrowser navigation captures screenshot/DOM/HAR via `browser_goto`; engagement loot via `PWN::Plugins::Vault` |
15
+ | `pwn` | `pwn_eval` · `sbom_scan` · `r2_functions` · `r2_disasm` · `gdb_run_to_crash` · `nmap_scan` · `nuclei_scan` · `browser_goto` · `loot_query` | `TOPLEVEL_BINDING.eval` in the live REPL process, after `ToolGuard`; SBOM engines via `PWN::Plugins::SBOM` store CVE leads as recon observations, not findings; r2pipe JSON via `PWN::Plugins::Radare2`; gdb MI3 crash reports via `PWN::Plugins::GDBMI`; nmap XML ingest/diff via `PWN::Plugins::NmapIt`, which skips a known port unless `refresh: true`; nuclei/httpx JSONL via `PWN::Plugins::Nuclei` writes a recon observation (a template match is not a finding) and accepts a `Recon.handoff` record; TransparentBrowser navigation captures screenshot/DOM/HAR via `browser_goto`; engagement loot via `PWN::Plugins::Vault` |
16
16
  | `mcp` | `mcp` | `PWN::AI::MCP` session broker → any `PWN::AI::MCP::*` stdio client |
17
17
  | `memory` | `memory_remember` · `memory_recall` · `memory_forget` · `memory_clear` · **`memory_lean`** | `PWN::Memory` → `~/.pwn/memory.json` |
18
18
  | `skills` | `skills_consolidate` · **`skills_recall`** · `skill_list` · `skill_view` · `skill_create` · `skill_add_reference` · `skill_delete` · `skill_migrate_legacy` | `~/.pwn/skills/<name>/SKILL.md` (**[agentskills.io](https://agentskills.io) spec**; legacy flat `*.md` auto-migrated) |
@@ -20,7 +20,7 @@ toolsets; the JSON-Schema for each tool is what the model actually sees.
20
20
  | `learning` | `learning_note_outcome` · `learning_reflect` · `learning_distill_skill` · `learning_stats` · `learning_outcomes` · `learning_consolidate` · `learning_reset` · `learning_auto_introspect_toggle` · **`learning_gc_stores`** · **`learning_purge_noise`** · **`mistakes_list`** · **`mistakes_record`** · **`mistakes_resolve`** · **`mistakes_reset`** · **`mistakes_lean`** · **`reward_judge`** · **`reward_prm`** · **`reward_sentinel`** · **`reward_preferences`** · **`reward_export_dpo`** · **`reward_warm_sentinel`** · **`reward_scrub_preferences`** · **`reward_preference_balance`** · **`curriculum_practice`** · **`curriculum_train`** · **`curriculum_hindsight`** · **`curriculum_offline_judge`** · **`curriculum_preference_balance`** | `PWN::AI::Agent::Learning` + `Mistakes` + `Reward` + `Curriculum` → `~/.pwn/learning.jsonl` + `~/.pwn/mistakes.json` + `~/.pwn/preferences.jsonl` + `~/.pwn/curriculum/` + `~/.pwn/finetune/` |
21
21
  | `reward` | **`reward_generator_mix`** | `PWN::AI::Agent::Reward.generator_mix` → online preference source-mix controller (`preferences.jsonl`) |
22
22
  | `curriculum` | **`curriculum_practice_kpi`** | `PWN::AI::Agent::Curriculum.practice_kpi` → `~/.pwn/curriculum_kpi.jsonl` |
23
- | `metrics` | `metrics_summary` · `metrics_reset` | `PWN::AI::Agent::Metrics` → `~/.pwn/metrics.json` |
23
+ | `metrics` | `metrics_summary` · `metrics_reset` | `PWN::AI::Agent::Metrics` → `~/.pwn/metrics.json`. ESR is `verified_exploit_tools / vulnerable_tools`. ASR is `successful_attacks / total_attack_attempts`. |
24
24
  | `policy` | **`policy_stats`** · **`policy_evaluate`** · **`policy_recommend`** | `PWN::AI::Agent::Policy` → `~/.pwn/policy.json` + `~/.pwn/policy_traj.jsonl` |
25
25
  | `extrospection` | `extro_snapshot` · `extro_drift` · `extro_observe` · `extro_observations` · `extro_intel` · **`extro_watch`** · **`extro_verify`** · **`extro_rf_tune`** · **`extro_osint`** · **`extro_serial`** · **`extro_telecomm`** · **`extro_packet`** · **`extro_vision`** · **`extro_voice`** · `extro_correlate` · `extro_stats` · `extro_reset` · `extro_auto_toggle` | `PWN::AI::Agent::Extrospection` (+ Serial/Packet/OCR/Voice/BareSIP/TransparentBrowser/GQRX) → `~/.pwn/extrospection.json` |
26
26
  | `cron` | `cron_list` · `cron_create` · `cron_run` · `cron_enable` · `cron_disable` · `cron_remove` | `PWN::Cron` → `~/.pwn/cron/jobs.yml` |
@@ -59,6 +59,16 @@ pwn_setup --migrate --fix # ~/.pwn state doctor + autofix (PWN::Migrate)
59
59
  See [Installation](Installation.md) for the full profile table, the
60
60
  `PWN::Setup` API and the `PWN::Migrate` state-file registry.
61
61
 
62
+ ## Offline policy evaluation with `pwn-ai`
63
+
64
+ `pwn-ai --policy evaluate --baseline PATH --candidate PATH` prints two frozen
65
+ held-out reports without opening a vault or starting an AI session. The separate
66
+ `--policy promote` and `--policy rollback` actions require an explicit target and
67
+ both `--approve-policy-change` and `--policy-writers-stopped`. They never start
68
+ network tasks or automatically promote on success. See
69
+ [Policy-Benchmark](Policy-Benchmark.md#offline-operator-cli) for runnable local
70
+ commands, receipt preservation, fresh replay gates, and limitations.
71
+
62
72
  ## Typical CI usage
63
73
 
64
74
  ```yaml
@@ -77,7 +77,9 @@ POLICY · EXTROSPECTION · RECENT TURNS) are re-injected into the next system
77
77
  prompt.
78
78
  Nightly hygiene cron trims `~/.pwn` stores. Curriculum practice, offline
79
79
  judge, and weekly LoRA train ship seeded but disabled; turn them on with
80
- `cron_enable` when you want that loop:
80
+ `cron_enable` when you want that loop.
81
+
82
+ `Metrics.esr` is `verified_exploit_tools / vulnerable_tools`. `Metrics.asr` is `successful_attacks / total_attack_attempts`. `Learning.rsi_tick` compares ESR with the previous snapshot and writes a lesson when it falls. A file hash, a nuclei template match, or a CVE id does not move ESR. Only a reproduced exploit does.
81
83
 
82
84
  ![Self-improvement loop](diagrams/pwn-ai-feedback-learning-loop.svg)
83
85
 
@@ -187,6 +187,7 @@ What `--migrate` does:
187
187
  directory layout) → **deep-merge** any keys the current
188
188
  `PWN::Config.env_template` added into your encrypted `~/.pwn/pwn.yaml`
189
189
  **without overwriting your values** (re-encrypted with the same key/IV).
190
+ Schema 5 adds `ai.agent.skill_review` (`recommend` when the key is absent).
190
191
 
191
192
  Everything is idempotent and dry-run capable. The plain `pwn` launcher also
192
193
  prints a one-line drift warning on startup whenever `~/.pwn/.schema`
@@ -268,6 +269,7 @@ The first `pwn` launch creates `~/.pwn/` and an **encrypted**
268
269
  - `cwe`
269
270
  - `capec`
270
271
  - `att&ck`
272
+ - `humanizer`
271
273
 
272
274
  Source lives in the gem at `etc/default_skills/`. `PWN::Config.install_default_skills`
273
275
  walks every `SKILL.md` recursively (idempotent SOP copies; never overwrites an
@@ -220,6 +220,82 @@ fields except `elapsed_seconds`. They contain no wall-clock timestamps, random
220
220
  IDs or temporary paths. The original training report still contains the timing
221
221
  and metadata variability described above.
222
222
 
223
+ ## Offline operator CLI
224
+
225
+ `pwn-ai --policy evaluate|promote|rollback` reuses `PolicyEvaluation`; it never
226
+ starts a session, loads the encrypted configuration, runs `Learning.rsi_tick`,
227
+ or invokes the agent loop. It cannot be combined with `--ai`, replay, mission,
228
+ or other session options. Evaluation accepts existing explicit frozen snapshots,
229
+ runs suites 0 and 1, and prints a JSON **array** suitable for `--reports`. It does
230
+ not train or write the live policy. The Ruby API supports other indices 0..7.
231
+
232
+ From a source checkout, this disposable demonstration stays entirely under a
233
+ new `/tmp` directory (use `pwn-ai` instead of `ruby -Ilib bin/pwn-ai` for an
234
+ installed executable):
235
+
236
+ ```sh
237
+ umask 077
238
+ work=$(mktemp -d /tmp/pwn-offline-policy.XXXXXX)
239
+ ruby scripts/benchmark_policy.rb --heldout \
240
+ --snapshot-dir "$work/snapshots" --output "$work/benchmark.json"
241
+ ruby -Ilib bin/pwn-ai --policy evaluate \
242
+ --baseline "$work/snapshots/off.json" --candidate "$work/snapshots/on.json" \
243
+ > "$work/reports.json"
244
+ cp "$work/snapshots/off.json" "$work/demo-live.json"
245
+ ```
246
+
247
+ Review `reports.json` before deciding whether to proceed. For a real target,
248
+ first stop **all** agent processes and other policy writers, take the baseline
249
+ copy only after stopping them, and keep writers stopped through promotion and
250
+ readback. The acknowledgement does not stop processes or acquire a writer lock.
251
+ Concurrent promotions are also unsupported. Do not replace a real policy with
252
+ these demonstration snapshots or infer live gains from this fixture experiment.
253
+
254
+ Only if approving the change to the **disposable demo target**, run:
255
+
256
+ ```sh
257
+ ruby -Ilib bin/pwn-ai --policy promote \
258
+ --baseline "$work/snapshots/off.json" --candidate "$work/snapshots/on.json" \
259
+ --reports "$work/reports.json" --live-policy "$work/demo-live.json" \
260
+ --approve-policy-change --policy-writers-stopped > "$work/promotion.json"
261
+ cmp "$work/demo-live.json" "$work/snapshots/on.json"
262
+ ```
263
+
264
+ The command re-executes both reports before considering replacement. It returns
265
+ nonzero on refusal or malformed input; check the exit status and JSON, not just
266
+ whether a redirected file exists. Keep the successful `promotion.json` receipt
267
+ and digest-named backup. Use distinct, new output paths: shell redirection opens
268
+ files before the CLI starts, so never redirect onto snapshots, the live target,
269
+ or an existing receipt. A failed retry must not overwrite the successful receipt.
270
+
271
+ Rollback is a separate explicit operator decision with the same stopped-writer
272
+ requirement; it refuses intervening changes to the target:
273
+
274
+ ```sh
275
+ ruby -Ilib bin/pwn-ai --policy rollback \
276
+ --receipt "$work/promotion.json" --live-policy "$work/demo-live.json" \
277
+ --approve-policy-change --policy-writers-stopped > "$work/rollback.json"
278
+ cmp "$work/demo-live.json" "$work/snapshots/off.json"
279
+ ```
280
+
281
+ Neither approval flag alone is sufficient. There is no default live path,
282
+ automatic promotion, target discovery, network task, or model training in this
283
+ CLI path. Reports and receipts are operator-owned local JSON, not executable
284
+ configuration; neither can supply approval flags. Existing `Learning.rsi_tick`
285
+ only snapshots measured rates and records a regression lesson. It does not call
286
+ this gate, generate candidates, schedule practice, or approve changes. The
287
+ broader online learning and curriculum paths remain unchanged and separate.
288
+
289
+ Focused verification commands (no hardware, providers, or live models):
290
+
291
+ ```sh
292
+ bundle exec rspec spec/lib/pwn/ai/cli_spec.rb \
293
+ spec/lib/pwn/ai/agent/policy_evaluation_spec.rb \
294
+ spec/lib/pwn/ai/agent/learning_spec.rb spec/lib/pwn/ai/agent/rsi_metrics_spec.rb
295
+ bundle exec rubocop lib/pwn/ai/cli.rb spec/lib/pwn/ai/cli_spec.rb \
296
+ spec/lib/pwn/ai/agent/rsi_metrics_spec.rb
297
+ ```
298
+
223
299
  ## Explicit promotion and rollback
224
300
 
225
301
  This is an **operator-invoked local eligibility gate**, not automatic online
@@ -22,9 +22,11 @@ Use `affected_asset: asset[:id]` and `evidence_paths: asset[:evidence_paths]` in
22
22
 
23
23
  `record_structured` requires title, CWE identifier (`CWE-<positive integer>`), complete CVSS 3.0/3.1 base vector and matching numeric base score, affected_asset, nonempty existing readable absolute evidence_paths, nonempty PoC command/code string, attack_chain_refs array, remediation string, and numeric confidence 0..1. Chain refs must already exist in the same engagement. Invalid input raises before appending. CVSS 2/4, temporal/environmental vectors are explicitly unsupported (rejected, not normalized).
24
24
 
25
- Legacy `Findings.record`, query/report, chain, render, and SARIF output remain callable; legacy record is not the strict API. Agent record and chain operations are strict. Existing query/export tool operations remain available. Accepted structured rows mark `verification_status: not_executed`; evidence hashes prove captured bytes, not exploit execution. Severity is derived from the validated score. No automatic escalation for linked findings.
25
+ Legacy `Findings.record`, query/report, chain, render, and SARIF output remain callable; legacy record is not the strict API. Agent record and chain operations are strict. Existing query/export tool operations remain available. Accepted structured rows store `severity: info` and `verification_status: not_executed`. A file hash proves the bytes were captured. It does not prove the PoC ran. The claimed score stays on the row, and the published severity rises only after `verify` returns `reproduced`. Linking findings does not raise severity by itself.
26
26
 
27
- `Findings.verify` / `finding_record op=verify` attests a working PoC from HTTP request/response files or a script execution log. The impact marker must appear in that evidence or the row stays `failed`. `Findings.retest` replays the same path after a fix (`still_open` vs `fixed`). `Findings.chain_impact` may raise combined severity only when a combined-impact file names every finding id. Issue work is unfinished while recorded findings remain `not_executed` (`issue_work_unverified`).
27
+ `Findings.verify` and `Findings.retest` run the stored PoC or the ordered reproduction commands, capture that transcript, and only then search it for the impact string. A file the caller already filled with the impact string does not count. `retest` uses the same runner, so `fixed` means that command no longer prints the impact. `chain` refuses evidence shorter than 40 characters and does not raise `composite_severity` above the parent. `chain_impact` can raise it only when the combined-impact file names every id and shows the second effect. `PWN::Reports` refuses to print `high` or `critical` for a linked pair that is not `reproduced`, or whose combined-impact file does not name every id.
28
+
29
+ Nuclei and SBOM write recon observations, not findings. A template match or a CVE id is a lead. `Recon.handoff` carries host, port, product, version, and the evidence path into `ExploitDev.ret2libc`, `Nuclei.scan`, and `Fuzz.triage`. `NmapIt.scan` and `Nuclei.scan` skip a port that already has an observation unless `refresh: true`.
28
30
 
29
31
  `PWN::AI::Agent::Swarm.ensure_specialists` / `agent_roster` writes recon, authz, injection, xss, and business_logic personas. SARIF export is `Findings.render`; `PWN::Plugins::Github.open_fix_pr` opens a remediation PR (tests stub the GitHub API).
30
32
 
@@ -8,8 +8,16 @@ local adapter.
8
8
  `Curriculum.practice` -> `Reward.export_dpo` -> `Curriculum.train_and_gate`
9
9
 
10
10
  That path can promote a new local adapter when the candidate beats the current
11
- one. Without a trainer it still **exports** the datasets and a manual CLI. Live
12
- improvement does not wait on weights.
11
+ one. Without a trainer it still **exports** the datasets and a manual CLI. Live improvement does not wait on weights.
12
+
13
+ ESR and ASR are the rates RSI actually compares. ESR is `verified_exploit_tools / vulnerable_tools`. ASR is `successful_attacks / total_attack_attempts`. `Learning.rsi_tick` stores the pair and writes an `rsi` lesson when ESR falls. Tool telemetry and the judge score are separate. A passing suite is not either rate.
14
+
15
+ RSI's rate tick does not generate or promote policy candidates. For the separate
16
+ **offline operator evaluation** path, use `pwn-ai --policy evaluate` and review
17
+ the fixed local held-out reports. Promotion and rollback each require explicit
18
+ `--approve-policy-change --policy-writers-stopped` plus a named live policy path;
19
+ neither runs automatically from RSI. See the executable commands, stopped-writer
20
+ requirements, and benchmark limits in [Policy-Benchmark](Policy-Benchmark.md#offline-operator-cli).
13
21
 
14
22
  ![Reinforcement-learning loop](diagrams/reinforcement-learning.svg)
15
23
 
@@ -24,6 +24,14 @@
24
24
  | `PWN::Plugins::JiraDataCenter` | Create issues from findings |
25
25
  | `PWN::Plugins::SlackClient` / `MailAgent` | Notify |
26
26
 
27
+ ## Severity and chain gates
28
+
29
+ `Findings.record` and `record_structured` store `severity: info` and `verification_status: not_executed`. `verify` runs the stored PoC and searches that transcript. The claimed severity is published only when the status is `reproduced`.
30
+
31
+ `PWN::Reports.report_payload` calls `refuse_unproven_combined!` before a chain leaves the process. It refuses to print `high` or `critical` for a linked pair that is not `reproduced`, or whose combined-impact file does not name every id. Max-of-two is not a chain.
32
+
33
+ Nuclei and SBOM leads stay recon observations until a later `verify` reproduces them.
34
+
27
35
  ## Example
28
36
 
29
37
  ```ruby
@@ -1,8 +1,6 @@
1
1
  # Memory · Skills · Learning · Mistakes · Metrics · Policy - Introspection
2
2
 
3
- The **inward-facing** half of the pwn-ai feedback loop: how the agent measures
4
- its own performance, turns wins into permanent capability, and - critically -
5
- **learns from its own mistakes so it does not repeat them**.
3
+ The inward half of the pwn-ai feedback loop: the agent writes what it did, and the next prompt can see it. Wins become skills. Failures become fingerprinted mistakes.
6
4
 
7
5
  ![Memory / Skills detail](diagrams/memory-skills-detailed.svg)
8
6
 
@@ -15,7 +13,17 @@ its own performance, turns wins into permanent capability, and - critically -
15
13
  | **Policy** | `policy.json` + `policy_traj.jsonl` | Loop `begin_episode` / `observe_step` / `finish` | `policy_stats` · `policy_evaluate` · `policy_recommend` | `POLICY` block - live tabular Q / REINFORCE. Advisory rank only. Disable with `ai.agent.policy: false`. |
16
14
  | **Learning** | `learning.jsonl` | `learning_note_outcome` · `learning_reflect` | `learning_outcomes` · `learning_stats` · `Learning.exemplars_for` | `LEARNING` block - recent outcomes + success_rate. Prior *successful* traces are also spliced in as **few-shot exemplars** for local models. |
17
15
  | **Mistakes** | `mistakes.json` | `mistakes_record` · `mistakes_resolve` · *auto on failure* | `mistakes_list` | `KNOWN MISTAKES` + `KNOWN FIXES` blocks - do-NOT-repeat + do-THIS-instead |
18
- | **Metrics** | `metrics.json` | *automatic* (every Dispatch) | `metrics_summary` | `TOOL EFFECTIVENESS` block - steer tool choice. **Segmented per engine** (`engine=...`) so a local model's telemetry never blends with a frontier model's. |
16
+ | **Metrics** | `metrics.json` | *automatic* (every Dispatch) plus `Metrics.record_attempt` | `metrics_summary` | `TOOL EFFECTIVENESS` block, plus ESR and ASR on the health line. |
17
+
18
+ ## Exploit and attack rates
19
+
20
+ `PWN::AI::Agent::Metrics` keeps two rates that RSI can compare across turns.
21
+
22
+ ESR is `verified_exploit_tools / vulnerable_tools`. A tool enters `verified_exploit_tools` only when an exploit attempt for that tool name succeeds. A tool enters `vulnerable_tools` when the caller marks it vulnerable, or when a successful exploit names that target. Repeating the same tool does not change the counts. The rate stays nil until at least one vulnerable tool exists.
23
+
24
+ ASR is `successful_attacks / total_attack_attempts`. Every attack attempt increments the denominator. A reproduced impact, a still-open retest, or an accepted `chain_impact` increments the numerator. This one is an attempt ratio, not a set of tool names.
25
+
26
+ `Learning.rsi_tick` compares the current ESR with the previous snapshot. A drop writes a lesson tagged `rsi`. `Loop` already calls `Learning.auto_introspect`, and that path calls `rsi_tick`. A green rake is not an ESR sample. The default suite does not hit a live target.
19
27
 
20
28
  ## Memory write guard
21
29
 
@@ -122,6 +130,9 @@ bundled skills into `~/.pwn/skills/` when the name is missing:
122
130
  | `cwe` | Exhaustive test procedure per CWE ID (`references/CWE-<id>.md`) |
123
131
  | `capec` | Exhaustive attack-pattern procedure per CAPEC ID (`references/CAPEC-<id>.md`) |
124
132
  | `att&ck` | Exhaustive test procedure per ATT&CK technique (`references/T1059.001.md`) |
133
+ | `humanizer` | Strip AI writing patterns from prose. Keep meaning and identifiers. |
134
+
135
+ `PWN::AI::Agent::SkillReview` runs after introspection. A routine success does not create a skill. A tested procedure, a resolved recurring mistake, or an explicit correction can recommend an update to the closest skill under `~/.pwn/skills`. The default mode is `recommend`. `auto-safe` writes only a small addition that an execution fixture passed in three sessions, and it keeps a backup. It does not create skills, rewrite generated `pwn/` module skills, or save secrets and raw tool output. Set `ai.agent.skill_review` to `off`, `recommend`, or `auto-safe`.
125
136
 
126
137
  Source: `etc/default_skills/` in the gem. SOP edits in `~/.pwn/skills/<name>/SKILL.md`
127
138
  are never overwritten. Generated `~/.pwn/skills/pwn/**/SKILL.md` module skills
@@ -34,7 +34,7 @@ Escalate [label="escalate\n>=N fails → Swarm hint", fillcolor="#c4b5fd"
34
34
  subgraph cluster_intro {
35
35
  label="INTROSPECTION (self)"; fontcolor="#a7f3d0";
36
36
  style=rounded; color="#047857"; bgcolor="#022c22"; penwidth=2;
37
- Metrics [label="Metrics\nper-tool · PER-ENGINE\nsuccess · avg ms", fillcolor="#6ee7b7"];
37
+ Metrics [label="Metrics\nESR = verified exploit tools / vulnerable tools\nASR = successful attacks / attempts", fillcolor="#6ee7b7"];
38
38
  Learning [label="Learning\nnote_outcome · reflect\nexemplars_for · SFT quality gate\nfact_check_local_final", fillcolor="#6ee7b7"];
39
39
  Mistakes [label="Mistakes\nrecord · resolve\nguard · correction_hint\n[REPEATING] · [REGRESSED]", fillcolor="#6ee7b7", penwidth=2];
40
40
  ReflectT [label="Reflect\nteacher-student\n(reflect_engine)", fillcolor="#6ee7b7", penwidth=2];
@@ -106,6 +106,7 @@ Escalate [label="escalate\n>=N fails → Swarm hint", fillcolor="#c4b5fd"
106
106
  /* L2 → L3 */
107
107
  Metrics -> Correlate [color="#34d399"];
108
108
  Learning -> Correlate [color="#34d399"];
109
+ Learning -> Metrics [label="rsi_tick\nESR drop writes a lesson", style=dashed, color="#34d399", constraint=false];
109
110
  Mistakes -> Correlate [color="#34d399"];
110
111
  Snapshot -> Correlate [color="#f59e0b"];
111
112
  Observe -> Correlate [color="#f59e0b"];
@@ -47,8 +47,9 @@ digraph "PWN_Reinforcement_Learning" {
47
47
  W1 [label="preference ledger\ntrajectory pairs\nscrub + geometry filter\nwrite+export <=40%/src", fillcolor="#fda4af"];
48
48
  W2 [label="train_and_gate\nSFT+DPO → LoRA vN+1\ngate + preference diet\nexport_ready without trainer", fillcolor="#fda4af", penwidth=2];
49
49
  W3 [label="calibrate\np(success) vs actual → Brier", fillcolor="#fda4af"];
50
+ Rates [label="ESR verified_exploit_tools / vulnerable_tools\nASR successful_attacks / total_attack_attempts\nrsi_tick on an ESR drop", fillcolor="#fda4af"];
50
51
  }
51
- {rank=same; W1; W2; W3}
52
+ {rank=same; W1; W2; W3; Rates}
52
53
 
53
54
  /* Persistence */
54
55
  subgraph cluster_files {
@@ -97,6 +98,7 @@ digraph "PWN_Reinforcement_Learning" {
97
98
  W2 -> Ollama [label="LoRA", color="#fb7185", penwidth=2];
98
99
  Fmis -> W2 [label="eval set", style=dashed, color="#fbbf24", constraint=false];
99
100
  W3 -> Loop [label="calibration", style=dashed, color="#fbbf24", constraint=false];
101
+ Loop -> Rates [label="rsi_tick", color="#34d399", style=dashed, constraint=false];
100
102
 
101
103
  /* cron seeds */
102
104
  Cron [label="⏱ PWN::Cron.install_defaults\npractice · offline_judge\ntrain dry_run · consolidate", fillcolor="#7dd3fc"];
@@ -17,6 +17,8 @@ digraph "PWN_Reports" {
17
17
  Fuzz [label="Plugins::Fuzz", fillcolor="#6ee7b7"];
18
18
  Uri [label="URI Buster", fillcolor="#6ee7b7"];
19
19
  Phone[label="BareSIP recon", fillcolor="#6ee7b7"];
20
+ Obs [label="Nuclei · SBOM\nrecon observations\nnot findings", fillcolor="#6ee7b7"];
21
+ Find [label="Findings.record\nseverity info\nuntil verify", fillcolor="#6ee7b7"];
20
22
  }
21
23
  subgraph cluster_gen {
22
24
  label="Generators"; fontcolor="#fde68a"; style=rounded;
@@ -33,12 +35,16 @@ digraph "PWN_Reports" {
33
35
  JSON [label="JSON", fillcolor="#c4b5fd"];
34
36
  DD [label="DefectDojo\nimport/reimport", fillcolor="#c4b5fd"];
35
37
  Jira [label="JiraDataCenter", fillcolor="#c4b5fd"];
38
+ Gate [label="refuse_unproven_combined!\nhigh/critical needs reproduced\nchain file naming every id", fillcolor="#fda4af"];
36
39
  }
37
40
 
38
41
  Sast -> Rsast [color="#f59e0b"];
39
42
  Fuzz -> Rfuzz [color="#f59e0b"];
40
43
  Uri -> Ruri [color="#f59e0b"];
41
44
  Phone -> Rphone [color="#f59e0b"];
45
+ Obs -> Find [label="handoff", color="#34d399"];
46
+ Find -> Gate [label="verify transcript", color="#f59e0b"];
47
+ Gate -> HTML [color="#a78bfa"];
42
48
  Rsast -> HTML [color="#a78bfa"];
43
49
  Rfuzz -> JSON [color="#a78bfa"];
44
50
  Ruri -> DD [color="#a78bfa"];
@@ -23,6 +23,8 @@ digraph "PWN_ZeroDay" {
23
23
  color="#a16207"; bgcolor="#422006";
24
24
  Fuzz [label="Fuzz · Sock\nPacket", fillcolor="#fcd34d"];
25
25
  TB [label="TransparentBrowser\nBurp replay", fillcolor="#fcd34d"];
26
+ Obs [label="Recon.observe\nhost port product\nversion evidence", fillcolor="#fcd34d"];
27
+ Ver [label="Findings.verify\ntranscript shows impact\nor severity stays info", fillcolor="#fcd34d"];
26
28
  }
27
29
  subgraph cluster_weapon {
28
30
  label="Weaponise"; fontcolor="#fecaca"; style=rounded;
@@ -39,8 +41,11 @@ digraph "PWN_ZeroDay" {
39
41
 
40
42
  Surface -> SAST [color="#38bdf8"];
41
43
  Surface -> AI [color="#38bdf8"];
42
- SAST -> Fuzz [color="#f59e0b"];
43
- AI -> TB [color="#f59e0b"];
44
+ SAST -> Obs [color="#f59e0b"];
45
+ AI -> Obs [color="#f59e0b"];
46
+ Obs -> Ver [label="lead, not a finding", color="#f59e0b"];
47
+ Ver -> Fuzz [color="#f59e0b"];
48
+ Ver -> TB [color="#f59e0b"];
44
49
  Fuzz -> Asm [color="#fb7185"];
45
50
  TB -> MSF [color="#fb7185"];
46
51
  Asm -> H1 [color="#a78bfa"];