pwn 0.5.722 → 0.5.724

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. checksums.yaml +4 -4
  2. data/Gemfile +2 -2
  3. data/bin/pwn_setup +5 -5
  4. data/documentation/AI-Integration.md +34 -1
  5. data/documentation/Policy-Benchmark.md +296 -0
  6. data/documentation/Reinforcement-Learning.md +81 -3
  7. data/etc/default_skills/pwn/ai/agent/engagement/SKILL.md +1 -0
  8. data/etc/default_skills/pwn/ai/agent/metrics/SKILL.md +3 -0
  9. data/etc/default_skills/pwn/ai/agent/policy/SKILL.md +4 -1
  10. data/etc/default_skills/pwn/ai/agent/policy_evaluation/SKILL.md +49 -0
  11. data/etc/default_skills/pwn/ai/agent/registry/SKILL.md +2 -0
  12. data/etc/default_skills/pwn/ai/agent/reward/SKILL.md +3 -0
  13. data/etc/default_skills/pwn/ai/agent/swarm/SKILL.md +6 -0
  14. data/etc/default_skills/pwn/ai/agent/tools/capabilities/SKILL.md +45 -0
  15. data/etc/default_skills/pwn/ai/agent/tools/context/SKILL.md +45 -0
  16. data/etc/default_skills/pwn/ai/agent/verification/SKILL.md +48 -0
  17. data/etc/default_skills/pwn/ai/context/SKILL.md +50 -0
  18. data/etc/default_skills/pwn/ai/http_retry/SKILL.md +7 -0
  19. data/etc/default_skills/pwn/ai/http_retry/references/urls.md +4 -0
  20. data/etc/default_skills/pwn/ai/open_ai/SKILL.md +1 -0
  21. data/etc/default_skills/pwn/ai/open_ai/references/urls.md +1 -0
  22. data/etc/default_skills/pwn/plugins/findings/SKILL.md +1 -0
  23. data/etc/default_skills/pwn/plugins/gdbmi/SKILL.md +55 -0
  24. data/etc/default_skills/pwn/plugins/ghidra_headless/SKILL.md +49 -0
  25. data/etc/default_skills/pwn/plugins/jobs/SKILL.md +5 -0
  26. data/etc/default_skills/pwn/plugins/packet/SKILL.md +3 -0
  27. data/etc/default_skills/pwn/plugins/preflight_checker/SKILL.md +1 -0
  28. data/etc/default_skills/pwn/plugins/radare2/SKILL.md +1 -0
  29. data/etc/default_skills/pwn/plugins/transparent_browser/SKILL.md +3 -0
  30. data/etc/default_skills/pwn/reports/engagement/SKILL.md +3 -2
  31. data/lib/pwn/ai/agent/curriculum.rb +37 -45
  32. data/lib/pwn/ai/agent/engagement.rb +59 -0
  33. data/lib/pwn/ai/agent/learning.rb +85 -52
  34. data/lib/pwn/ai/agent/loop.rb +72 -9
  35. data/lib/pwn/ai/agent/metrics.rb +52 -2
  36. data/lib/pwn/ai/agent/policy.rb +288 -49
  37. data/lib/pwn/ai/agent/policy_evaluation.rb +230 -0
  38. data/lib/pwn/ai/agent/registry.rb +44 -4
  39. data/lib/pwn/ai/agent/reward.rb +188 -46
  40. data/lib/pwn/ai/agent/swarm.rb +235 -35
  41. data/lib/pwn/ai/agent/tool_guard.rb +13 -1
  42. data/lib/pwn/ai/agent/tools/artifacts.rb +42 -0
  43. data/lib/pwn/ai/agent/tools/capabilities.rb +19 -0
  44. data/lib/pwn/ai/agent/tools/context.rb +38 -0
  45. data/lib/pwn/ai/agent/tools/finding_record.rb +18 -0
  46. data/lib/pwn/ai/agent/tools/fuzz_campaign.rb +10 -1
  47. data/lib/pwn/ai/agent/tools/job_run.rb +32 -0
  48. data/lib/pwn/ai/agent/tools/learning.rb +5 -6
  49. data/lib/pwn/ai/agent/tools/metrics.rb +16 -0
  50. data/lib/pwn/ai/agent/tools/pty_session.rb +4 -4
  51. data/lib/pwn/ai/agent/tools/ruby_eval.rb +6 -5
  52. data/lib/pwn/ai/agent/tools/shell.rb +10 -1
  53. data/lib/pwn/ai/agent/tools/skills.rb +30 -0
  54. data/lib/pwn/ai/agent/tools/swarm.rb +8 -2
  55. data/lib/pwn/ai/agent/verification.rb +202 -0
  56. data/lib/pwn/ai/agent.rb +2 -0
  57. data/lib/pwn/ai/context.rb +193 -0
  58. data/lib/pwn/ai/http_retry.rb +53 -7
  59. data/lib/pwn/ai/open_ai.rb +315 -45
  60. data/lib/pwn/ai.rb +1 -0
  61. data/lib/pwn/migrate.rb +10 -1
  62. data/lib/pwn/plugins/artifact_registry.rb +13 -6
  63. data/lib/pwn/plugins/findings.rb +48 -8
  64. data/lib/pwn/plugins/gdbmi.rb +128 -0
  65. data/lib/pwn/plugins/ghidra_headless.rb +104 -0
  66. data/lib/pwn/plugins/jobs.rb +53 -0
  67. data/lib/pwn/plugins/packet.rb +51 -0
  68. data/lib/pwn/plugins/preflight_checker.rb +29 -0
  69. data/lib/pwn/plugins/process_tube.rb +24 -7
  70. data/lib/pwn/plugins/radare2.rb +14 -2
  71. data/lib/pwn/plugins/transparent_browser.rb +64 -0
  72. data/lib/pwn/plugins.rb +2 -0
  73. data/lib/pwn/reports/engagement.rb +19 -0
  74. data/lib/pwn/sessions.rb +3 -1
  75. data/lib/pwn/version.rb +1 -1
  76. data/scripts/benchmark_policy.rb +376 -0
  77. data/spec/documentation/installation_md_spec.rb +18 -4
  78. data/spec/integration/reinforced_feedback_loop_spec.rb +33 -19
  79. data/spec/lib/pwn/ai/agent/curriculum_spec.rb +267 -0
  80. data/spec/lib/pwn/ai/agent/engagement_spec.rb +12 -0
  81. data/spec/lib/pwn/ai/agent/learning_spec.rb +100 -3
  82. data/spec/lib/pwn/ai/agent/loop_spec.rb +104 -0
  83. data/spec/lib/pwn/ai/agent/metrics_spec.rb +44 -0
  84. data/spec/lib/pwn/ai/agent/policy_evaluation_spec.rb +221 -0
  85. data/spec/lib/pwn/ai/agent/policy_spec.rb +246 -6
  86. data/spec/lib/pwn/ai/agent/registry_spec.rb +123 -0
  87. data/spec/lib/pwn/ai/agent/reward_spec.rb +196 -12
  88. data/spec/lib/pwn/ai/agent/swarm_spec.rb +121 -1
  89. data/spec/lib/pwn/ai/agent/tool_guard_spec.rb +6 -0
  90. data/spec/lib/pwn/ai/agent/tools/capabilities_spec.rb +14 -0
  91. data/spec/lib/pwn/ai/agent/tools/context_spec.rb +14 -0
  92. data/spec/lib/pwn/ai/agent/tools/job_run_spec.rb +2 -0
  93. data/spec/lib/pwn/ai/agent/tools/learning_spec.rb +25 -0
  94. data/spec/lib/pwn/ai/agent/verification_spec.rb +172 -0
  95. data/spec/lib/pwn/ai/context_spec.rb +48 -0
  96. data/spec/lib/pwn/ai/http_retry_spec.rb +27 -0
  97. data/spec/lib/pwn/ai/open_ai_oauth_transport_spec.rb +245 -0
  98. data/spec/lib/pwn/ai/open_ai_spec.rb +235 -0
  99. data/spec/lib/pwn/migrate_spec.rb +24 -0
  100. data/spec/lib/pwn/plugins/artifact_registry_spec.rb +10 -0
  101. data/spec/lib/pwn/plugins/findings_spec.rb +2 -0
  102. data/spec/lib/pwn/plugins/gdbmi_spec.rb +17 -0
  103. data/spec/lib/pwn/plugins/ghidra_headless_spec.rb +17 -0
  104. data/third_party/pwn_rdoc.jsonl +100 -4
  105. metadata +30 -5
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 31f54d1cd6a7bfbc4cef3bcc989650930b1b31bbf2bbc47654b20d6f20158775
4
- data.tar.gz: 50bfda31eecd4b99d17324a8c4c548f14b8df10561c183f27bc9e7157c8f88bf
3
+ metadata.gz: b1c409d6b05e3c061837ef187c74d22ff5e24d361aee4531be0943714a8f713b
4
+ data.tar.gz: dacfce51246aee0f02c84b5a8eae1651fe2853db2e0f192343c5fc8cfbc5462b
5
5
  SHA512:
6
- metadata.gz: 3a43118f36da1f9712dbfb908b5f0c9d4650d3bfde63239240d7c9cefcc50d29d78468e14e9cd9bf44f9fdba5df971b48659b348619dc79a7e17667028930025
7
- data.tar.gz: 867d72ec34f85caaf18fa5abeb38b466d5663c1361becf17cb4af87115dd0d45411527b0f5616dd54874c241544416fb0b9427f16476248ed38a864b3951daef
6
+ metadata.gz: 6c29fd2cde7775f1923249f166af195dcc99ed0614c86536d3aeb56bcc7a3d29a28a3483b1b2c25bb736b2b78b28ab92a0906d690cd71d002ff335c41e1e5d78
7
+ data.tar.gz: cbcf28e1681df68a21a1da6f247bf645de861ee2a752af2f3735f09f2a59501bdd9d5fb5bf238dcc39ae2734ea173a3d515ddb3df1de2d7548de4fbbb67c5ea7
data/Gemfile CHANGED
@@ -49,7 +49,7 @@ gem 'jwt', '3.2.0'
49
49
  gem 'libusb', '0.8.0'
50
50
  gem 'luhn', '3.0.0'
51
51
  gem 'mail', '2.9.1'
52
- gem 'mcp', '1.4.0'
52
+ gem 'mcp', '1.5.0'
53
53
  gem 'meshtastic', '0.0.175'
54
54
  gem 'metasm', '1.0.6'
55
55
  gem 'mongo', '2.25.0'
@@ -77,7 +77,7 @@ gem 'rbvmomi2', '3.10.0'
77
77
  gem 'rdoc', '7.0.4'
78
78
  gem 'rest-client', '2.1.0'
79
79
  gem 'rex', '2.0.13'
80
- gem 'rmagick', '7.1.3'
80
+ gem 'rmagick', '7.1.4'
81
81
  gem 'rqrcode', '3.2.0'
82
82
  gem 'rspec', '3.13.2'
83
83
  gem 'rtesseract', '3.1.4'
data/bin/pwn_setup CHANGED
@@ -25,8 +25,10 @@ pwn_driver = PWN::Driver::Parser.new do |options|
25
25
  end
26
26
 
27
27
  options.on('-l', '--list-profiles',
28
- '<Optional - List capability profiles and exit>') do |o|
29
- opts[:list] = o
28
+ '<Optional - List capability profiles and exit>') do
29
+ # Public metadata, like --help: no environment decryption or bootstrap.
30
+ PWN::Setup.list_profiles
31
+ exit 0
30
32
  end
31
33
 
32
34
  options.on('-m', '--migrate',
@@ -66,9 +68,7 @@ pwn_driver.parse!
66
68
  begin
67
69
  status = 0
68
70
 
69
- if opts[:list]
70
- PWN::Setup.list_profiles
71
- elsif opts[:migrate]
71
+ if opts[:migrate]
72
72
  r = PWN::Setup.migrate(
73
73
  fix: opts[:fix] || opts[:yes],
74
74
  dry_run: opts[:dry_run]
@@ -10,7 +10,7 @@ agent code never cares which model is behind it.
10
10
 
11
11
  | Engine | Client | Auth | Notes |
12
12
  |---|---|---|---|
13
- | `openai` | `PWN::AI::OpenAI` | `key:` | function-calling native |
13
+ | `openai` | `PWN::AI::OpenAI` | ChatGPT/Codex `oauth:` or `key:` | OAuth preferred when configured; subscription chat uses the Codex Responses backend, API keys use the Platform API |
14
14
  | `anthropic` | `PWN::AI::Anthropic` | `key:` | tool-use native |
15
15
  | `grok` | `PWN::AI::Grok` | `key:` **or** `oauth: true` | OAuth = RFC-8628 device-code flow using xAI's public Grok-CLI client id (no secret) - see skill `xai_grok_oauth_device_flow` |
16
16
  | `gemini` | `PWN::AI::Gemini` | `key:` | function-calling native |
@@ -38,6 +38,39 @@ ai:
38
38
  PWN::Env[:ai][:active] = :ollama
39
39
  ```
40
40
 
41
+ ## OpenAI OAuth enrollment and persistence
42
+
43
+ To prefer subscription OAuth, set `ai.openai.oauth.enroll: true` through
44
+ `pwn-vault`, then restart `pwn-ai` and follow the device-code consent prompt.
45
+ An existing bearer or refresh token is preferred over `ai.openai.key`.
46
+ The account must have access to the requested model through Codex; OAuth
47
+ does not grant all Platform API permissions or bypass subscription limits.
48
+
49
+ You can also enroll explicitly from a PWN Ruby console:
50
+
51
+ ```ruby
52
+ PWN::AI::OpenAI.obtain_oauth_bearer_token; nil
53
+ ```
54
+
55
+ The trailing `nil` prevents the console from echoing the returned bearer.
56
+ Successful enrollment updates the live environment and calls the private
57
+ `persist_oauth_to_vault` helper, as token refresh does. It uses the configured
58
+ `driver_opts.pwn_env_path` and `pwn_dec_path`, defaulting to
59
+ `~/.pwn/pwn.yaml` and its `.decryptor` file. Existing decryption artifacts are
60
+ required; missing artifacts leave tokens in the session only and produce a
61
+ notice. Tokens are not printed by the enrollment success message.
62
+
63
+ Persistence retains unrelated settings and the existing key/IV, encrypts a
64
+ private temporary file, and replaces the vault only after encryption succeeds.
65
+ The resulting vault has mode `0600`. The original vault remains unchanged on
66
+ a persistence failure; the helper does not create or rotate decryptor secrets.
67
+
68
+ Subscription OAuth requests must not go to the default Platform
69
+ `/v1/responses` endpoint: that mismatch can produce `Missing scopes:
70
+ api.responses.write`. OAuth chat uses
71
+ `https://chatgpt.com/backend-api/codex/responses` instead. An actual Platform
72
+ API key still needs the relevant project and endpoint permissions.
73
+
41
74
  ## Engine-aware behavior
42
75
 
43
76
  The harness adapts to the *class* of engine, not the model name:
@@ -0,0 +1,296 @@
1
+ # Independent local Policy/Registry benchmark
2
+
3
+ This is a **deterministic controller benchmark, not proof of live LLM improvement**.
4
+ It executes real local file tasks and trains the actual tabular
5
+ `PWN::AI::Agent::Policy` implementation, then selects actions through the actual
6
+ `Registry.rank`. No model responses, provider usage, or benchmark gains are
7
+ simulated. The action handlers are explicitly hand-written local algorithms,
8
+ not a simulated language model.
9
+
10
+ ## Run from a source checkout
11
+
12
+ ```sh
13
+ ruby scripts/benchmark_policy.rb --self-check
14
+ ruby scripts/benchmark_policy.rb --output /tmp/pwn-policy-benchmark.json
15
+ bundle exec rubocop scripts/benchmark_policy.rb
16
+ ```
17
+
18
+ The experiment uses Ruby and its standard libraries; it does not require the
19
+ full application boot sequence. Use a fresh Ruby process, not the live agent
20
+ console. JSON is printed to stdout and optionally written to `--output`.
21
+ `--self-check` runs independent scorer tests, both complete experimental arms,
22
+ and two fresh snapshot workers; it prints short pass messages instead of a
23
+ report. Use the options separately.
24
+ The benchmark exits nonzero if persistence changes during evaluation, splits
25
+ overlap, a negative control succeeds, or a positive control fails. Self-checks
26
+ also require actual training updates in the on arm and none in the off arm.
27
+ They deliberately do **not** require an improvement of a predetermined size.
28
+
29
+ ## Relation to existing evaluation
30
+
31
+ `lib/pwn/ai/agent/curriculum.rb` provides live self-play, judge-based practice,
32
+ mistake-derived evaluation prompts, and adapter training/promotion gates. This
33
+ standalone script does not invoke those paths. It provides a small, independently
34
+ scored experiment where the training labels come from actual task achievement,
35
+ not `Reward.judge`, model prose, or a previously assigned reward. It calls
36
+ `Policy.begin_episode`, `observe_step`, and `finish` with those labels during
37
+ training. This is not a test of Reward's verification-record binding, the live
38
+ Loop, Metrics learning, memory retrieval, or adapter training.
39
+
40
+ The experimental driver remains in `scripts/`. The opt-in `PolicyEvaluation`
41
+ module launches only this fixed runner; it does not load the live agent in its
42
+ workers. Its module autoload does not enable evaluation or promotion.
43
+
44
+ ## Protocol
45
+
46
+ 1. **Isolation.** Create a private `/tmp/pwn-policy-benchmark-*` directory. Clear
47
+ the process environment, set `HOME` and `TMPDIR` to that directory, then load
48
+ only Policy and Registry. Assert their persistence constants resolve below
49
+ the temporary `HOME/.pwn`. Do not load user configuration, credentials,
50
+ registered application tools, providers, or network libraries. Delete the
51
+ entire temporary directory on normal completion or Ruby exception and restore
52
+ the process environment. A force-killed process may leave its temporary files.
53
+ This is isolation for trusted fixed handlers, not a sandbox for untrusted code.
54
+ 2. **Paired arms.** Run `off`, then `on`, resetting Policy between them. Both arms
55
+ start empty and execute the identical training schedule. The only learning
56
+ switch is `PWN::Env[:ai][:agent][:policy]`. No Q entries, visits, episode counts,
57
+ or rewards are seeded directly. Other learning modules are not loaded.
58
+ 3. **Training only.** There are four training fixtures: two numeric sorting tasks
59
+ and two active-inventory filtering tasks. Six fixed exploration rounds execute
60
+ every one of the three candidate actions for each fixture: 72 real handler
61
+ calls per arm. Every action gets equal exposure; the scheduler does not use
62
+ answer keys to choose actions. Each call writes into a fresh task directory.
63
+ Exact artifact correctness supplies a binary training label to Policy; the
64
+ off arm executes the same work but Policy declines to update.
65
+ Each call has an explicit action ID. Only after the independent artifact
66
+ check, `finish` receives a `controlled_comparison` attribution receipt naming
67
+ that ID. This is a one-action isolated experiment, not a claim that the last
68
+ action in an arbitrary live trace caused the outcome.
69
+ 4. **Held-out evaluation.** Only after training, materialize eight distinct
70
+ held-out inputs with literal answer keys: four numeric and four inventory
71
+ tasks. Assert no training input appears in evaluation. Task families and
72
+ request wording are intentionally shared with training. The held-out units
73
+ are **input instances, not unseen families, prompts, or tools**. This is
74
+ in-distribution transfer within a small public fixture set, not a blind test
75
+ or a generalization claim about arbitrary operator requests.
76
+ 5. **Frozen controller.** Rank the task family's three entries using
77
+ `Registry.rank(query: ..., entries: ..., preference: [])`. There are no
78
+ evaluation `begin_episode`, `observe_step`, or `finish` calls. The policy uses
79
+ Registry's normal fallback state rather than a fabricated state/table. Allow
80
+ at most three attempts, stopping only when the independent artifact checker
81
+ passes. The controller does not get corrective feedback or adapt between
82
+ attempts. Hash all persisted `.pwn` files before evaluation and after controls;
83
+ abort if they changed.
84
+ 6. **Separate controls.** For each held-out input, force a claim-only handler and
85
+ a handler that writes a wrong JSON object. Both return convincing
86
+ `PASS: completed successfully...verified` prose and `ok: true`. All 16 forced
87
+ negative controls must fail objective scoring. Also execute the correct
88
+ algorithm on each input; all eight positive controls must pass. Controls are
89
+ reported separately and never trained on or included in evaluation rates.
90
+
91
+ ## Real task behavior and independent scoring
92
+
93
+ Numeric candidates perform lexical sorting, numeric sorting, or no artifact
94
+ write. Inventory candidates filter active rows, include every row, or write
95
+ nothing. Handlers receive only input/output paths; the literal expected values
96
+ are not passed to them. All handlers claim success, intentionally making textual
97
+ claims unreliable. The checker ignores their prose and action names. It accepts
98
+ only a regular non-symlink output whose parsed JSON exactly equals the answer key,
99
+ with the source input unchanged. Missing output, invalid JSON, a plausible wrong
100
+ artifact, or an altered input is failure. Self-checks separately test missing,
101
+ wrong, prose-only, symlink, and correct artifacts.
102
+
103
+ Each family has equal keyword-fit descriptions, with preference disabled. The
104
+ unlearned deterministic tie-break favors lexical sort in one family and the
105
+ correct active filter in the other; it is not configured to lose every task.
106
+ Lexical sorting can genuinely solve some numeric fixtures and receives credit
107
+ when it does. The learned controller can change these tied rankings from
108
+ observed outcomes. This construction deliberately isolates the learning-to-router
109
+ connection; it does not measure natural-language tool-selection quality.
110
+
111
+ The answer keys and algorithms live in the same public script but do not call
112
+ one another. Independence here means scoring actual artifact/task achievement
113
+ without trusting action claims or training rewards, not process-level secrecy or
114
+ an external audit. Extending the task set requires reviewing literal answer keys
115
+ and adding both positive and negative controls before interpreting new results.
116
+
117
+ ## Report definitions
118
+
119
+ Each arm contains full training/evaluation traces, fixture inputs and answer keys,
120
+ control traces, Policy statistics, persistence fingerprints, and split/freeze
121
+ checks. Source hashes identify the Policy, Registry, and harness revisions used.
122
+ No fixed gains are embedded in the report.
123
+
124
+ - **Completion:** tasks with at least one objectively successful attempt divided
125
+ by evaluated tasks. A failed attempt does not count as task completion.
126
+ - **Artifact check score:** each attempt earns 1.0 only for an exact correct
127
+ artifact with unchanged input, otherwise 0.0; phase score is its mean across
128
+ attempts. Report rows retain the actual artifact body as well as its hash.
129
+ - **False-success count/rate:** attempts claiming `ok: true` without achieving the
130
+ task; rate denominator is executed attempts, not tasks. This is false reporting
131
+ by the handler, not acceptance of that report by the independent checker.
132
+ - **Repeated mistakes:** every failed attempt after the first occurrence of the
133
+ same `(family, action, checker failure)` signature within that phase and arm.
134
+ This includes both repeated attempts on one task and recurrence on later
135
+ held-out inputs. It is not the production Mistakes store's count.
136
+ - **Calls:** `tool_calls` counts actual local handler invocations. Evaluation makes
137
+ one Registry ranking decision per attempted handler call. Training invokes
138
+ begin/observe/finish once per training call; the returned update reports are
139
+ preserved. Each separately listed control row is one additional handler call.
140
+ - **Elapsed:** monotonic measured seconds. Row elapsed time measures the handler;
141
+ phase elapsed time also includes setup, ranking/checking, and training updates
142
+ as applicable. Evaluation phase time excludes separately listed controls;
143
+ top-level elapsed includes both arms and controls but not final JSON output or
144
+ temporary-directory teardown. Tiny timings vary with caching and filesystem
145
+ load; fixed arm order is not a timing-performance study.
146
+ - **Cost:** zero LLM calls and zero provider tokens because no provider is invoked.
147
+ Monetary cost is `null` (not estimated), not a claim that local computing is
148
+ economically free. No token-price or electricity estimates are invented.
149
+
150
+ Training completion is descriptive coverage under forced exploration, not a
151
+ learned-policy score. Only the held-out evaluation rates compare the controllers.
152
+ Repeated executions should reproduce actions, counts, fixture hashes, and update
153
+ counts for the same source revisions. Wall times, temporary paths, timestamps,
154
+ and timestamp-bearing persistence hashes are expected to differ.
155
+
156
+ There is deliberately no external-runner plug-in: accepting arbitrary commands
157
+ would undermine the no-network/no-credentials guarantee. A future live-model
158
+ study should use a separately reviewed runner, identical model/tool budgets,
159
+ external objective verifiers, frozen held-out evaluation, and actual provider
160
+ usage records. Do not present this controller experiment as that study.
161
+
162
+ ## Opt-in independent snapshot evaluation (R5)
163
+
164
+ The original commands and `off`/`on` report shape still work. To additionally
165
+ export snapshots and run repeated evaluations in **fresh subprocesses**, use:
166
+
167
+ ```sh
168
+ ruby scripts/benchmark_policy.rb --heldout \
169
+ --snapshot-dir /tmp/pwn-policy-snapshots \
170
+ --output /tmp/pwn-policy-heldout.json
171
+ ```
172
+
173
+ The snapshot directory must be **new**, below `/tmp`, with no symlink parents.
174
+ `off.json` and `on.json` are genuine Policy JSON tables after the same real
175
+ training schedule. Exported `updated_at` metadata is normalized to `null` for
176
+ reproducible snapshot digests; no Q entries or rewards are fabricated. Export
177
+ happens before that arm's evaluation. Neither `--heldout` nor `--snapshot-dir`
178
+ promotes anything or discovers/reads a real home-directory policy.
179
+
180
+ The added `heldout` array contains protocol `pwn-policy-heldout-v2`, for suite
181
+ indices 0 and 1. Each worker runs three frozen arms: `off` (baseline snapshot,
182
+ policy disabled), `baseline` (baseline enabled), and `candidate` (candidate
183
+ enabled). Workers never call begin/observe/finish, warmup, or reset the caller's
184
+ policy. Resets and snapshot installation occur only inside temporary HOME.
185
+ Only fixed local handlers are registered; no provider, shell tool, credential,
186
+ user config, or network client is loaded. Environment variables, including Ruby
187
+ startup hooks, are removed before spawning Ruby. Workers have a 30-second
188
+ deadline; stalled children are killed and reaped. Snapshot inputs must be regular
189
+ non-symlink files, at most 4 MiB, with valid numeric `q`, `h`, `visits`, `returns`,
190
+ `n_updates`, and `td_abs_sum` fields. Parent directories cannot be symlinks.
191
+
192
+ Suite indices 0..7 are bounded deterministic variations, **not random trials**.
193
+ Numeric inputs and separate literal answer keys are scaled by `seed + 1`;
194
+ inventory IDs and separate answer keys are offset by `100 * seed`. This does not
195
+ call a candidate algorithm to construct its answer key. Even indices use flat
196
+ paths and compact JSON; odd indices use nested paths containing spaces, pretty
197
+ JSON, and read-only inputs. Every arm runs eight tasks, 16 negative controls and
198
+ eight positive controls. Training fixtures remain unchanged and disjoint.
199
+ These are two task families and two filesystem configurations, not unseen tools
200
+ or broad environment generalization. Both environment types are required for
201
+ promotion eligibility.
202
+ For externally supplied snapshots, `disjoint_inputs` describes the harness's
203
+ fixture sets, not proof of the snapshot's training history; that history is not
204
+ attested by this runner.
205
+
206
+ Explicit snapshots from another controlled experiment can be evaluated without
207
+ booting the application:
208
+
209
+ ```ruby
210
+ require './lib/pwn/ai/agent/policy_evaluation'
211
+ evaluator = PWN::AI::Agent::PolicyEvaluation
212
+ reports = [0, 1].map do |seed|
213
+ evaluator.evaluate(baseline: '/tmp/baseline.json',
214
+ candidate: '/tmp/candidate.json', seed: seed)
215
+ end
216
+ ```
217
+
218
+ With fixed source revisions and snapshot bytes, snapshot reports reproduce all
219
+ fields except `elapsed_seconds`. They contain no wall-clock timestamps, random
220
+ IDs or temporary paths. The original training report still contains the timing
221
+ and metadata variability described above.
222
+
223
+ ## Explicit promotion and rollback
224
+
225
+ This is an **operator-invoked local eligibility gate**, not automatic online
226
+ policy promotion. `Policy.finish` and Loop do not call it. The existing online
227
+ learning behavior is not redirected or promoted by this module. No live policy
228
+ path is defaulted, and writes require `enabled: true` **and** `quiescent: true`.
229
+ The latter is an operator assertion: **stop all agent processes and policy
230
+ writers first**. The existing Policy writer does not share a transaction lock
231
+ with this module; concurrent live learning or concurrent promotions are not
232
+ supported. Restart writers only after the operation and readback complete.
233
+
234
+ Promotion requires 2..8 reports with distinct valid suite indices covering both
235
+ filesystem configurations. It then **reruns each suite in a fresh worker** using
236
+ the specified snapshot bytes. Every non-timing report field must match the fresh
237
+ execution, including source/harness digests, snapshot digests, input hashes,
238
+ artifact bodies and hashes, action choices, scores, controls, and frozen-policy
239
+ checks. Hashes alone are not signatures or evidence of trusted authorship;
240
+ re-execution is the authority. Model-written `passed: true`, edited scores,
241
+ invented artifact bodies, stale source revisions, or copied duplicate reports
242
+ cannot substitute for those executions. Supplied elapsed times are discarded;
243
+ only newly measured times enter the gate.
244
+
245
+ For **every** repeated suite, compared with both baseline-on and off:
246
+
247
+ - completion and mean artifact-check score must not decrease;
248
+ - no previously solved individual task may become unsolved (aggregate gains
249
+ cannot hide a task/family regression);
250
+ - false-success count **and rate**, repeated mistakes and tool calls must not rise;
251
+ - elapsed time must be at most `baseline_seconds * 1.25 + 0.02`, a fixed local
252
+ jitter allowance rather than evidence of a statistically established speedup.
253
+
254
+ Each suite must also improve completion, false-success count, repeated mistakes,
255
+ or calls relative to baseline-on. Better training returns, more updates, or a
256
+ timing-only change cannot qualify. The gate fails closed when verification fails.
257
+ Snapshots, source digests and live-baseline bytes must still match. An explicit
258
+ live target must already exist and equal the evaluated baseline byte-for-byte.
259
+ The previous policy is saved beside it as a digest-named rollback JSON before a
260
+ same-directory atomic replacement and exact readback. No trajectory file is
261
+ modified. Preserve the returned receipt for rollback.
262
+
263
+ A disposable demonstration using the opt-in benchmark output above:
264
+
265
+ ```ruby
266
+ require './lib/pwn/ai/agent/policy_evaluation'
267
+ evaluator = PWN::AI::Agent::PolicyEvaluation
268
+ baseline = '/tmp/pwn-policy-snapshots/off.json'
269
+ candidate = '/tmp/pwn-policy-snapshots/on.json'
270
+ reports = JSON.parse(File.read('/tmp/pwn-policy-heldout.json'), symbolize_names: true).fetch(:heldout)
271
+ live = '/tmp/pwn-policy-demo-live.json' # NOT the real online policy
272
+ File.open(live, File::WRONLY | File::CREAT | File::EXCL, 0o600) { |f| f.write(File.binread(baseline)) }
273
+ receipt = evaluator.promote(enabled: true, quiescent: true,
274
+ baseline: baseline, candidate: candidate,
275
+ reports: reports, live_path: live)
276
+ raise receipt.inspect unless receipt[:promoted]
277
+ restored = evaluator.rollback(enabled: true, quiescent: true,
278
+ live_path: live, receipt: receipt)
279
+ raise restored.inspect unless restored[:rolled_back]
280
+ ```
281
+
282
+ Rollback verifies the receipt's explicit target, backup path and prior digest,
283
+ validates the backup schema, and refuses if the live file no longer matches the
284
+ promoted candidate digest. Missing, altered or symlink backups/targets fail
285
+ closed. Both methods default to a disabled result; rejected operations return
286
+ `promoted: false` or `rolled_back: false` with a reason. `evaluate` raises on an
287
+ invalid snapshot, worker failure, or deadline. Force-killing a worker can leave
288
+ its temporary directory; this is not an OS sandbox for untrusted code.
289
+
290
+ **Limit:** this gate measures only the fixed benchmark action vocabulary and
291
+ public task families. A real online policy containing unrelated tools may show
292
+ no gain and be rejected; passing does not validate those unrelated routes or
293
+ establish live LLM gains. A production rollout still needs separately reviewed,
294
+ representative objective tasks and operator judgment. Do not interpret this
295
+ small public held-out set as a secret test or optimize repeatedly against it
296
+ and then claim independent generalization.
@@ -47,10 +47,10 @@ This is the live numeric controller. It does not replace planning.
47
47
 
48
48
  | Piece | What it is |
49
49
  |---|---|
50
- | State | request kind, task family, plan quality, answer completeness, usable-result, last action, fail bin, and engine |
50
+ | State | request kind, task family, plan quality, answer completeness, usable-result, last action, fail bin, engine, and sanitized host-observed capability/verification scope |
51
51
  | Action | tool name, or `final` |
52
52
  | Step reward | 0; −0.01 per tool after 8 |
53
- | Terminal reward | `Reward.judge` × confidence (sole large R). `plan_coverage` is a tag, not the score. |
53
+ | Terminal reward | Resolved training score mapped to −1..1 × confidence (sole large R), attributed only to independently linked actions. `plan_coverage` is a tag, not the score. |
54
54
  | Updates | Q-learning (`alpha=0.15`, `gamma=0.85`) and REINFORCE (`alpha=0.05`). Stored trajectories replay twice on warmup so a short table is not empty advice. |
55
55
  | Budget | Eight finished episodes (live or warmup-credited) unlock greedy suggestions. Until then the prompt omits them. |
56
56
  | Steer | Q-advantage in `Registry.rank` once the episode budget is met; keyword fit and CORE_TOOLS still come first. Suggested actions follow `Registry.preference_order` (`ai.agent.tool_preference`). |
@@ -65,6 +65,84 @@ high-return / high-score episodes.
65
65
 
66
66
  ## Reward signal (`PWN::AI::Agent::Reward`)
67
67
 
68
+ `Reward.resolve_outcome` is the shared decision used by live learning and
69
+ offline practice. The ledger retains the raw judge score, source, confidence,
70
+ verdict, and verification evidence. Heuristic guesses, evaluator errors, and
71
+ unresolved high-score/critic disagreements are unverified, with a nil
72
+ `training_score`; they do not become policy updates, failures, or supervised
73
+ success examples. Requeuing a disputed result rejudges its original session
74
+ instead of raising its score automatically.
75
+
76
+ Writing “PASS” in an answer, returning exit zero, or confirming one claim does
77
+ not establish completion. A trusted host verifier can call
78
+ `Reward.record_verification` after checking every original-request criterion,
79
+ providing actual boolean check results and evidence. Records are bound to the
80
+ current session request and invalidated by subsequent tool execution. The API
81
+ trusts the host verifier to cover the complete request; it cannot infer missing
82
+ criteria or automatically verify arbitrary tasks. LLM judgments remain fallible.
83
+
84
+ ### Executed acceptance checks and artifact attribution
85
+
86
+ For stronger verification, pass a host-owned `verification_contract` to
87
+ `Loop.run`, or use `Reward.run_verification(request:, session_id:, contract:)`.
88
+ The contract contains `root`, `requirements` (distinct verbatim clauses of the
89
+ original request), and `checks`. Each check names its `requirement` and kind:
90
+
91
+ - `file`: an in-root regular `path` whose bytes equal `expected`.
92
+ - `json`: an in-root regular `path` whose parsed JSON equals `expected`.
93
+ - `command`: a literal `argv`, expected `exit_code` (default zero), and optional
94
+ expected stdout. Requires `allow_commands: true`.
95
+ - `http`: an exact `url` listed in `allowed_urls`, expected response bytes, and
96
+ expected `status` (default 200). GET only, with no redirects.
97
+
98
+ Check failures are negative evidence. Missing requirements, unavailable checks,
99
+ and timeouts stay unknown even if the LLM judge awards a high score. Later tool
100
+ execution invalidates a runner report without falling back to presumed success.
101
+ The runner records output digests, not raw command output or expected values.
102
+ Artifact reads use no-follow descriptors, validate their actual location through
103
+ Linux `/proc/self/fd`, and enforce the byte limit while reading. If descriptor
104
+ location validation is unavailable, the check stays unknown rather than using
105
+ an unsafe pathname fallback.
106
+ The caller must supply the complete acceptance checklist: matching clauses to
107
+ the original text is not semantic proof that the checklist covers every intent.
108
+
109
+ Commands have a cleared environment, private HOME set to the selected root,
110
+ bounded output and timeout, and process-group cleanup. This is **not an OS
111
+ sandbox**: opt-in commands retain the process's filesystem/network permissions.
112
+ Only run trusted checks, using an external sandbox for untrusted programs.
113
+
114
+ At the final boundary, Loop executes the contract and publishes a verification
115
+ event through `on_tool`. Around dispatched actions it snapshots declared
116
+ artifact digests. Matching the final checked digest to its last observed writer
117
+ produces an `independent_verifier` attribution receipt. This establishes artifact
118
+ provenance, not universal causal proof. Controlled comparisons in isolated
119
+ evaluation can supply a separate `controlled_comparison` receipt.
120
+
121
+ Policy distributes the existing terminal reward across uniquely linked action
122
+ IDs. Unlinked actions receive no positive training credit; an accounting-only
123
+ final row retains unattributed terminal reward. Adding successful no-op commands
124
+ does not create more reward. Known per-step cost remains; unknown task outcomes
125
+ still do not train. Ordinary model-judged runs without attribution may retain
126
+ outcome/lesson records, but do not positively reinforce guessed tool contributions.
127
+
128
+ `Loop.run(trusted_context:)` accepts host observations, not model arguments.
129
+ Policy stores only fixed environment/capability/failure/verification categories;
130
+ state backoff stays within that observed scope. Registry checks declared tool
131
+ prerequisites before ranking, including core tools, so a known absent prerequisite
132
+ cannot be outweighed by a historical success score. Unknown availability is not
133
+ treated as absence. Default local scope observes the Ruby runtime and `/bin/sh`;
134
+ callers must supply observations for remote/container-specific capabilities.
135
+
136
+ Policy observations also capture allowlisted operation, argument-role/type
137
+ features, and result classification, without retaining raw argument values.
138
+ These condition next-tool ranking only after enough contextual samples exist,
139
+ with backoff to broad tool scores. Ranking remains advisory.
140
+
141
+ For an independently checked learning-on/off experiment, see
142
+ [Policy-Benchmark.md](Policy-Benchmark.md). It exercises real local handlers and
143
+ controller updates on separate training and held-out inputs; it does not prove
144
+ that a live language model performs better.
145
+
68
146
  | Method | What it does |
69
147
  |--------|--------------|
70
148
  | **R1** `.judge` | Cheap LLM outcome score on `(request, final)` -> `{score:0..1, verdict:, rationale:, key_step:, source:}`. Calls the active engine `.chat` with a short timeout (default 12s). `Reflect.on` is used only when `module_reflection` is on. Fallback scores completeness, plan cover, claims, and tool-trace echo. Token overlap is only a small on-topic gate. |
@@ -147,7 +225,7 @@ This is the live control list. Track the outcomes, not source comments.
147
225
  |-------|------|
148
226
  | **E1** `Metrics.changepoints` + `Loop.attribute_cause` | Env-drift-attributed failures get `cause: :env_drift` and do not inflate `[REPEATING]`. |
149
227
  | **E2** `Extrospection.correlate` | Lead-lag style joins ("tool X started failing after toolchain Y changed"). |
150
- | **E3** `Reward.verify_as_reward` | Browser-backed claim checks can floor/cap the outcome score. |
228
+ | **E3** `Reward.verify_as_reward` | Browser-backed claim checks provide diagnostics and refutation caps; one confirmed claim cannot promote whole-request success. |
151
229
 
152
230
  ## Config (`PWN::Env[:ai][:agent]`)
153
231
 
@@ -39,6 +39,7 @@ PWN::AI::Agent::Engagement.open(opts)
39
39
  - `current_name`
40
40
  - `in_scope`
41
41
  - `deny_if_out_of_scope`
42
+ - `load_roe`
42
43
  - `authors`
43
44
  - `help`
44
45
  - `in_scope?`
@@ -39,6 +39,9 @@ PWN::AI::Agent::Metrics.load(opts)
39
39
  - `append_jsonl`
40
40
  - `summary`
41
41
  - `snapshot`
42
+ - `record_tokens`
43
+ - `usage`
44
+ - `routing`
42
45
  - `to_context`
43
46
  - `proxy_trust`
44
47
  - `ucb`
@@ -12,7 +12,7 @@ metadata:
12
12
 
13
13
  # PWN::AI::Agent::Policy
14
14
 
15
- PWN::AI::Agent::Policy is the LIVE tabular RL controller that pwn-ai did not have before R5. Everything else in the harness is retrieval-plus-policy: scores are written to disk and re-injected as prose, or exported later for optional LoRA. This module is the missing MDP: state s — discretized (kind, task, plan, completeness, usable, last, fail) action a — tool name, or "final" reward r — step: 0 (spam cost −0.01 after 8 tools); terminal: judge × confidence next s' — state after the tool result Each Loop turn is one episode. Transitions land in ~/.pwn/policy_traj.jsonl. Q(s,a) and REINFORCE logits H(s,a) are updated from those tuples and persisted in ~/.pwn/policy.json. The learned Q values are an ADVISORY term in Registry.rank. They never replace TaskSummarizer planning, plan_first, or CORE_TOOLS. Disable with PWN::Env[:ai][:agent][:policy] = false.
15
+ PWN::AI::Agent::Policy is the LIVE tabular RL controller that pwn-ai did not have before R5. Everything else in the harness is retrieval-plus-policy: scores are written to disk and re-injected as prose, or exported later for optional LoRA. This module is the missing MDP: state s — discretized (kind, task, plan, completeness, usable, last, fail) action a — tool name, or "final" reward r — step: 0 (spam cost −0.01 after 8 tools); terminal: judge × confidence next s' — state after the tool result Each Loop turn is one episode. Trusted environment/prerequisite bins scope fallback history; independently evidenced terminal attribution uses isolated action targets, not future-return credit for busywork. Transitions land in ~/.pwn/policy_traj.jsonl. Q(s,a) and REINFORCE logits H(s,a) are updated from those tuples and persisted in ~/.pwn/policy.json. The learned Q values are an ADVISORY term in Registry.rank. They never replace TaskSummarizer planning, plan_first, or CORE_TOOLS. Disable with PWN::Env[:ai][:agent][:policy] = false.
16
16
 
17
17
  ## When to use
18
18
 
@@ -34,6 +34,8 @@ PWN::AI::Agent::Policy.state(opts)
34
34
  ## Public methods
35
35
 
36
36
  - `state`
37
+ - `observed_state`
38
+ - `observed_context`
37
39
  - `cold`
38
40
  - `warm`
39
41
  - `episode_budget_met`
@@ -48,6 +50,7 @@ PWN::AI::Agent::Policy.state(opts)
48
50
  - `recommend`
49
51
  - `current_state`
50
52
  - `current_episode`
53
+ - `current_context_state`
51
54
  - `detach_episode`
52
55
  - `attach_episode`
53
56
  - `load`
@@ -0,0 +1,49 @@
1
+ ---
2
+ name: pwn-ai-agent-policyevaluation
3
+ description: Drive PWN::AI::Agent::PolicyEvaluation from pwn_eval.
4
+ license: MIT
5
+ allowed-tools: [pwn, pwn_eval]
6
+ metadata:
7
+ bundled: true
8
+ generated: true
9
+ module: PWN::AI::Agent::PolicyEvaluation
10
+ source: pwn/ai/agent/policy_evaluation.rb
11
+ ---
12
+
13
+ # PWN::AI::Agent::PolicyEvaluation
14
+
15
+ Opt-in, fixed local held-out evaluation. Never loaded by the online loop.
16
+
17
+ ## When to use
18
+
19
+ Call `PWN::AI::Agent::PolicyEvaluation` from `pwn_eval` when the task needs this module.
20
+ Do not reimplement it in shell.
21
+
22
+ ## Methodologies
23
+
24
+ Generated from `pwn/ai/agent/policy_evaluation.rb`. Prefer the public class methods below.
25
+ Class methods take `(opts = {})` and read `opts`.
26
+
27
+ ## How to call
28
+
29
+ ```ruby
30
+ PWN::AI::Agent::PolicyEvaluation.help
31
+ PWN::AI::Agent::PolicyEvaluation.evaluate(opts)
32
+ ```
33
+
34
+ ## Public methods
35
+
36
+ - `evaluate`
37
+ - `authors`
38
+ - `promote`
39
+ - `rollback`
40
+ - `help`
41
+
42
+ ## Source
43
+
44
+ `pwn/ai/agent/policy_evaluation.rb`
45
+
46
+ ## Verification
47
+
48
+ `PWN::AI::Agent::PolicyEvaluation.respond_to?(:evaluate)` after the
49
+ module is loaded. Read the source for parameter names.
@@ -44,8 +44,10 @@ PWN::AI::Agent::Registry.register(opts)
44
44
  - `discover`
45
45
  - `eager_load`
46
46
  - `selftest`
47
+ - `available`
47
48
  - `authors`
48
49
  - `help`
50
+ - `available?`
49
51
  - `eager_load!`
50
52
 
51
53
  ## Source
@@ -34,6 +34,9 @@ PWN::AI::Agent::Reward.judge(opts)
34
34
  ## Public methods
35
35
 
36
36
  - `judge`
37
+ - `resolve_outcome`
38
+ - `run_verification`
39
+ - `record_verification`
37
40
  - `promote_to_success`
38
41
  - `prm`
39
42
  - `plan_coverage`
@@ -47,6 +47,12 @@ PWN::AI::Agent::Swarm.personas(opts)
47
47
  - `fact_record`
48
48
  - `facts_prompt`
49
49
  - `claim`
50
+ - `pack_specialist`
51
+ - `child_inbox`
52
+ - `child_honesty`
53
+ - `honesty_unmet`
54
+ - `view_graph`
55
+ - `migrate_personas`
50
56
  - `authors`
51
57
  - `help`
52
58