pwn 0.5.722 → 0.5.724
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/Gemfile +2 -2
- data/bin/pwn_setup +5 -5
- data/documentation/AI-Integration.md +34 -1
- data/documentation/Policy-Benchmark.md +296 -0
- data/documentation/Reinforcement-Learning.md +81 -3
- data/etc/default_skills/pwn/ai/agent/engagement/SKILL.md +1 -0
- data/etc/default_skills/pwn/ai/agent/metrics/SKILL.md +3 -0
- data/etc/default_skills/pwn/ai/agent/policy/SKILL.md +4 -1
- data/etc/default_skills/pwn/ai/agent/policy_evaluation/SKILL.md +49 -0
- data/etc/default_skills/pwn/ai/agent/registry/SKILL.md +2 -0
- data/etc/default_skills/pwn/ai/agent/reward/SKILL.md +3 -0
- data/etc/default_skills/pwn/ai/agent/swarm/SKILL.md +6 -0
- data/etc/default_skills/pwn/ai/agent/tools/capabilities/SKILL.md +45 -0
- data/etc/default_skills/pwn/ai/agent/tools/context/SKILL.md +45 -0
- data/etc/default_skills/pwn/ai/agent/verification/SKILL.md +48 -0
- data/etc/default_skills/pwn/ai/context/SKILL.md +50 -0
- data/etc/default_skills/pwn/ai/http_retry/SKILL.md +7 -0
- data/etc/default_skills/pwn/ai/http_retry/references/urls.md +4 -0
- data/etc/default_skills/pwn/ai/open_ai/SKILL.md +1 -0
- data/etc/default_skills/pwn/ai/open_ai/references/urls.md +1 -0
- data/etc/default_skills/pwn/plugins/findings/SKILL.md +1 -0
- data/etc/default_skills/pwn/plugins/gdbmi/SKILL.md +55 -0
- data/etc/default_skills/pwn/plugins/ghidra_headless/SKILL.md +49 -0
- data/etc/default_skills/pwn/plugins/jobs/SKILL.md +5 -0
- data/etc/default_skills/pwn/plugins/packet/SKILL.md +3 -0
- data/etc/default_skills/pwn/plugins/preflight_checker/SKILL.md +1 -0
- data/etc/default_skills/pwn/plugins/radare2/SKILL.md +1 -0
- data/etc/default_skills/pwn/plugins/transparent_browser/SKILL.md +3 -0
- data/etc/default_skills/pwn/reports/engagement/SKILL.md +3 -2
- data/lib/pwn/ai/agent/curriculum.rb +37 -45
- data/lib/pwn/ai/agent/engagement.rb +59 -0
- data/lib/pwn/ai/agent/learning.rb +85 -52
- data/lib/pwn/ai/agent/loop.rb +72 -9
- data/lib/pwn/ai/agent/metrics.rb +52 -2
- data/lib/pwn/ai/agent/policy.rb +288 -49
- data/lib/pwn/ai/agent/policy_evaluation.rb +230 -0
- data/lib/pwn/ai/agent/registry.rb +44 -4
- data/lib/pwn/ai/agent/reward.rb +188 -46
- data/lib/pwn/ai/agent/swarm.rb +235 -35
- data/lib/pwn/ai/agent/tool_guard.rb +13 -1
- data/lib/pwn/ai/agent/tools/artifacts.rb +42 -0
- data/lib/pwn/ai/agent/tools/capabilities.rb +19 -0
- data/lib/pwn/ai/agent/tools/context.rb +38 -0
- data/lib/pwn/ai/agent/tools/finding_record.rb +18 -0
- data/lib/pwn/ai/agent/tools/fuzz_campaign.rb +10 -1
- data/lib/pwn/ai/agent/tools/job_run.rb +32 -0
- data/lib/pwn/ai/agent/tools/learning.rb +5 -6
- data/lib/pwn/ai/agent/tools/metrics.rb +16 -0
- data/lib/pwn/ai/agent/tools/pty_session.rb +4 -4
- data/lib/pwn/ai/agent/tools/ruby_eval.rb +6 -5
- data/lib/pwn/ai/agent/tools/shell.rb +10 -1
- data/lib/pwn/ai/agent/tools/skills.rb +30 -0
- data/lib/pwn/ai/agent/tools/swarm.rb +8 -2
- data/lib/pwn/ai/agent/verification.rb +202 -0
- data/lib/pwn/ai/agent.rb +2 -0
- data/lib/pwn/ai/context.rb +193 -0
- data/lib/pwn/ai/http_retry.rb +53 -7
- data/lib/pwn/ai/open_ai.rb +315 -45
- data/lib/pwn/ai.rb +1 -0
- data/lib/pwn/migrate.rb +10 -1
- data/lib/pwn/plugins/artifact_registry.rb +13 -6
- data/lib/pwn/plugins/findings.rb +48 -8
- data/lib/pwn/plugins/gdbmi.rb +128 -0
- data/lib/pwn/plugins/ghidra_headless.rb +104 -0
- data/lib/pwn/plugins/jobs.rb +53 -0
- data/lib/pwn/plugins/packet.rb +51 -0
- data/lib/pwn/plugins/preflight_checker.rb +29 -0
- data/lib/pwn/plugins/process_tube.rb +24 -7
- data/lib/pwn/plugins/radare2.rb +14 -2
- data/lib/pwn/plugins/transparent_browser.rb +64 -0
- data/lib/pwn/plugins.rb +2 -0
- data/lib/pwn/reports/engagement.rb +19 -0
- data/lib/pwn/sessions.rb +3 -1
- data/lib/pwn/version.rb +1 -1
- data/scripts/benchmark_policy.rb +376 -0
- data/spec/documentation/installation_md_spec.rb +18 -4
- data/spec/integration/reinforced_feedback_loop_spec.rb +33 -19
- data/spec/lib/pwn/ai/agent/curriculum_spec.rb +267 -0
- data/spec/lib/pwn/ai/agent/engagement_spec.rb +12 -0
- data/spec/lib/pwn/ai/agent/learning_spec.rb +100 -3
- data/spec/lib/pwn/ai/agent/loop_spec.rb +104 -0
- data/spec/lib/pwn/ai/agent/metrics_spec.rb +44 -0
- data/spec/lib/pwn/ai/agent/policy_evaluation_spec.rb +221 -0
- data/spec/lib/pwn/ai/agent/policy_spec.rb +246 -6
- data/spec/lib/pwn/ai/agent/registry_spec.rb +123 -0
- data/spec/lib/pwn/ai/agent/reward_spec.rb +196 -12
- data/spec/lib/pwn/ai/agent/swarm_spec.rb +121 -1
- data/spec/lib/pwn/ai/agent/tool_guard_spec.rb +6 -0
- data/spec/lib/pwn/ai/agent/tools/capabilities_spec.rb +14 -0
- data/spec/lib/pwn/ai/agent/tools/context_spec.rb +14 -0
- data/spec/lib/pwn/ai/agent/tools/job_run_spec.rb +2 -0
- data/spec/lib/pwn/ai/agent/tools/learning_spec.rb +25 -0
- data/spec/lib/pwn/ai/agent/verification_spec.rb +172 -0
- data/spec/lib/pwn/ai/context_spec.rb +48 -0
- data/spec/lib/pwn/ai/http_retry_spec.rb +27 -0
- data/spec/lib/pwn/ai/open_ai_oauth_transport_spec.rb +245 -0
- data/spec/lib/pwn/ai/open_ai_spec.rb +235 -0
- data/spec/lib/pwn/migrate_spec.rb +24 -0
- data/spec/lib/pwn/plugins/artifact_registry_spec.rb +10 -0
- data/spec/lib/pwn/plugins/findings_spec.rb +2 -0
- data/spec/lib/pwn/plugins/gdbmi_spec.rb +17 -0
- data/spec/lib/pwn/plugins/ghidra_headless_spec.rb +17 -0
- data/third_party/pwn_rdoc.jsonl +100 -4
- metadata +30 -5
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: b1c409d6b05e3c061837ef187c74d22ff5e24d361aee4531be0943714a8f713b
|
|
4
|
+
data.tar.gz: dacfce51246aee0f02c84b5a8eae1651fe2853db2e0f192343c5fc8cfbc5462b
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 6c29fd2cde7775f1923249f166af195dcc99ed0614c86536d3aeb56bcc7a3d29a28a3483b1b2c25bb736b2b78b28ab92a0906d690cd71d002ff335c41e1e5d78
|
|
7
|
+
data.tar.gz: cbcf28e1681df68a21a1da6f247bf645de861ee2a752af2f3735f09f2a59501bdd9d5fb5bf238dcc39ae2734ea173a3d515ddb3df1de2d7548de4fbbb67c5ea7
|
data/Gemfile
CHANGED
|
@@ -49,7 +49,7 @@ gem 'jwt', '3.2.0'
|
|
|
49
49
|
gem 'libusb', '0.8.0'
|
|
50
50
|
gem 'luhn', '3.0.0'
|
|
51
51
|
gem 'mail', '2.9.1'
|
|
52
|
-
gem 'mcp', '1.
|
|
52
|
+
gem 'mcp', '1.5.0'
|
|
53
53
|
gem 'meshtastic', '0.0.175'
|
|
54
54
|
gem 'metasm', '1.0.6'
|
|
55
55
|
gem 'mongo', '2.25.0'
|
|
@@ -77,7 +77,7 @@ gem 'rbvmomi2', '3.10.0'
|
|
|
77
77
|
gem 'rdoc', '7.0.4'
|
|
78
78
|
gem 'rest-client', '2.1.0'
|
|
79
79
|
gem 'rex', '2.0.13'
|
|
80
|
-
gem 'rmagick', '7.1.
|
|
80
|
+
gem 'rmagick', '7.1.4'
|
|
81
81
|
gem 'rqrcode', '3.2.0'
|
|
82
82
|
gem 'rspec', '3.13.2'
|
|
83
83
|
gem 'rtesseract', '3.1.4'
|
data/bin/pwn_setup
CHANGED
|
@@ -25,8 +25,10 @@ pwn_driver = PWN::Driver::Parser.new do |options|
|
|
|
25
25
|
end
|
|
26
26
|
|
|
27
27
|
options.on('-l', '--list-profiles',
|
|
28
|
-
'<Optional - List capability profiles and exit>') do
|
|
29
|
-
|
|
28
|
+
'<Optional - List capability profiles and exit>') do
|
|
29
|
+
# Public metadata, like --help: no environment decryption or bootstrap.
|
|
30
|
+
PWN::Setup.list_profiles
|
|
31
|
+
exit 0
|
|
30
32
|
end
|
|
31
33
|
|
|
32
34
|
options.on('-m', '--migrate',
|
|
@@ -66,9 +68,7 @@ pwn_driver.parse!
|
|
|
66
68
|
begin
|
|
67
69
|
status = 0
|
|
68
70
|
|
|
69
|
-
if opts[:
|
|
70
|
-
PWN::Setup.list_profiles
|
|
71
|
-
elsif opts[:migrate]
|
|
71
|
+
if opts[:migrate]
|
|
72
72
|
r = PWN::Setup.migrate(
|
|
73
73
|
fix: opts[:fix] || opts[:yes],
|
|
74
74
|
dry_run: opts[:dry_run]
|
|
@@ -10,7 +10,7 @@ agent code never cares which model is behind it.
|
|
|
10
10
|
|
|
11
11
|
| Engine | Client | Auth | Notes |
|
|
12
12
|
|---|---|---|---|
|
|
13
|
-
| `openai` | `PWN::AI::OpenAI` | `key:` |
|
|
13
|
+
| `openai` | `PWN::AI::OpenAI` | ChatGPT/Codex `oauth:` or `key:` | OAuth preferred when configured; subscription chat uses the Codex Responses backend, API keys use the Platform API |
|
|
14
14
|
| `anthropic` | `PWN::AI::Anthropic` | `key:` | tool-use native |
|
|
15
15
|
| `grok` | `PWN::AI::Grok` | `key:` **or** `oauth: true` | OAuth = RFC-8628 device-code flow using xAI's public Grok-CLI client id (no secret) - see skill `xai_grok_oauth_device_flow` |
|
|
16
16
|
| `gemini` | `PWN::AI::Gemini` | `key:` | function-calling native |
|
|
@@ -38,6 +38,39 @@ ai:
|
|
|
38
38
|
PWN::Env[:ai][:active] = :ollama
|
|
39
39
|
```
|
|
40
40
|
|
|
41
|
+
## OpenAI OAuth enrollment and persistence
|
|
42
|
+
|
|
43
|
+
To prefer subscription OAuth, set `ai.openai.oauth.enroll: true` through
|
|
44
|
+
`pwn-vault`, then restart `pwn-ai` and follow the device-code consent prompt.
|
|
45
|
+
An existing bearer or refresh token is preferred over `ai.openai.key`.
|
|
46
|
+
The account must have access to the requested model through Codex; OAuth
|
|
47
|
+
does not grant all Platform API permissions or bypass subscription limits.
|
|
48
|
+
|
|
49
|
+
You can also enroll explicitly from a PWN Ruby console:
|
|
50
|
+
|
|
51
|
+
```ruby
|
|
52
|
+
PWN::AI::OpenAI.obtain_oauth_bearer_token; nil
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
The trailing `nil` prevents the console from echoing the returned bearer.
|
|
56
|
+
Successful enrollment updates the live environment and calls the private
|
|
57
|
+
`persist_oauth_to_vault` helper, as token refresh does. It uses the configured
|
|
58
|
+
`driver_opts.pwn_env_path` and `pwn_dec_path`, defaulting to
|
|
59
|
+
`~/.pwn/pwn.yaml` and its `.decryptor` file. Existing decryption artifacts are
|
|
60
|
+
required; missing artifacts leave tokens in the session only and produce a
|
|
61
|
+
notice. Tokens are not printed by the enrollment success message.
|
|
62
|
+
|
|
63
|
+
Persistence retains unrelated settings and the existing key/IV, encrypts a
|
|
64
|
+
private temporary file, and replaces the vault only after encryption succeeds.
|
|
65
|
+
The resulting vault has mode `0600`. The original vault remains unchanged on
|
|
66
|
+
a persistence failure; the helper does not create or rotate decryptor secrets.
|
|
67
|
+
|
|
68
|
+
Subscription OAuth requests must not go to the default Platform
|
|
69
|
+
`/v1/responses` endpoint: that mismatch can produce `Missing scopes:
|
|
70
|
+
api.responses.write`. OAuth chat uses
|
|
71
|
+
`https://chatgpt.com/backend-api/codex/responses` instead. An actual Platform
|
|
72
|
+
API key still needs the relevant project and endpoint permissions.
|
|
73
|
+
|
|
41
74
|
## Engine-aware behavior
|
|
42
75
|
|
|
43
76
|
The harness adapts to the *class* of engine, not the model name:
|
|
@@ -0,0 +1,296 @@
|
|
|
1
|
+
# Independent local Policy/Registry benchmark
|
|
2
|
+
|
|
3
|
+
This is a **deterministic controller benchmark, not proof of live LLM improvement**.
|
|
4
|
+
It executes real local file tasks and trains the actual tabular
|
|
5
|
+
`PWN::AI::Agent::Policy` implementation, then selects actions through the actual
|
|
6
|
+
`Registry.rank`. No model responses, provider usage, or benchmark gains are
|
|
7
|
+
simulated. The action handlers are explicitly hand-written local algorithms,
|
|
8
|
+
not a simulated language model.
|
|
9
|
+
|
|
10
|
+
## Run from a source checkout
|
|
11
|
+
|
|
12
|
+
```sh
|
|
13
|
+
ruby scripts/benchmark_policy.rb --self-check
|
|
14
|
+
ruby scripts/benchmark_policy.rb --output /tmp/pwn-policy-benchmark.json
|
|
15
|
+
bundle exec rubocop scripts/benchmark_policy.rb
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
The experiment uses Ruby and its standard libraries; it does not require the
|
|
19
|
+
full application boot sequence. Use a fresh Ruby process, not the live agent
|
|
20
|
+
console. JSON is printed to stdout and optionally written to `--output`.
|
|
21
|
+
`--self-check` runs independent scorer tests, both complete experimental arms,
|
|
22
|
+
and two fresh snapshot workers; it prints short pass messages instead of a
|
|
23
|
+
report. Use the options separately.
|
|
24
|
+
The benchmark exits nonzero if persistence changes during evaluation, splits
|
|
25
|
+
overlap, a negative control succeeds, or a positive control fails. Self-checks
|
|
26
|
+
also require actual training updates in the on arm and none in the off arm.
|
|
27
|
+
They deliberately do **not** require an improvement of a predetermined size.
|
|
28
|
+
|
|
29
|
+
## Relation to existing evaluation
|
|
30
|
+
|
|
31
|
+
`lib/pwn/ai/agent/curriculum.rb` provides live self-play, judge-based practice,
|
|
32
|
+
mistake-derived evaluation prompts, and adapter training/promotion gates. This
|
|
33
|
+
standalone script does not invoke those paths. It provides a small, independently
|
|
34
|
+
scored experiment where the training labels come from actual task achievement,
|
|
35
|
+
not `Reward.judge`, model prose, or a previously assigned reward. It calls
|
|
36
|
+
`Policy.begin_episode`, `observe_step`, and `finish` with those labels during
|
|
37
|
+
training. This is not a test of Reward's verification-record binding, the live
|
|
38
|
+
Loop, Metrics learning, memory retrieval, or adapter training.
|
|
39
|
+
|
|
40
|
+
The experimental driver remains in `scripts/`. The opt-in `PolicyEvaluation`
|
|
41
|
+
module launches only this fixed runner; it does not load the live agent in its
|
|
42
|
+
workers. Its module autoload does not enable evaluation or promotion.
|
|
43
|
+
|
|
44
|
+
## Protocol
|
|
45
|
+
|
|
46
|
+
1. **Isolation.** Create a private `/tmp/pwn-policy-benchmark-*` directory. Clear
|
|
47
|
+
the process environment, set `HOME` and `TMPDIR` to that directory, then load
|
|
48
|
+
only Policy and Registry. Assert their persistence constants resolve below
|
|
49
|
+
the temporary `HOME/.pwn`. Do not load user configuration, credentials,
|
|
50
|
+
registered application tools, providers, or network libraries. Delete the
|
|
51
|
+
entire temporary directory on normal completion or Ruby exception and restore
|
|
52
|
+
the process environment. A force-killed process may leave its temporary files.
|
|
53
|
+
This is isolation for trusted fixed handlers, not a sandbox for untrusted code.
|
|
54
|
+
2. **Paired arms.** Run `off`, then `on`, resetting Policy between them. Both arms
|
|
55
|
+
start empty and execute the identical training schedule. The only learning
|
|
56
|
+
switch is `PWN::Env[:ai][:agent][:policy]`. No Q entries, visits, episode counts,
|
|
57
|
+
or rewards are seeded directly. Other learning modules are not loaded.
|
|
58
|
+
3. **Training only.** There are four training fixtures: two numeric sorting tasks
|
|
59
|
+
and two active-inventory filtering tasks. Six fixed exploration rounds execute
|
|
60
|
+
every one of the three candidate actions for each fixture: 72 real handler
|
|
61
|
+
calls per arm. Every action gets equal exposure; the scheduler does not use
|
|
62
|
+
answer keys to choose actions. Each call writes into a fresh task directory.
|
|
63
|
+
Exact artifact correctness supplies a binary training label to Policy; the
|
|
64
|
+
off arm executes the same work but Policy declines to update.
|
|
65
|
+
Each call has an explicit action ID. Only after the independent artifact
|
|
66
|
+
check, `finish` receives a `controlled_comparison` attribution receipt naming
|
|
67
|
+
that ID. This is a one-action isolated experiment, not a claim that the last
|
|
68
|
+
action in an arbitrary live trace caused the outcome.
|
|
69
|
+
4. **Held-out evaluation.** Only after training, materialize eight distinct
|
|
70
|
+
held-out inputs with literal answer keys: four numeric and four inventory
|
|
71
|
+
tasks. Assert no training input appears in evaluation. Task families and
|
|
72
|
+
request wording are intentionally shared with training. The held-out units
|
|
73
|
+
are **input instances, not unseen families, prompts, or tools**. This is
|
|
74
|
+
in-distribution transfer within a small public fixture set, not a blind test
|
|
75
|
+
or a generalization claim about arbitrary operator requests.
|
|
76
|
+
5. **Frozen controller.** Rank the task family's three entries using
|
|
77
|
+
`Registry.rank(query: ..., entries: ..., preference: [])`. There are no
|
|
78
|
+
evaluation `begin_episode`, `observe_step`, or `finish` calls. The policy uses
|
|
79
|
+
Registry's normal fallback state rather than a fabricated state/table. Allow
|
|
80
|
+
at most three attempts, stopping only when the independent artifact checker
|
|
81
|
+
passes. The controller does not get corrective feedback or adapt between
|
|
82
|
+
attempts. Hash all persisted `.pwn` files before evaluation and after controls;
|
|
83
|
+
abort if they changed.
|
|
84
|
+
6. **Separate controls.** For each held-out input, force a claim-only handler and
|
|
85
|
+
a handler that writes a wrong JSON object. Both return convincing
|
|
86
|
+
`PASS: completed successfully...verified` prose and `ok: true`. All 16 forced
|
|
87
|
+
negative controls must fail objective scoring. Also execute the correct
|
|
88
|
+
algorithm on each input; all eight positive controls must pass. Controls are
|
|
89
|
+
reported separately and never trained on or included in evaluation rates.
|
|
90
|
+
|
|
91
|
+
## Real task behavior and independent scoring
|
|
92
|
+
|
|
93
|
+
Numeric candidates perform lexical sorting, numeric sorting, or no artifact
|
|
94
|
+
write. Inventory candidates filter active rows, include every row, or write
|
|
95
|
+
nothing. Handlers receive only input/output paths; the literal expected values
|
|
96
|
+
are not passed to them. All handlers claim success, intentionally making textual
|
|
97
|
+
claims unreliable. The checker ignores their prose and action names. It accepts
|
|
98
|
+
only a regular non-symlink output whose parsed JSON exactly equals the answer key,
|
|
99
|
+
with the source input unchanged. Missing output, invalid JSON, a plausible wrong
|
|
100
|
+
artifact, or an altered input is failure. Self-checks separately test missing,
|
|
101
|
+
wrong, prose-only, symlink, and correct artifacts.
|
|
102
|
+
|
|
103
|
+
Each family has equal keyword-fit descriptions, with preference disabled. The
|
|
104
|
+
unlearned deterministic tie-break favors lexical sort in one family and the
|
|
105
|
+
correct active filter in the other; it is not configured to lose every task.
|
|
106
|
+
Lexical sorting can genuinely solve some numeric fixtures and receives credit
|
|
107
|
+
when it does. The learned controller can change these tied rankings from
|
|
108
|
+
observed outcomes. This construction deliberately isolates the learning-to-router
|
|
109
|
+
connection; it does not measure natural-language tool-selection quality.
|
|
110
|
+
|
|
111
|
+
The answer keys and algorithms live in the same public script but do not call
|
|
112
|
+
one another. Independence here means scoring actual artifact/task achievement
|
|
113
|
+
without trusting action claims or training rewards, not process-level secrecy or
|
|
114
|
+
an external audit. Extending the task set requires reviewing literal answer keys
|
|
115
|
+
and adding both positive and negative controls before interpreting new results.
|
|
116
|
+
|
|
117
|
+
## Report definitions
|
|
118
|
+
|
|
119
|
+
Each arm contains full training/evaluation traces, fixture inputs and answer keys,
|
|
120
|
+
control traces, Policy statistics, persistence fingerprints, and split/freeze
|
|
121
|
+
checks. Source hashes identify the Policy, Registry, and harness revisions used.
|
|
122
|
+
No fixed gains are embedded in the report.
|
|
123
|
+
|
|
124
|
+
- **Completion:** tasks with at least one objectively successful attempt divided
|
|
125
|
+
by evaluated tasks. A failed attempt does not count as task completion.
|
|
126
|
+
- **Artifact check score:** each attempt earns 1.0 only for an exact correct
|
|
127
|
+
artifact with unchanged input, otherwise 0.0; phase score is its mean across
|
|
128
|
+
attempts. Report rows retain the actual artifact body as well as its hash.
|
|
129
|
+
- **False-success count/rate:** attempts claiming `ok: true` without achieving the
|
|
130
|
+
task; rate denominator is executed attempts, not tasks. This is false reporting
|
|
131
|
+
by the handler, not acceptance of that report by the independent checker.
|
|
132
|
+
- **Repeated mistakes:** every failed attempt after the first occurrence of the
|
|
133
|
+
same `(family, action, checker failure)` signature within that phase and arm.
|
|
134
|
+
This includes both repeated attempts on one task and recurrence on later
|
|
135
|
+
held-out inputs. It is not the production Mistakes store's count.
|
|
136
|
+
- **Calls:** `tool_calls` counts actual local handler invocations. Evaluation makes
|
|
137
|
+
one Registry ranking decision per attempted handler call. Training invokes
|
|
138
|
+
begin/observe/finish once per training call; the returned update reports are
|
|
139
|
+
preserved. Each separately listed control row is one additional handler call.
|
|
140
|
+
- **Elapsed:** monotonic measured seconds. Row elapsed time measures the handler;
|
|
141
|
+
phase elapsed time also includes setup, ranking/checking, and training updates
|
|
142
|
+
as applicable. Evaluation phase time excludes separately listed controls;
|
|
143
|
+
top-level elapsed includes both arms and controls but not final JSON output or
|
|
144
|
+
temporary-directory teardown. Tiny timings vary with caching and filesystem
|
|
145
|
+
load; fixed arm order is not a timing-performance study.
|
|
146
|
+
- **Cost:** zero LLM calls and zero provider tokens because no provider is invoked.
|
|
147
|
+
Monetary cost is `null` (not estimated), not a claim that local computing is
|
|
148
|
+
economically free. No token-price or electricity estimates are invented.
|
|
149
|
+
|
|
150
|
+
Training completion is descriptive coverage under forced exploration, not a
|
|
151
|
+
learned-policy score. Only the held-out evaluation rates compare the controllers.
|
|
152
|
+
Repeated executions should reproduce actions, counts, fixture hashes, and update
|
|
153
|
+
counts for the same source revisions. Wall times, temporary paths, timestamps,
|
|
154
|
+
and timestamp-bearing persistence hashes are expected to differ.
|
|
155
|
+
|
|
156
|
+
There is deliberately no external-runner plug-in: accepting arbitrary commands
|
|
157
|
+
would undermine the no-network/no-credentials guarantee. A future live-model
|
|
158
|
+
study should use a separately reviewed runner, identical model/tool budgets,
|
|
159
|
+
external objective verifiers, frozen held-out evaluation, and actual provider
|
|
160
|
+
usage records. Do not present this controller experiment as that study.
|
|
161
|
+
|
|
162
|
+
## Opt-in independent snapshot evaluation (R5)
|
|
163
|
+
|
|
164
|
+
The original commands and `off`/`on` report shape still work. To additionally
|
|
165
|
+
export snapshots and run repeated evaluations in **fresh subprocesses**, use:
|
|
166
|
+
|
|
167
|
+
```sh
|
|
168
|
+
ruby scripts/benchmark_policy.rb --heldout \
|
|
169
|
+
--snapshot-dir /tmp/pwn-policy-snapshots \
|
|
170
|
+
--output /tmp/pwn-policy-heldout.json
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
The snapshot directory must be **new**, below `/tmp`, with no symlink parents.
|
|
174
|
+
`off.json` and `on.json` are genuine Policy JSON tables after the same real
|
|
175
|
+
training schedule. Exported `updated_at` metadata is normalized to `null` for
|
|
176
|
+
reproducible snapshot digests; no Q entries or rewards are fabricated. Export
|
|
177
|
+
happens before that arm's evaluation. Neither `--heldout` nor `--snapshot-dir`
|
|
178
|
+
promotes anything or discovers/reads a real home-directory policy.
|
|
179
|
+
|
|
180
|
+
The added `heldout` array contains protocol `pwn-policy-heldout-v2`, for suite
|
|
181
|
+
indices 0 and 1. Each worker runs three frozen arms: `off` (baseline snapshot,
|
|
182
|
+
policy disabled), `baseline` (baseline enabled), and `candidate` (candidate
|
|
183
|
+
enabled). Workers never call begin/observe/finish, warmup, or reset the caller's
|
|
184
|
+
policy. Resets and snapshot installation occur only inside temporary HOME.
|
|
185
|
+
Only fixed local handlers are registered; no provider, shell tool, credential,
|
|
186
|
+
user config, or network client is loaded. Environment variables, including Ruby
|
|
187
|
+
startup hooks, are removed before spawning Ruby. Workers have a 30-second
|
|
188
|
+
deadline; stalled children are killed and reaped. Snapshot inputs must be regular
|
|
189
|
+
non-symlink files, at most 4 MiB, with valid numeric `q`, `h`, `visits`, `returns`,
|
|
190
|
+
`n_updates`, and `td_abs_sum` fields. Parent directories cannot be symlinks.
|
|
191
|
+
|
|
192
|
+
Suite indices 0..7 are bounded deterministic variations, **not random trials**.
|
|
193
|
+
Numeric inputs and separate literal answer keys are scaled by `seed + 1`;
|
|
194
|
+
inventory IDs and separate answer keys are offset by `100 * seed`. This does not
|
|
195
|
+
call a candidate algorithm to construct its answer key. Even indices use flat
|
|
196
|
+
paths and compact JSON; odd indices use nested paths containing spaces, pretty
|
|
197
|
+
JSON, and read-only inputs. Every arm runs eight tasks, 16 negative controls and
|
|
198
|
+
eight positive controls. Training fixtures remain unchanged and disjoint.
|
|
199
|
+
These are two task families and two filesystem configurations, not unseen tools
|
|
200
|
+
or broad environment generalization. Both environment types are required for
|
|
201
|
+
promotion eligibility.
|
|
202
|
+
For externally supplied snapshots, `disjoint_inputs` describes the harness's
|
|
203
|
+
fixture sets, not proof of the snapshot's training history; that history is not
|
|
204
|
+
attested by this runner.
|
|
205
|
+
|
|
206
|
+
Explicit snapshots from another controlled experiment can be evaluated without
|
|
207
|
+
booting the application:
|
|
208
|
+
|
|
209
|
+
```ruby
|
|
210
|
+
require './lib/pwn/ai/agent/policy_evaluation'
|
|
211
|
+
evaluator = PWN::AI::Agent::PolicyEvaluation
|
|
212
|
+
reports = [0, 1].map do |seed|
|
|
213
|
+
evaluator.evaluate(baseline: '/tmp/baseline.json',
|
|
214
|
+
candidate: '/tmp/candidate.json', seed: seed)
|
|
215
|
+
end
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
With fixed source revisions and snapshot bytes, snapshot reports reproduce all
|
|
219
|
+
fields except `elapsed_seconds`. They contain no wall-clock timestamps, random
|
|
220
|
+
IDs or temporary paths. The original training report still contains the timing
|
|
221
|
+
and metadata variability described above.
|
|
222
|
+
|
|
223
|
+
## Explicit promotion and rollback
|
|
224
|
+
|
|
225
|
+
This is an **operator-invoked local eligibility gate**, not automatic online
|
|
226
|
+
policy promotion. `Policy.finish` and Loop do not call it. The existing online
|
|
227
|
+
learning behavior is not redirected or promoted by this module. No live policy
|
|
228
|
+
path is defaulted, and writes require `enabled: true` **and** `quiescent: true`.
|
|
229
|
+
The latter is an operator assertion: **stop all agent processes and policy
|
|
230
|
+
writers first**. The existing Policy writer does not share a transaction lock
|
|
231
|
+
with this module; concurrent live learning or concurrent promotions are not
|
|
232
|
+
supported. Restart writers only after the operation and readback complete.
|
|
233
|
+
|
|
234
|
+
Promotion requires 2..8 reports with distinct valid suite indices covering both
|
|
235
|
+
filesystem configurations. It then **reruns each suite in a fresh worker** using
|
|
236
|
+
the specified snapshot bytes. Every non-timing report field must match the fresh
|
|
237
|
+
execution, including source/harness digests, snapshot digests, input hashes,
|
|
238
|
+
artifact bodies and hashes, action choices, scores, controls, and frozen-policy
|
|
239
|
+
checks. Hashes alone are not signatures or evidence of trusted authorship;
|
|
240
|
+
re-execution is the authority. Model-written `passed: true`, edited scores,
|
|
241
|
+
invented artifact bodies, stale source revisions, or copied duplicate reports
|
|
242
|
+
cannot substitute for those executions. Supplied elapsed times are discarded;
|
|
243
|
+
only newly measured times enter the gate.
|
|
244
|
+
|
|
245
|
+
For **every** repeated suite, compared with both baseline-on and off:
|
|
246
|
+
|
|
247
|
+
- completion and mean artifact-check score must not decrease;
|
|
248
|
+
- no previously solved individual task may become unsolved (aggregate gains
|
|
249
|
+
cannot hide a task/family regression);
|
|
250
|
+
- false-success count **and rate**, repeated mistakes and tool calls must not rise;
|
|
251
|
+
- elapsed time must be at most `baseline_seconds * 1.25 + 0.02`, a fixed local
|
|
252
|
+
jitter allowance rather than evidence of a statistically established speedup.
|
|
253
|
+
|
|
254
|
+
Each suite must also improve completion, false-success count, repeated mistakes,
|
|
255
|
+
or calls relative to baseline-on. Better training returns, more updates, or a
|
|
256
|
+
timing-only change cannot qualify. The gate fails closed when verification fails.
|
|
257
|
+
Snapshots, source digests and live-baseline bytes must still match. An explicit
|
|
258
|
+
live target must already exist and equal the evaluated baseline byte-for-byte.
|
|
259
|
+
The previous policy is saved beside it as a digest-named rollback JSON before a
|
|
260
|
+
same-directory atomic replacement and exact readback. No trajectory file is
|
|
261
|
+
modified. Preserve the returned receipt for rollback.
|
|
262
|
+
|
|
263
|
+
A disposable demonstration using the opt-in benchmark output above:
|
|
264
|
+
|
|
265
|
+
```ruby
|
|
266
|
+
require './lib/pwn/ai/agent/policy_evaluation'
|
|
267
|
+
evaluator = PWN::AI::Agent::PolicyEvaluation
|
|
268
|
+
baseline = '/tmp/pwn-policy-snapshots/off.json'
|
|
269
|
+
candidate = '/tmp/pwn-policy-snapshots/on.json'
|
|
270
|
+
reports = JSON.parse(File.read('/tmp/pwn-policy-heldout.json'), symbolize_names: true).fetch(:heldout)
|
|
271
|
+
live = '/tmp/pwn-policy-demo-live.json' # NOT the real online policy
|
|
272
|
+
File.open(live, File::WRONLY | File::CREAT | File::EXCL, 0o600) { |f| f.write(File.binread(baseline)) }
|
|
273
|
+
receipt = evaluator.promote(enabled: true, quiescent: true,
|
|
274
|
+
baseline: baseline, candidate: candidate,
|
|
275
|
+
reports: reports, live_path: live)
|
|
276
|
+
raise receipt.inspect unless receipt[:promoted]
|
|
277
|
+
restored = evaluator.rollback(enabled: true, quiescent: true,
|
|
278
|
+
live_path: live, receipt: receipt)
|
|
279
|
+
raise restored.inspect unless restored[:rolled_back]
|
|
280
|
+
```
|
|
281
|
+
|
|
282
|
+
Rollback verifies the receipt's explicit target, backup path and prior digest,
|
|
283
|
+
validates the backup schema, and refuses if the live file no longer matches the
|
|
284
|
+
promoted candidate digest. Missing, altered or symlink backups/targets fail
|
|
285
|
+
closed. Both methods default to a disabled result; rejected operations return
|
|
286
|
+
`promoted: false` or `rolled_back: false` with a reason. `evaluate` raises on an
|
|
287
|
+
invalid snapshot, worker failure, or deadline. Force-killing a worker can leave
|
|
288
|
+
its temporary directory; this is not an OS sandbox for untrusted code.
|
|
289
|
+
|
|
290
|
+
**Limit:** this gate measures only the fixed benchmark action vocabulary and
|
|
291
|
+
public task families. A real online policy containing unrelated tools may show
|
|
292
|
+
no gain and be rejected; passing does not validate those unrelated routes or
|
|
293
|
+
establish live LLM gains. A production rollout still needs separately reviewed,
|
|
294
|
+
representative objective tasks and operator judgment. Do not interpret this
|
|
295
|
+
small public held-out set as a secret test or optimize repeatedly against it
|
|
296
|
+
and then claim independent generalization.
|
|
@@ -47,10 +47,10 @@ This is the live numeric controller. It does not replace planning.
|
|
|
47
47
|
|
|
48
48
|
| Piece | What it is |
|
|
49
49
|
|---|---|
|
|
50
|
-
| State | request kind, task family, plan quality, answer completeness, usable-result, last action, fail bin, and
|
|
50
|
+
| State | request kind, task family, plan quality, answer completeness, usable-result, last action, fail bin, engine, and sanitized host-observed capability/verification scope |
|
|
51
51
|
| Action | tool name, or `final` |
|
|
52
52
|
| Step reward | 0; −0.01 per tool after 8 |
|
|
53
|
-
| Terminal reward |
|
|
53
|
+
| Terminal reward | Resolved training score mapped to −1..1 × confidence (sole large R), attributed only to independently linked actions. `plan_coverage` is a tag, not the score. |
|
|
54
54
|
| Updates | Q-learning (`alpha=0.15`, `gamma=0.85`) and REINFORCE (`alpha=0.05`). Stored trajectories replay twice on warmup so a short table is not empty advice. |
|
|
55
55
|
| Budget | Eight finished episodes (live or warmup-credited) unlock greedy suggestions. Until then the prompt omits them. |
|
|
56
56
|
| Steer | Q-advantage in `Registry.rank` once the episode budget is met; keyword fit and CORE_TOOLS still come first. Suggested actions follow `Registry.preference_order` (`ai.agent.tool_preference`). |
|
|
@@ -65,6 +65,84 @@ high-return / high-score episodes.
|
|
|
65
65
|
|
|
66
66
|
## Reward signal (`PWN::AI::Agent::Reward`)
|
|
67
67
|
|
|
68
|
+
`Reward.resolve_outcome` is the shared decision used by live learning and
|
|
69
|
+
offline practice. The ledger retains the raw judge score, source, confidence,
|
|
70
|
+
verdict, and verification evidence. Heuristic guesses, evaluator errors, and
|
|
71
|
+
unresolved high-score/critic disagreements are unverified, with a nil
|
|
72
|
+
`training_score`; they do not become policy updates, failures, or supervised
|
|
73
|
+
success examples. Requeuing a disputed result rejudges its original session
|
|
74
|
+
instead of raising its score automatically.
|
|
75
|
+
|
|
76
|
+
Writing “PASS” in an answer, returning exit zero, or confirming one claim does
|
|
77
|
+
not establish completion. A trusted host verifier can call
|
|
78
|
+
`Reward.record_verification` after checking every original-request criterion,
|
|
79
|
+
providing actual boolean check results and evidence. Records are bound to the
|
|
80
|
+
current session request and invalidated by subsequent tool execution. The API
|
|
81
|
+
trusts the host verifier to cover the complete request; it cannot infer missing
|
|
82
|
+
criteria or automatically verify arbitrary tasks. LLM judgments remain fallible.
|
|
83
|
+
|
|
84
|
+
### Executed acceptance checks and artifact attribution
|
|
85
|
+
|
|
86
|
+
For stronger verification, pass a host-owned `verification_contract` to
|
|
87
|
+
`Loop.run`, or use `Reward.run_verification(request:, session_id:, contract:)`.
|
|
88
|
+
The contract contains `root`, `requirements` (distinct verbatim clauses of the
|
|
89
|
+
original request), and `checks`. Each check names its `requirement` and kind:
|
|
90
|
+
|
|
91
|
+
- `file`: an in-root regular `path` whose bytes equal `expected`.
|
|
92
|
+
- `json`: an in-root regular `path` whose parsed JSON equals `expected`.
|
|
93
|
+
- `command`: a literal `argv`, expected `exit_code` (default zero), and optional
|
|
94
|
+
expected stdout. Requires `allow_commands: true`.
|
|
95
|
+
- `http`: an exact `url` listed in `allowed_urls`, expected response bytes, and
|
|
96
|
+
expected `status` (default 200). GET only, with no redirects.
|
|
97
|
+
|
|
98
|
+
Check failures are negative evidence. Missing requirements, unavailable checks,
|
|
99
|
+
and timeouts stay unknown even if the LLM judge awards a high score. Later tool
|
|
100
|
+
execution invalidates a runner report without falling back to presumed success.
|
|
101
|
+
The runner records output digests, not raw command output or expected values.
|
|
102
|
+
Artifact reads use no-follow descriptors, validate their actual location through
|
|
103
|
+
Linux `/proc/self/fd`, and enforce the byte limit while reading. If descriptor
|
|
104
|
+
location validation is unavailable, the check stays unknown rather than using
|
|
105
|
+
an unsafe pathname fallback.
|
|
106
|
+
The caller must supply the complete acceptance checklist: matching clauses to
|
|
107
|
+
the original text is not semantic proof that the checklist covers every intent.
|
|
108
|
+
|
|
109
|
+
Commands have a cleared environment, private HOME set to the selected root,
|
|
110
|
+
bounded output and timeout, and process-group cleanup. This is **not an OS
|
|
111
|
+
sandbox**: opt-in commands retain the process's filesystem/network permissions.
|
|
112
|
+
Only run trusted checks, using an external sandbox for untrusted programs.
|
|
113
|
+
|
|
114
|
+
At the final boundary, Loop executes the contract and publishes a verification
|
|
115
|
+
event through `on_tool`. Around dispatched actions it snapshots declared
|
|
116
|
+
artifact digests. Matching the final checked digest to its last observed writer
|
|
117
|
+
produces an `independent_verifier` attribution receipt. This establishes artifact
|
|
118
|
+
provenance, not universal causal proof. Controlled comparisons in isolated
|
|
119
|
+
evaluation can supply a separate `controlled_comparison` receipt.
|
|
120
|
+
|
|
121
|
+
Policy distributes the existing terminal reward across uniquely linked action
|
|
122
|
+
IDs. Unlinked actions receive no positive training credit; an accounting-only
|
|
123
|
+
final row retains unattributed terminal reward. Adding successful no-op commands
|
|
124
|
+
does not create more reward. Known per-step cost remains; unknown task outcomes
|
|
125
|
+
still do not train. Ordinary model-judged runs without attribution may retain
|
|
126
|
+
outcome/lesson records, but do not positively reinforce guessed tool contributions.
|
|
127
|
+
|
|
128
|
+
`Loop.run(trusted_context:)` accepts host observations, not model arguments.
|
|
129
|
+
Policy stores only fixed environment/capability/failure/verification categories;
|
|
130
|
+
state backoff stays within that observed scope. Registry checks declared tool
|
|
131
|
+
prerequisites before ranking, including core tools, so a known absent prerequisite
|
|
132
|
+
cannot be outweighed by a historical success score. Unknown availability is not
|
|
133
|
+
treated as absence. Default local scope observes the Ruby runtime and `/bin/sh`;
|
|
134
|
+
callers must supply observations for remote/container-specific capabilities.
|
|
135
|
+
|
|
136
|
+
Policy observations also capture allowlisted operation, argument-role/type
|
|
137
|
+
features, and result classification, without retaining raw argument values.
|
|
138
|
+
These condition next-tool ranking only after enough contextual samples exist,
|
|
139
|
+
with backoff to broad tool scores. Ranking remains advisory.
|
|
140
|
+
|
|
141
|
+
For an independently checked learning-on/off experiment, see
|
|
142
|
+
[Policy-Benchmark.md](Policy-Benchmark.md). It exercises real local handlers and
|
|
143
|
+
controller updates on separate training and held-out inputs; it does not prove
|
|
144
|
+
that a live language model performs better.
|
|
145
|
+
|
|
68
146
|
| Method | What it does |
|
|
69
147
|
|--------|--------------|
|
|
70
148
|
| **R1** `.judge` | Cheap LLM outcome score on `(request, final)` -> `{score:0..1, verdict:, rationale:, key_step:, source:}`. Calls the active engine `.chat` with a short timeout (default 12s). `Reflect.on` is used only when `module_reflection` is on. Fallback scores completeness, plan cover, claims, and tool-trace echo. Token overlap is only a small on-topic gate. |
|
|
@@ -147,7 +225,7 @@ This is the live control list. Track the outcomes, not source comments.
|
|
|
147
225
|
|-------|------|
|
|
148
226
|
| **E1** `Metrics.changepoints` + `Loop.attribute_cause` | Env-drift-attributed failures get `cause: :env_drift` and do not inflate `[REPEATING]`. |
|
|
149
227
|
| **E2** `Extrospection.correlate` | Lead-lag style joins ("tool X started failing after toolchain Y changed"). |
|
|
150
|
-
| **E3** `Reward.verify_as_reward` | Browser-backed claim checks
|
|
228
|
+
| **E3** `Reward.verify_as_reward` | Browser-backed claim checks provide diagnostics and refutation caps; one confirmed claim cannot promote whole-request success. |
|
|
151
229
|
|
|
152
230
|
## Config (`PWN::Env[:ai][:agent]`)
|
|
153
231
|
|
|
@@ -12,7 +12,7 @@ metadata:
|
|
|
12
12
|
|
|
13
13
|
# PWN::AI::Agent::Policy
|
|
14
14
|
|
|
15
|
-
PWN::AI::Agent::Policy is the LIVE tabular RL controller that pwn-ai did not have before R5. Everything else in the harness is retrieval-plus-policy: scores are written to disk and re-injected as prose, or exported later for optional LoRA. This module is the missing MDP: state s — discretized (kind, task, plan, completeness, usable, last, fail) action a — tool name, or "final" reward r — step: 0 (spam cost −0.01 after 8 tools); terminal: judge × confidence next s' — state after the tool result Each Loop turn is one episode. Transitions land in ~/.pwn/policy_traj.jsonl. Q(s,a) and REINFORCE logits H(s,a) are updated from those tuples and persisted in ~/.pwn/policy.json. The learned Q values are an ADVISORY term in Registry.rank. They never replace TaskSummarizer planning, plan_first, or CORE_TOOLS. Disable with PWN::Env[:ai][:agent][:policy] = false.
|
|
15
|
+
PWN::AI::Agent::Policy is the LIVE tabular RL controller that pwn-ai did not have before R5. Everything else in the harness is retrieval-plus-policy: scores are written to disk and re-injected as prose, or exported later for optional LoRA. This module is the missing MDP: state s — discretized (kind, task, plan, completeness, usable, last, fail) action a — tool name, or "final" reward r — step: 0 (spam cost −0.01 after 8 tools); terminal: judge × confidence next s' — state after the tool result Each Loop turn is one episode. Trusted environment/prerequisite bins scope fallback history; independently evidenced terminal attribution uses isolated action targets, not future-return credit for busywork. Transitions land in ~/.pwn/policy_traj.jsonl. Q(s,a) and REINFORCE logits H(s,a) are updated from those tuples and persisted in ~/.pwn/policy.json. The learned Q values are an ADVISORY term in Registry.rank. They never replace TaskSummarizer planning, plan_first, or CORE_TOOLS. Disable with PWN::Env[:ai][:agent][:policy] = false.
|
|
16
16
|
|
|
17
17
|
## When to use
|
|
18
18
|
|
|
@@ -34,6 +34,8 @@ PWN::AI::Agent::Policy.state(opts)
|
|
|
34
34
|
## Public methods
|
|
35
35
|
|
|
36
36
|
- `state`
|
|
37
|
+
- `observed_state`
|
|
38
|
+
- `observed_context`
|
|
37
39
|
- `cold`
|
|
38
40
|
- `warm`
|
|
39
41
|
- `episode_budget_met`
|
|
@@ -48,6 +50,7 @@ PWN::AI::Agent::Policy.state(opts)
|
|
|
48
50
|
- `recommend`
|
|
49
51
|
- `current_state`
|
|
50
52
|
- `current_episode`
|
|
53
|
+
- `current_context_state`
|
|
51
54
|
- `detach_episode`
|
|
52
55
|
- `attach_episode`
|
|
53
56
|
- `load`
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: pwn-ai-agent-policyevaluation
|
|
3
|
+
description: Drive PWN::AI::Agent::PolicyEvaluation from pwn_eval.
|
|
4
|
+
license: MIT
|
|
5
|
+
allowed-tools: [pwn, pwn_eval]
|
|
6
|
+
metadata:
|
|
7
|
+
bundled: true
|
|
8
|
+
generated: true
|
|
9
|
+
module: PWN::AI::Agent::PolicyEvaluation
|
|
10
|
+
source: pwn/ai/agent/policy_evaluation.rb
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
# PWN::AI::Agent::PolicyEvaluation
|
|
14
|
+
|
|
15
|
+
Opt-in, fixed local held-out evaluation. Never loaded by the online loop.
|
|
16
|
+
|
|
17
|
+
## When to use
|
|
18
|
+
|
|
19
|
+
Call `PWN::AI::Agent::PolicyEvaluation` from `pwn_eval` when the task needs this module.
|
|
20
|
+
Do not reimplement it in shell.
|
|
21
|
+
|
|
22
|
+
## Methodologies
|
|
23
|
+
|
|
24
|
+
Generated from `pwn/ai/agent/policy_evaluation.rb`. Prefer the public class methods below.
|
|
25
|
+
Class methods take `(opts = {})` and read `opts`.
|
|
26
|
+
|
|
27
|
+
## How to call
|
|
28
|
+
|
|
29
|
+
```ruby
|
|
30
|
+
PWN::AI::Agent::PolicyEvaluation.help
|
|
31
|
+
PWN::AI::Agent::PolicyEvaluation.evaluate(opts)
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Public methods
|
|
35
|
+
|
|
36
|
+
- `evaluate`
|
|
37
|
+
- `authors`
|
|
38
|
+
- `promote`
|
|
39
|
+
- `rollback`
|
|
40
|
+
- `help`
|
|
41
|
+
|
|
42
|
+
## Source
|
|
43
|
+
|
|
44
|
+
`pwn/ai/agent/policy_evaluation.rb`
|
|
45
|
+
|
|
46
|
+
## Verification
|
|
47
|
+
|
|
48
|
+
`PWN::AI::Agent::PolicyEvaluation.respond_to?(:evaluate)` after the
|
|
49
|
+
module is loaded. Read the source for parameter names.
|