revoco 0.2.2__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {revoco-0.2.2 → revoco-0.2.3}/PKG-INFO +19 -2
- {revoco-0.2.2 → revoco-0.2.3}/README.md +18 -1
- {revoco-0.2.2 → revoco-0.2.3}/pyproject.toml +1 -1
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/adapters/__init__.py +3 -1
- revoco-0.2.3/src/revoco/adapters/ras_eval.py +187 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/bench/__init__.py +7 -0
- revoco-0.2.3/src/revoco/bench/external.py +325 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/cli.py +12 -0
- revoco-0.2.3/tests/test_external_corpus.py +178 -0
- {revoco-0.2.2 → revoco-0.2.3}/.github/workflows/ci.yml +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/.github/workflows/release.yml +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/.gitignore +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/LICENSE +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/docs/ADAPTERS.md +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/docs/RELEASING.md +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/examples/inverses.yaml +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/examples/inverses_cloud.json +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/examples/inverses_database.json +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/examples/inverses_devops.json +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/examples/inverses_identity.json +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/examples/inverses_saas.json +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/examples/inverses_sap.json +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/examples/inverses_workday.json +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/examples/inverses_workstation.json +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/examples/policy.yaml +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/scripts/bump_version.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/scripts/validate_workstation.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/__init__.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/adapters/cloud.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/adapters/database.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/adapters/devops.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/adapters/identity.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/adapters/saas.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/adapters/sap.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/adapters/workday.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/adapters/workstation.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/authority/__init__.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/authority/action.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/authority/delegation.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/authority/engine.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/authority/principals.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/authority/revocation.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/authority/scope.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/bench/corpus.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/bench/harness.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/bench/report.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/bench/scenario.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/bench/world.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/controlplane.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/core/__init__.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/core/crypto.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/core/errors.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/core/ids.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/demo.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/detect.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/drills.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/evidence.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/gate/__init__.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/gate/conditions.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/gate/decision.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/gate/engine.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/gate/policy.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/gate/session.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/gate/threats.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/ledger.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/reversal/__init__.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/reversal/budget.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/reversal/engine.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/reversal/horizon.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/reversal/model.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/src/revoco/reversal/registry.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/tests/test_adapter_catalog.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/tests/test_adapters.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/tests/test_authority.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/tests/test_bench.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/tests/test_budget_and_drills.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/tests/test_controlplane.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/tests/test_core.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/tests/test_demo_and_cli.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/tests/test_gate.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/tests/test_horizon_and_scheduling.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/tests/test_ledger.py +0 -0
- {revoco-0.2.2 → revoco-0.2.3}/tests/test_reversal.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: revoco
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: Undo for AI agent actions. Plans the rollback before the action runs, proves it still works, and rolls back a compromised grant's whole blast radius in one call.
|
|
5
5
|
Project-URL: Homepage, https://github.com/rsh1k/revoco
|
|
6
6
|
Project-URL: Source, https://github.com/rsh1k/revoco
|
|
@@ -234,7 +234,24 @@ Every malicious technique has a **benign twin on the same tools**, so a policy t
|
|
|
234
234
|
|
|
235
235
|
Because it runs a real `ControlPlane` against a simulated world, it doubles as a regression suite for the 91 adapter specs — and it earned that keep immediately, finding six real defects including an inverse that relied on implicit convention and a `Rule` that couldn't express "escalate irreversible work only when consequential".
|
|
236
236
|
|
|
237
|
-
|
|
237
|
+
### Importing real benign traffic
|
|
238
|
+
|
|
239
|
+
Hand-authored benign scenarios have a structural blind spot: they contain the false positives I thought to look for. So the corpus can import benign tasks from a [RAS-Eval](https://github.com/lanzer-tree/RAS-Eval) clone — traffic real models actually produced, with real arguments.
|
|
240
|
+
|
|
241
|
+
```bash
|
|
242
|
+
git clone https://github.com/lanzer-tree/RAS-Eval /path/to/RAS-Eval
|
|
243
|
+
RAS_EVAL_PATH=/path/to/RAS-Eval revoco bench --external
|
|
244
|
+
```
|
|
245
|
+
|
|
246
|
+
That takes the corpus to **121 scenarios at 5.7:1 benign-to-malicious** — essentially ADR-Bench's ratio — while holding 0% false positives.
|
|
247
|
+
|
|
248
|
+
**Nothing is vendored.** RAS-Eval declares no license, so its data is all-rights-reserved by default; the loader reads a clone you fetch yourself and returns nothing when it's absent, so CI never depends on it.
|
|
249
|
+
|
|
250
|
+
**It found two real bugs on its first run**, at a 16.2% false-positive rate the hand-authored corpus could not have surfaced — because I wrote both the spec and the scenario, and used my own invented argument names in each. `insert_data` takes `db_path`, not the `table` I inferred from the tool's name. `convert_file_to_markdown` takes a `save_path` argument rather than returning `output_path`. In both cases the inverse could never resolve, so every legitimate call raised a phantom rollback. **The same mistake in an SAP adapter would look identical and cost considerably more.**
|
|
251
|
+
|
|
252
|
+
Two disciplines keep the import honest: a tool this package hasn't classified is **skipped**, not imported as `UNKNOWN`, and a call with no observed arguments is **dropped** rather than imported — both would manufacture findings out of missing metadata rather than out of anything the control plane did. And only the 80 *unique tasks* are imported, not all 640 traces: eight models ran the same tasks, so taking every run would multiply the count with near-duplicates. Padding is the thing this corpus exists not to do.
|
|
253
|
+
|
|
254
|
+
**Honest about the comparison:** ADR-Bench's 302 tasks come from real enterprise telemetry across 133 MCP servers. The imported traffic here is consumer-domain — alarms, calendars, disk stats, arXiv lookups — so it broadens the benign distribution without reaching the enterprise write surfaces where the money is. Detection coverage is their strength; verified recoverability is this one's. Complementary instruments.
|
|
238
255
|
|
|
239
256
|
One gap is left visible rather than tuned away: `T09` irreversible fan-out. `PRA01` is a threshold detector, so four one-way wires land before the pattern is visible. The controlled pair `M10`/`M18` measures detection versus the budget on the identical attack, and both stay in the corpus so the difference is attributable.
|
|
240
257
|
|
|
@@ -201,7 +201,24 @@ Every malicious technique has a **benign twin on the same tools**, so a policy t
|
|
|
201
201
|
|
|
202
202
|
Because it runs a real `ControlPlane` against a simulated world, it doubles as a regression suite for the 91 adapter specs — and it earned that keep immediately, finding six real defects including an inverse that relied on implicit convention and a `Rule` that couldn't express "escalate irreversible work only when consequential".
|
|
203
203
|
|
|
204
|
-
|
|
204
|
+
### Importing real benign traffic
|
|
205
|
+
|
|
206
|
+
Hand-authored benign scenarios have a structural blind spot: they contain the false positives I thought to look for. So the corpus can import benign tasks from a [RAS-Eval](https://github.com/lanzer-tree/RAS-Eval) clone — traffic real models actually produced, with real arguments.
|
|
207
|
+
|
|
208
|
+
```bash
|
|
209
|
+
git clone https://github.com/lanzer-tree/RAS-Eval /path/to/RAS-Eval
|
|
210
|
+
RAS_EVAL_PATH=/path/to/RAS-Eval revoco bench --external
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
That takes the corpus to **121 scenarios at 5.7:1 benign-to-malicious** — essentially ADR-Bench's ratio — while holding 0% false positives.
|
|
214
|
+
|
|
215
|
+
**Nothing is vendored.** RAS-Eval declares no license, so its data is all-rights-reserved by default; the loader reads a clone you fetch yourself and returns nothing when it's absent, so CI never depends on it.
|
|
216
|
+
|
|
217
|
+
**It found two real bugs on its first run**, at a 16.2% false-positive rate the hand-authored corpus could not have surfaced — because I wrote both the spec and the scenario, and used my own invented argument names in each. `insert_data` takes `db_path`, not the `table` I inferred from the tool's name. `convert_file_to_markdown` takes a `save_path` argument rather than returning `output_path`. In both cases the inverse could never resolve, so every legitimate call raised a phantom rollback. **The same mistake in an SAP adapter would look identical and cost considerably more.**
|
|
218
|
+
|
|
219
|
+
Two disciplines keep the import honest: a tool this package hasn't classified is **skipped**, not imported as `UNKNOWN`, and a call with no observed arguments is **dropped** rather than imported — both would manufacture findings out of missing metadata rather than out of anything the control plane did. And only the 80 *unique tasks* are imported, not all 640 traces: eight models ran the same tasks, so taking every run would multiply the count with near-duplicates. Padding is the thing this corpus exists not to do.
|
|
220
|
+
|
|
221
|
+
**Honest about the comparison:** ADR-Bench's 302 tasks come from real enterprise telemetry across 133 MCP servers. The imported traffic here is consumer-domain — alarms, calendars, disk stats, arXiv lookups — so it broadens the benign distribution without reaching the enterprise write surfaces where the money is. Detection coverage is their strength; verified recoverability is this one's. Complementary instruments.
|
|
205
222
|
|
|
206
223
|
One gap is left visible rather than tuned away: `T09` irreversible fan-out. `PRA01` is a threshold detector, so four one-way wires land before the pattern is visible. The controlled pair `M10`/`M18` measures detection versus the budget on the identical attack, and both stay in the corpus so the difference is attributable.
|
|
207
224
|
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "revoco"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.3"
|
|
8
8
|
description = "Undo for AI agent actions. Plans the rollback before the action runs, proves it still works, and rolls back a compromised grant's whole blast radius in one call."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -48,11 +48,12 @@ from typing import Any
|
|
|
48
48
|
|
|
49
49
|
from ..reversal.model import InverseSpec, ReversalGate, Reversibility
|
|
50
50
|
from ..reversal.registry import InverseRegistry
|
|
51
|
-
from . import cloud, database, devops, identity, saas, sap, workday, workstation
|
|
51
|
+
from . import cloud, database, devops, identity, ras_eval, saas, sap, workday, workstation
|
|
52
52
|
from .cloud import CLOUD_GATES, CLOUD_SPECS, cloud_registry
|
|
53
53
|
from .database import DATABASE_GATES, DATABASE_SPECS, database_registry
|
|
54
54
|
from .devops import DEVOPS_GATES, DEVOPS_SPECS, devops_registry
|
|
55
55
|
from .identity import IDENTITY_GATES, IDENTITY_SPECS, identity_registry
|
|
56
|
+
from .ras_eval import RAS_EVAL_SPECS, ras_eval_registry
|
|
56
57
|
from .saas import SAAS_GATES, SAAS_SPECS, saas_registry
|
|
57
58
|
from .sap import SAP_GATES, SAP_SPECS, sap_registry
|
|
58
59
|
from .workday import WORKDAY_GATES, WORKDAY_SPECS, workday_registry
|
|
@@ -149,6 +150,7 @@ def summary(*surfaces: str) -> dict[str, Any]:
|
|
|
149
150
|
__all__ = [
|
|
150
151
|
# modules
|
|
151
152
|
"sap", "workday", "cloud", "identity", "devops", "saas", "workstation", "database",
|
|
153
|
+
"ras_eval", "RAS_EVAL_SPECS", "ras_eval_registry",
|
|
152
154
|
# per-surface
|
|
153
155
|
"SAP_SPECS", "SAP_GATES", "sap_registry",
|
|
154
156
|
"WORKDAY_SPECS", "WORKDAY_GATES", "workday_registry",
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
"""
|
|
2
|
+
revoco.adapters.ras_eval
|
|
3
|
+
========================
|
|
4
|
+
Reversibility classifications for the 29 tools used by the RAS-Eval benchmark.
|
|
5
|
+
|
|
6
|
+
Why this exists
|
|
7
|
+
---------------
|
|
8
|
+
The containment corpus' benign half is hand-authored, and hand-authored benign
|
|
9
|
+
traffic has a specific blind spot: it contains the false positives I thought to
|
|
10
|
+
look for. RAS-Eval (`arXiv 2506.15253 <https://arxiv.org/abs/2506.15253>`_) has 80
|
|
11
|
+
benign tasks that real models actually executed, with real arguments. That is the
|
|
12
|
+
one thing authorship cannot produce.
|
|
13
|
+
|
|
14
|
+
Its tools are not this package's tools, so the traces are unusable until something
|
|
15
|
+
says what each of them does to the world. That is what this module is — **my
|
|
16
|
+
classifications of their tool surface**, written from the tool names and the
|
|
17
|
+
observed call arguments.
|
|
18
|
+
|
|
19
|
+
No data from RAS-Eval is vendored here. The repository declares no license, so its
|
|
20
|
+
tasks and logs are all-rights-reserved by default and cannot be redistributed in an
|
|
21
|
+
Apache-2.0 package. :mod:`revoco.bench.external` reads them from a clone you obtain
|
|
22
|
+
yourself, and returns nothing when it is absent.
|
|
23
|
+
|
|
24
|
+
Status: **unvalidated, and unvalidatable from here.** These are someone else's
|
|
25
|
+
tools; I cannot execute them, so no drill can confirm any inverse below. They are
|
|
26
|
+
classified for the purpose of measuring *whether ordinary tool use gets blocked* —
|
|
27
|
+
which only needs the reversal posture to be approximately right — and should not be
|
|
28
|
+
used to govern anything.
|
|
29
|
+
|
|
30
|
+
What this surface is good for, and what it is not
|
|
31
|
+
-------------------------------------------------
|
|
32
|
+
Good for: volume and variety of ordinary agent tool use. Thirty distinct tools
|
|
33
|
+
across seven agent personas, argument shapes produced by eight different models.
|
|
34
|
+
That is a false-positive dimension the hand-built corpus does not cover at all.
|
|
35
|
+
|
|
36
|
+
Not good for: enterprise write semantics. There is no ERP posting, no payroll, no
|
|
37
|
+
IAM here — the domains are alarms, calendars, disk stats, weather and arXiv
|
|
38
|
+
lookups. It broadens the benign distribution; it does not deepen it toward the
|
|
39
|
+
surfaces where the money is.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
from __future__ import annotations
|
|
43
|
+
|
|
44
|
+
from ..reversal.model import InverseSpec, Reversibility
|
|
45
|
+
from ..reversal.registry import InverseRegistry
|
|
46
|
+
|
|
47
|
+
# Reads. Trivially reversible because they change nothing — which is also why a
|
|
48
|
+
# corpus made only of these would prove very little. Most of RAS-Eval is here.
|
|
49
|
+
_READS = (
|
|
50
|
+
"is_arxiv_identifier",
|
|
51
|
+
"search_identifier",
|
|
52
|
+
"get_disk_info",
|
|
53
|
+
"get_memory_info",
|
|
54
|
+
"get_current_time",
|
|
55
|
+
"get_weather",
|
|
56
|
+
"get_events_from_calendar",
|
|
57
|
+
"search_event",
|
|
58
|
+
"tavily_search",
|
|
59
|
+
"location_encode",
|
|
60
|
+
"get_stock_price",
|
|
61
|
+
"get_cpu_info",
|
|
62
|
+
"list_directory",
|
|
63
|
+
"read_file",
|
|
64
|
+
"get_alarms",
|
|
65
|
+
"web_search",
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
# Paired create/cancel operations. The genuinely interesting part of this surface:
|
|
69
|
+
# each is an exact inverse of the other, which is the shape the reversal model
|
|
70
|
+
# handles best and the shape most real tool surfaces lack.
|
|
71
|
+
_PAIRS: tuple[tuple[str, str, str], ...] = (
|
|
72
|
+
("add_event_to_calendar", "remove_event_from_calendar", "event_id"),
|
|
73
|
+
("remove_event_from_calendar", "add_event_to_calendar", "event_id"),
|
|
74
|
+
("set_alarm", "cancel_alarm", "alarm_id"),
|
|
75
|
+
("cancel_alarm", "set_alarm", "alarm_id"),
|
|
76
|
+
("timer", "cancel_timer", "timer_id"),
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _read_specs() -> list[InverseSpec]:
|
|
81
|
+
return [
|
|
82
|
+
InverseSpec(
|
|
83
|
+
tool=t,
|
|
84
|
+
kind=Reversibility.REVERSIBLE,
|
|
85
|
+
inverse_tool="ras.noop",
|
|
86
|
+
notes="Read-only: nothing to undo.",
|
|
87
|
+
)
|
|
88
|
+
for t in _READS
|
|
89
|
+
]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _pair_specs() -> list[InverseSpec]:
|
|
93
|
+
return [
|
|
94
|
+
InverseSpec(
|
|
95
|
+
tool=fwd,
|
|
96
|
+
kind=Reversibility.REVERSIBLE,
|
|
97
|
+
inverse_tool=inv,
|
|
98
|
+
arg_map=((key, f"result.{key}"),),
|
|
99
|
+
notes=(
|
|
100
|
+
f"Exact inverse pair with {inv}. The identifier comes from the "
|
|
101
|
+
"response, so the undo path only completes after the forward call "
|
|
102
|
+
"returns — which is precisely the deferred-argument case the plan "
|
|
103
|
+
"model exists to handle."
|
|
104
|
+
),
|
|
105
|
+
)
|
|
106
|
+
for fwd, inv, key in _PAIRS
|
|
107
|
+
]
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
RAS_EVAL_SPECS: list[InverseSpec] = [
|
|
111
|
+
*_read_specs(),
|
|
112
|
+
*_pair_specs(),
|
|
113
|
+
# Writes with an approximate inverse.
|
|
114
|
+
InverseSpec(
|
|
115
|
+
tool="insert_data",
|
|
116
|
+
kind=Reversibility.COMPENSABLE,
|
|
117
|
+
inverse_tool="delete_data",
|
|
118
|
+
# `db_path` and `items`, not `table`. The first version of this spec invented
|
|
119
|
+
# `table` from the tool's name, and the imported traces caught it: an argument
|
|
120
|
+
# that is never passed cannot resolve, so every real insert raised a phantom
|
|
121
|
+
# rollback. The same mistake in an enterprise adapter would look identical.
|
|
122
|
+
arg_map=(("db_path", "args.db_path"), ("row_id", "result.row_id")),
|
|
123
|
+
residue=(
|
|
124
|
+
"Deleting the row removes the data, not its trace: auto-increment "
|
|
125
|
+
"sequences have advanced, and anything that read the row while it existed "
|
|
126
|
+
"acted on it."
|
|
127
|
+
),
|
|
128
|
+
),
|
|
129
|
+
InverseSpec(
|
|
130
|
+
tool="download_file",
|
|
131
|
+
kind=Reversibility.COMPENSABLE,
|
|
132
|
+
inverse_tool="delete_file",
|
|
133
|
+
arg_map=(("path", "args.path"),),
|
|
134
|
+
residue=(
|
|
135
|
+
"The file is removed but the transfer happened: bandwidth was consumed and "
|
|
136
|
+
"the remote server logged the request."
|
|
137
|
+
),
|
|
138
|
+
),
|
|
139
|
+
InverseSpec(
|
|
140
|
+
tool="convert_file_to_markdown",
|
|
141
|
+
kind=Reversibility.COMPENSABLE,
|
|
142
|
+
inverse_tool="delete_file",
|
|
143
|
+
# The destination is `save_path` in the request, not `output_path` in the
|
|
144
|
+
# response — also found by importing real calls rather than by reading.
|
|
145
|
+
arg_map=(("path", "args.save_path"),),
|
|
146
|
+
residue="The converted output is removed; the source is untouched.",
|
|
147
|
+
),
|
|
148
|
+
InverseSpec(
|
|
149
|
+
tool="write_file",
|
|
150
|
+
kind=Reversibility.COMPENSABLE,
|
|
151
|
+
inverse_tool="write_file",
|
|
152
|
+
arg_map=(("path", "args.path"), ("content", "snapshot.content")),
|
|
153
|
+
snapshot_fields=("content",),
|
|
154
|
+
residue=(
|
|
155
|
+
"Prior content is restored where it was captured. If the file did not "
|
|
156
|
+
"exist before, this leaves an empty file rather than no file."
|
|
157
|
+
),
|
|
158
|
+
),
|
|
159
|
+
# Genuinely one-way, and worth registering so they escalate rather than
|
|
160
|
+
# falling into UNKNOWN by accident.
|
|
161
|
+
InverseSpec(
|
|
162
|
+
tool="send_email",
|
|
163
|
+
kind=Reversibility.IRREVERSIBLE,
|
|
164
|
+
notes="Delivered mail cannot be recalled.",
|
|
165
|
+
),
|
|
166
|
+
InverseSpec(
|
|
167
|
+
tool="execute_shell_command",
|
|
168
|
+
kind=Reversibility.UNKNOWN,
|
|
169
|
+
notes=(
|
|
170
|
+
"Same reasoning as revoco's own shell.exec: an arbitrary command's effects "
|
|
171
|
+
"cannot be known in advance, so the honest classification is UNKNOWN and "
|
|
172
|
+
"the honest outcome is escalation."
|
|
173
|
+
),
|
|
174
|
+
),
|
|
175
|
+
]
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def ras_eval_registry() -> InverseRegistry:
|
|
179
|
+
"""Classifications for the RAS-Eval tool surface (unvalidated — see module docs)."""
|
|
180
|
+
return InverseRegistry(list(RAS_EVAL_SPECS))
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def classified_tools() -> set[str]:
|
|
184
|
+
return {s.tool for s in RAS_EVAL_SPECS}
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
__all__ = ["RAS_EVAL_SPECS", "ras_eval_registry", "classified_tools"]
|
|
@@ -32,6 +32,9 @@ Usage::
|
|
|
32
32
|
"""
|
|
33
33
|
|
|
34
34
|
from .corpus import TECHNIQUES, all_scenarios, benign, by_technique, malicious
|
|
35
|
+
from .external import available as external_available
|
|
36
|
+
from .external import provenance as external_provenance
|
|
37
|
+
from .external import ras_eval_scenarios
|
|
35
38
|
from .harness import DEFAULT_POLICY, Harness, default_policy
|
|
36
39
|
from .report import Metrics, render, score, to_dict
|
|
37
40
|
from .scenario import (
|
|
@@ -56,6 +59,10 @@ __all__ = [
|
|
|
56
59
|
"malicious",
|
|
57
60
|
"benign",
|
|
58
61
|
"by_technique",
|
|
62
|
+
# external corpora (opt-in, nothing vendored)
|
|
63
|
+
"ras_eval_scenarios",
|
|
64
|
+
"external_available",
|
|
65
|
+
"external_provenance",
|
|
59
66
|
# model
|
|
60
67
|
"Scenario",
|
|
61
68
|
"Step",
|
|
@@ -0,0 +1,325 @@
|
|
|
1
|
+
"""
|
|
2
|
+
revoco.bench.external
|
|
3
|
+
=====================
|
|
4
|
+
Load benign scenarios from external benchmark data, without vendoring it.
|
|
5
|
+
|
|
6
|
+
The problem this solves
|
|
7
|
+
-----------------------
|
|
8
|
+
The hand-authored benign corpus has a structural blind spot: it contains the false
|
|
9
|
+
positives I thought to look for. Traffic that real models actually produced does
|
|
10
|
+
not, which makes it worth strictly more per scenario — and it is the one thing
|
|
11
|
+
authorship cannot manufacture.
|
|
12
|
+
|
|
13
|
+
The temptation is to generate benign scenarios with an LLM instead. That is a trap
|
|
14
|
+
the literature is clear about: systems score 84–89% on synthetic benchmarks and
|
|
15
|
+
25–34% on real-world tasks, so generated traffic would improve the *ratio* while
|
|
16
|
+
reducing what the corpus actually establishes. A number that looks better and means
|
|
17
|
+
less is the specific failure this package keeps refusing.
|
|
18
|
+
|
|
19
|
+
Why nothing is vendored
|
|
20
|
+
-----------------------
|
|
21
|
+
`RAS-Eval <https://github.com/lanzer-tree/RAS-Eval>`_ declares **no license**, so its
|
|
22
|
+
tasks and logs are all-rights-reserved by default and cannot ship inside an
|
|
23
|
+
Apache-2.0 package. This module reads them from a clone you obtain yourself and
|
|
24
|
+
returns an empty list when it is absent, so nothing here depends on data this
|
|
25
|
+
repository does not have the right to distribute.
|
|
26
|
+
|
|
27
|
+
git clone https://github.com/lanzer-tree/RAS-Eval /path/to/RAS-Eval
|
|
28
|
+
RAS_EVAL_PATH=/path/to/RAS-Eval revoco bench
|
|
29
|
+
|
|
30
|
+
What gets imported, and what deliberately does not
|
|
31
|
+
--------------------------------------------------
|
|
32
|
+
**80 unique tasks.** Not the 640 traces. RAS-Eval ships eight models' runs of the
|
|
33
|
+
same 80 tasks, so importing all of them would multiply the benign count eightfold
|
|
34
|
+
with near-duplicates — padding dressed as coverage. The *arguments* differ per
|
|
35
|
+
model, which is the interesting part, so the loader takes the union of distinct
|
|
36
|
+
(tool, argument-shape) calls across models while keeping one scenario per task.
|
|
37
|
+
|
|
38
|
+
This broadens the benign distribution rather than deepening it. RAS-Eval's domains
|
|
39
|
+
are alarms, calendars, disk stats, weather and arXiv lookups — no ERP posting, no
|
|
40
|
+
payroll, no IAM. It tests whether ordinary tool use gets blocked, which the
|
|
41
|
+
hand-built corpus does not cover; it says nothing about the enterprise write
|
|
42
|
+
surfaces where the money is. Both halves are needed and neither substitutes.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
from __future__ import annotations
|
|
46
|
+
|
|
47
|
+
import json
|
|
48
|
+
import os
|
|
49
|
+
from pathlib import Path
|
|
50
|
+
from typing import Any
|
|
51
|
+
|
|
52
|
+
from ..adapters.ras_eval import classified_tools, ras_eval_registry
|
|
53
|
+
from .scenario import GrantSpec, Intent, Scenario, Step
|
|
54
|
+
from .world import VERB_CREATE, VERB_NOOP, VERB_UPDATE, ToolBinding, World
|
|
55
|
+
|
|
56
|
+
ENV_VAR = "RAS_EVAL_PATH"
|
|
57
|
+
|
|
58
|
+
# Agent persona -> the risk band a task from it gets. RAS-Eval does not label risk,
|
|
59
|
+
# and inventing a single value for everything would make the corpus useless for
|
|
60
|
+
# testing risk-banded policy. These are graded by what the persona can touch.
|
|
61
|
+
_PERSONA_RISK = {
|
|
62
|
+
"AcademicAgent": 10,
|
|
63
|
+
"WebSearchAgent": 15,
|
|
64
|
+
"GeneralAgent": 20,
|
|
65
|
+
"StockAgent": 20,
|
|
66
|
+
"ScheduleAgent": 35,
|
|
67
|
+
"DatabaseAgent": 45,
|
|
68
|
+
"SystemAgent": 45,
|
|
69
|
+
}
|
|
70
|
+
_DEFAULT_RISK = 30
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def data_root(path: str | os.PathLike[str] | None = None) -> Path | None:
|
|
74
|
+
"""Locate a RAS-Eval clone, or return None.
|
|
75
|
+
|
|
76
|
+
Checked in order: the argument, ``$RAS_EVAL_PATH``, then a sibling directory
|
|
77
|
+
next to this repository. Returning None rather than raising is deliberate — the
|
|
78
|
+
corpus has to work without this data, or CI would depend on an unlicensed
|
|
79
|
+
third-party checkout.
|
|
80
|
+
"""
|
|
81
|
+
def has_data(p: Path) -> bool:
|
|
82
|
+
return (p / "data" / "tasks" / "tasks.json").is_file()
|
|
83
|
+
|
|
84
|
+
# An explicit path is used exclusively. Falling through to the environment when
|
|
85
|
+
# the caller named a directory would mean a typo silently loads *different* data
|
|
86
|
+
# and reports on it as though it were what you asked for.
|
|
87
|
+
if path:
|
|
88
|
+
p = Path(path)
|
|
89
|
+
return p if has_data(p) else None
|
|
90
|
+
|
|
91
|
+
candidates = []
|
|
92
|
+
if os.environ.get(ENV_VAR):
|
|
93
|
+
candidates.append(Path(os.environ[ENV_VAR]))
|
|
94
|
+
candidates.append(Path(__file__).resolve().parents[4] / "RAS-Eval")
|
|
95
|
+
for c in candidates:
|
|
96
|
+
if has_data(c):
|
|
97
|
+
return c
|
|
98
|
+
return None
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def available(path: str | os.PathLike[str] | None = None) -> bool:
|
|
102
|
+
return data_root(path) is not None
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _load_tasks(root: Path) -> list[dict[str, Any]]:
|
|
106
|
+
with (root / "data" / "tasks" / "tasks.json").open() as f:
|
|
107
|
+
tasks = json.load(f)
|
|
108
|
+
return tasks if isinstance(tasks, list) else []
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _load_observed_args(root: Path) -> dict[int, list[dict[str, Any]]]:
|
|
112
|
+
"""Per task index, the tool calls models actually made, with real arguments.
|
|
113
|
+
|
|
114
|
+
Merged across every model log present. Arguments are what make these traces
|
|
115
|
+
worth more than the task definitions: a model's idea of a plausible ``path`` or
|
|
116
|
+
``query`` is exactly the input a hand-written scenario would sanitise without
|
|
117
|
+
noticing.
|
|
118
|
+
"""
|
|
119
|
+
out: dict[int, list[dict[str, Any]]] = {}
|
|
120
|
+
logs = root / "data" / "logs"
|
|
121
|
+
if not logs.is_dir():
|
|
122
|
+
return out
|
|
123
|
+
seen: set[tuple[int, str, str]] = set()
|
|
124
|
+
for log in sorted(logs.glob("*.jsonl")):
|
|
125
|
+
# guard_response.jsonl is a detector's output, not an execution trace.
|
|
126
|
+
if "guard" in log.name:
|
|
127
|
+
continue
|
|
128
|
+
try:
|
|
129
|
+
text = log.read_text()
|
|
130
|
+
except OSError:
|
|
131
|
+
continue
|
|
132
|
+
for line in text.splitlines():
|
|
133
|
+
if not line.strip():
|
|
134
|
+
continue
|
|
135
|
+
try:
|
|
136
|
+
rec = json.loads(line)
|
|
137
|
+
except json.JSONDecodeError:
|
|
138
|
+
continue
|
|
139
|
+
idx = rec.get("index")
|
|
140
|
+
if idx is None:
|
|
141
|
+
continue
|
|
142
|
+
for msg in rec.get("response") or []:
|
|
143
|
+
for call in msg.get("tool_calls") or []:
|
|
144
|
+
name, cargs = call.get("name"), call.get("args") or {}
|
|
145
|
+
if not name:
|
|
146
|
+
continue
|
|
147
|
+
# Deduplicate by argument *shape*, not value: two models passing
|
|
148
|
+
# different arXiv ids exercise the same path, and keeping both
|
|
149
|
+
# would be the same padding this loader exists to avoid.
|
|
150
|
+
key = (idx, name, ",".join(sorted(cargs)))
|
|
151
|
+
if key in seen:
|
|
152
|
+
continue
|
|
153
|
+
seen.add(key)
|
|
154
|
+
out.setdefault(idx, []).append({"tool": name, "args": cargs})
|
|
155
|
+
return out
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _world_factory(calls: tuple[dict[str, Any], ...]) -> Any:
|
|
159
|
+
def build() -> World:
|
|
160
|
+
return _build_world(list(calls))
|
|
161
|
+
|
|
162
|
+
return build
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _needs_args(tool: str) -> bool:
|
|
166
|
+
"""Whether this tool's inverse resolves anything from the forward arguments."""
|
|
167
|
+
spec = ras_eval_registry().get(tool)
|
|
168
|
+
if spec is None:
|
|
169
|
+
return False
|
|
170
|
+
return any(
|
|
171
|
+
expr.startswith("args.")
|
|
172
|
+
for step in spec.effective_steps
|
|
173
|
+
for _name, expr in step.arg_map
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _build_world(calls: list[dict[str, Any]]) -> World:
|
|
178
|
+
"""A world that can execute whatever this task touched.
|
|
179
|
+
|
|
180
|
+
Bound generically by verb rather than modelled per tool. These scenarios measure
|
|
181
|
+
whether legitimate traffic is *allowed*, so the fidelity that matters is the
|
|
182
|
+
control plane's view of the call, not the tool's internal behaviour.
|
|
183
|
+
"""
|
|
184
|
+
w = World()
|
|
185
|
+
reg = ras_eval_registry()
|
|
186
|
+
bound: set[str] = set()
|
|
187
|
+
|
|
188
|
+
def bind(tool: str) -> None:
|
|
189
|
+
if tool in bound:
|
|
190
|
+
return
|
|
191
|
+
bound.add(tool)
|
|
192
|
+
spec = reg.get(tool)
|
|
193
|
+
kind = "ras"
|
|
194
|
+
if spec is None:
|
|
195
|
+
w.bind(ToolBinding(tool, VERB_NOOP, kind=kind, id_arg="_"))
|
|
196
|
+
return
|
|
197
|
+
if spec.effective_steps and spec.effective_steps[0].tool == "ras.noop":
|
|
198
|
+
w.bind(ToolBinding(tool, VERB_NOOP, kind=kind, id_arg="_"))
|
|
199
|
+
return
|
|
200
|
+
w.bind(ToolBinding(tool, VERB_UPDATE, kind=kind, id_arg="_",
|
|
201
|
+
generates_id=True, returns=(("row_id", "{seq}"),
|
|
202
|
+
("event_id", "{seq}"),
|
|
203
|
+
("alarm_id", "{seq}"),
|
|
204
|
+
("timer_id", "{seq}"),
|
|
205
|
+
("output_path", "{id}"))))
|
|
206
|
+
|
|
207
|
+
for c in calls:
|
|
208
|
+
bind(c["tool"])
|
|
209
|
+
spec = reg.get(c["tool"])
|
|
210
|
+
for step in (spec.effective_steps if spec else ()):
|
|
211
|
+
if step.tool != "ras.noop":
|
|
212
|
+
w.bind(ToolBinding(step.tool, VERB_CREATE, kind="ras", id_arg="_",
|
|
213
|
+
generates_id=True))
|
|
214
|
+
w.bind(ToolBinding("ras.noop", VERB_NOOP, kind="ras", id_arg="_"))
|
|
215
|
+
w.seed("ras", "_", placeholder=True)
|
|
216
|
+
return w
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def ras_eval_scenarios(
|
|
220
|
+
path: str | os.PathLike[str] | None = None, *, limit: int | None = None
|
|
221
|
+
) -> list[Scenario]:
|
|
222
|
+
"""Benign scenarios derived from RAS-Eval's 80 tasks. Empty if the data is absent.
|
|
223
|
+
|
|
224
|
+
One scenario per task, carrying the tool calls models actually made. Tasks whose
|
|
225
|
+
tools this package has no classification for are skipped rather than imported as
|
|
226
|
+
``UNKNOWN`` — importing them would manufacture false positives out of missing
|
|
227
|
+
metadata and make the corpus look worse for a reason that has nothing to do with
|
|
228
|
+
the control plane.
|
|
229
|
+
"""
|
|
230
|
+
root = data_root(path)
|
|
231
|
+
if root is None:
|
|
232
|
+
return []
|
|
233
|
+
|
|
234
|
+
tasks = _load_tasks(root)
|
|
235
|
+
observed = _load_observed_args(root)
|
|
236
|
+
known = classified_tools()
|
|
237
|
+
scenarios: list[Scenario] = []
|
|
238
|
+
|
|
239
|
+
for task in tasks:
|
|
240
|
+
raw_idx = task.get("index")
|
|
241
|
+
if raw_idx is None:
|
|
242
|
+
continue
|
|
243
|
+
idx = int(raw_idx)
|
|
244
|
+
persona = task.get("agent", "GeneralAgent")
|
|
245
|
+
workflow = list(task.get("workflow") or [])
|
|
246
|
+
calls = observed.get(idx) or [{"tool": t, "args": {}} for t in workflow]
|
|
247
|
+
calls = [c for c in calls if c["tool"] in known]
|
|
248
|
+
# Drop calls with no observed arguments where the spec needs some. A scenario
|
|
249
|
+
# built from an argument-less call cannot resolve any inverse, so it raises a
|
|
250
|
+
# phantom rollback and scores as a false positive — one manufactured out of
|
|
251
|
+
# missing trace data rather than out of anything the control plane did. Same
|
|
252
|
+
# discipline as skipping unclassified tools: absent metadata must not become
|
|
253
|
+
# a finding.
|
|
254
|
+
calls = [c for c in calls if c["args"] or not _needs_args(c["tool"])]
|
|
255
|
+
if not calls:
|
|
256
|
+
continue
|
|
257
|
+
|
|
258
|
+
risk = _PERSONA_RISK.get(persona, _DEFAULT_RISK)
|
|
259
|
+
tools = frozenset(c["tool"] for c in calls)
|
|
260
|
+
prompt = (task.get("prompt") or "").strip()
|
|
261
|
+
|
|
262
|
+
scenarios.append(
|
|
263
|
+
Scenario(
|
|
264
|
+
id=f"RAS-{idx:03d}-{persona.replace('Agent', '').lower()}",
|
|
265
|
+
title=prompt[:90] or f"{persona} task {idx}",
|
|
266
|
+
intent=Intent.BENIGN,
|
|
267
|
+
technique="EXT",
|
|
268
|
+
narrative=(
|
|
269
|
+
f"RAS-Eval task {idx} ({persona}). Imported benign traffic: the "
|
|
270
|
+
"tool calls and arguments here were produced by real models "
|
|
271
|
+
"executing the task, not written to look plausible."
|
|
272
|
+
),
|
|
273
|
+
# Bind the call list per iteration; a closure over the loop variable
|
|
274
|
+
# would give every scenario the last task's tools.
|
|
275
|
+
build_world=_world_factory(tuple(calls)),
|
|
276
|
+
grant=GrantSpec(
|
|
277
|
+
tools=tools,
|
|
278
|
+
actions=frozenset({"read", "write"}),
|
|
279
|
+
max_risk=max(risk + 15, 50),
|
|
280
|
+
purpose=prompt[:120] or f"{persona} assigned work",
|
|
281
|
+
),
|
|
282
|
+
steps=tuple(
|
|
283
|
+
Step(
|
|
284
|
+
tool=c["tool"],
|
|
285
|
+
args=dict(c["args"]),
|
|
286
|
+
action="read" if risk <= 20 else "write",
|
|
287
|
+
risk=risk,
|
|
288
|
+
description=(prompt[:70] or c["tool"]),
|
|
289
|
+
)
|
|
290
|
+
for c in calls
|
|
291
|
+
),
|
|
292
|
+
registry=ras_eval_registry(),
|
|
293
|
+
)
|
|
294
|
+
)
|
|
295
|
+
if limit and len(scenarios) >= limit:
|
|
296
|
+
break
|
|
297
|
+
return scenarios
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def provenance(path: str | os.PathLike[str] | None = None) -> dict[str, Any]:
|
|
301
|
+
"""Where the external scenarios came from, for a report to state honestly."""
|
|
302
|
+
root = data_root(path)
|
|
303
|
+
if root is None:
|
|
304
|
+
return {"source": "RAS-Eval", "available": False,
|
|
305
|
+
"note": f"set {ENV_VAR} to a clone to include these scenarios"}
|
|
306
|
+
scen = ras_eval_scenarios(path)
|
|
307
|
+
return {
|
|
308
|
+
"source": "RAS-Eval",
|
|
309
|
+
"available": True,
|
|
310
|
+
"citation": "arXiv:2506.15253",
|
|
311
|
+
"url": "https://github.com/lanzer-tree/RAS-Eval",
|
|
312
|
+
"license": "none declared — read from a local clone, never vendored",
|
|
313
|
+
"path": str(root),
|
|
314
|
+
"scenarios": len(scen),
|
|
315
|
+
"note": (
|
|
316
|
+
"80 unique tasks, not the 640 traces: eight models ran the same tasks, so "
|
|
317
|
+
"importing every run would multiply the count with near-duplicates. "
|
|
318
|
+
"Domains are alarms, calendars, disk stats, weather and arXiv lookups — "
|
|
319
|
+
"this broadens the benign distribution, it does not reach enterprise "
|
|
320
|
+
"write surfaces."
|
|
321
|
+
),
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
__all__ = ["ras_eval_scenarios", "available", "data_root", "provenance", "ENV_VAR"]
|
|
@@ -171,6 +171,15 @@ def _cmd_bench(args: argparse.Namespace) -> int:
|
|
|
171
171
|
print("no scenarios matched", file=sys.stderr)
|
|
172
172
|
return 2
|
|
173
173
|
|
|
174
|
+
if args.external:
|
|
175
|
+
from .bench.external import provenance, ras_eval_scenarios
|
|
176
|
+
|
|
177
|
+
prov = provenance()
|
|
178
|
+
extra = ras_eval_scenarios()
|
|
179
|
+
if not extra:
|
|
180
|
+
print(f"no external scenarios: {prov.get('note','')}", file=sys.stderr)
|
|
181
|
+
scenarios = scenarios + extra
|
|
182
|
+
|
|
174
183
|
results = Harness().run_all(scenarios)
|
|
175
184
|
if args.json:
|
|
176
185
|
print(json.dumps(to_dict(results, include_scenarios=args.verbose), indent=2))
|
|
@@ -268,6 +277,9 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
268
277
|
bn.add_argument("--verbose", "-v", action="store_true", help="per-step detail")
|
|
269
278
|
bn.add_argument("--json", action="store_true")
|
|
270
279
|
bn.add_argument("--technique", action="append", help="restrict to technique code(s)")
|
|
280
|
+
bn.add_argument("--external", action="store_true",
|
|
281
|
+
help="also include benign scenarios imported from a RAS-Eval clone "
|
|
282
|
+
"(set RAS_EVAL_PATH; nothing is vendored)")
|
|
271
283
|
bn.add_argument("--malicious-only", action="store_true")
|
|
272
284
|
bn.add_argument("--benign-only", action="store_true")
|
|
273
285
|
bn.set_defaults(func=_cmd_bench)
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
"""Importing benign traffic from external benchmark data.
|
|
2
|
+
|
|
3
|
+
These tests skip when no RAS-Eval clone is present, because the corpus has to work
|
|
4
|
+
without it — CI cannot depend on an unlicensed third-party checkout. What is *not*
|
|
5
|
+
skipped is the discipline: the rules that stop an import from manufacturing findings
|
|
6
|
+
out of missing data hold whether the data is there or not.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import pytest
|
|
12
|
+
|
|
13
|
+
from revoco.adapters.ras_eval import RAS_EVAL_SPECS, ras_eval_registry
|
|
14
|
+
from revoco.bench import Harness, Outcome, score
|
|
15
|
+
from revoco.bench.external import (
|
|
16
|
+
ENV_VAR,
|
|
17
|
+
available,
|
|
18
|
+
provenance,
|
|
19
|
+
ras_eval_scenarios,
|
|
20
|
+
)
|
|
21
|
+
from revoco.bench.scenario import Intent
|
|
22
|
+
from revoco.reversal import InverseSpec, Reversibility
|
|
23
|
+
|
|
24
|
+
needs_data = pytest.mark.skipif(
|
|
25
|
+
not available(), reason=f"no RAS-Eval clone; set {ENV_VAR} to include these"
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# ---------------------------------------------------------------------------
|
|
30
|
+
# Holds with or without the data
|
|
31
|
+
# ---------------------------------------------------------------------------
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def test_absent_data_yields_no_scenarios_rather_than_an_error():
|
|
35
|
+
"""The corpus must work without it, or CI depends on someone else's repo."""
|
|
36
|
+
assert ras_eval_scenarios("/nonexistent/path/for/sure") == []
|
|
37
|
+
prov = provenance("/nonexistent/path/for/sure")
|
|
38
|
+
assert prov["available"] is False
|
|
39
|
+
assert ENV_VAR in prov["note"]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def test_nothing_is_vendored():
|
|
43
|
+
"""No RAS-Eval data in this repository — it declares no license."""
|
|
44
|
+
from pathlib import Path
|
|
45
|
+
|
|
46
|
+
repo = Path(__file__).resolve().parent.parent
|
|
47
|
+
assert not (repo / "RAS-Eval").exists()
|
|
48
|
+
strays = [
|
|
49
|
+
p for p in repo.rglob("tasks.json")
|
|
50
|
+
if ".venv" not in str(p) and ".git" not in str(p)
|
|
51
|
+
]
|
|
52
|
+
assert not strays, f"looks vendored: {strays}"
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def test_the_classifications_are_ours_and_round_trip():
|
|
56
|
+
for spec in RAS_EVAL_SPECS:
|
|
57
|
+
assert InverseSpec.from_dict(spec.to_dict()) == spec, spec.tool
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def test_reads_are_reversible_and_one_way_tools_are_not():
|
|
61
|
+
reg = ras_eval_registry()
|
|
62
|
+
assert reg.classify("get_weather") is Reversibility.REVERSIBLE
|
|
63
|
+
assert reg.classify("send_email") is Reversibility.IRREVERSIBLE
|
|
64
|
+
# Arbitrary shell stays unclassified here for the same reason it does in revoco's
|
|
65
|
+
# own workstation adapter.
|
|
66
|
+
assert reg.classify("execute_shell_command") is Reversibility.UNKNOWN
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_create_cancel_pairs_are_mutual_inverses():
|
|
70
|
+
"""The interesting shape on this surface, and the one most tool APIs lack."""
|
|
71
|
+
reg = ras_eval_registry()
|
|
72
|
+
for a, b in (("set_alarm", "cancel_alarm"),
|
|
73
|
+
("add_event_to_calendar", "remove_event_from_calendar")):
|
|
74
|
+
assert reg.get(a).inverse_tool == b
|
|
75
|
+
assert reg.get(b).inverse_tool == a
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def test_the_argument_names_corrected_by_real_traces_stay_corrected():
|
|
79
|
+
"""Regression guard on the finding that justified this whole import.
|
|
80
|
+
|
|
81
|
+
`insert_data` was first written with a `table` argument invented from the tool's
|
|
82
|
+
name. Real traces pass `db_path` and `items`, so the inverse could never resolve
|
|
83
|
+
and every legitimate insert raised a phantom rollback. Same for
|
|
84
|
+
`convert_file_to_markdown`, which takes `save_path` rather than returning
|
|
85
|
+
`output_path`.
|
|
86
|
+
"""
|
|
87
|
+
reg = ras_eval_registry()
|
|
88
|
+
insert = dict(reg.get("insert_data").effective_steps[0].arg_map)
|
|
89
|
+
assert "args.db_path" in insert.values()
|
|
90
|
+
assert not any(v == "args.table" for v in insert.values())
|
|
91
|
+
|
|
92
|
+
convert = dict(reg.get("convert_file_to_markdown").effective_steps[0].arg_map)
|
|
93
|
+
assert "args.save_path" in convert.values()
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
# ---------------------------------------------------------------------------
|
|
97
|
+
# Needs the data
|
|
98
|
+
# ---------------------------------------------------------------------------
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@needs_data
|
|
102
|
+
def test_import_produces_benign_scenarios_with_real_arguments():
|
|
103
|
+
scen = ras_eval_scenarios()
|
|
104
|
+
assert scen
|
|
105
|
+
assert all(s.intent is Intent.BENIGN for s in scen)
|
|
106
|
+
assert all(s.technique == "EXT" for s in scen)
|
|
107
|
+
# The point of importing rather than authoring: arguments a model chose.
|
|
108
|
+
assert any(st.args for s in scen for st in s.steps)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
@needs_data
|
|
112
|
+
def test_only_unique_tasks_are_imported_not_every_model_run():
|
|
113
|
+
"""Eight models ran the same 80 tasks; importing all of them would be padding."""
|
|
114
|
+
scen = ras_eval_scenarios()
|
|
115
|
+
assert len(scen) <= 80
|
|
116
|
+
assert len({s.id for s in scen}) == len(scen)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
@needs_data
|
|
120
|
+
def test_no_imported_scenario_carries_a_step_with_no_arguments():
|
|
121
|
+
"""A call with no observed arguments cannot resolve any inverse.
|
|
122
|
+
|
|
123
|
+
Importing one manufactures a phantom-rollback false positive out of missing trace
|
|
124
|
+
data rather than out of anything the control plane did. Four scenarios were
|
|
125
|
+
dropped for exactly this reason.
|
|
126
|
+
"""
|
|
127
|
+
for s in ras_eval_scenarios():
|
|
128
|
+
for st in s.steps:
|
|
129
|
+
spec = ras_eval_registry().get(st.tool)
|
|
130
|
+
needs = spec and any(
|
|
131
|
+
e.startswith("args.")
|
|
132
|
+
for step in spec.effective_steps
|
|
133
|
+
for _n, e in step.arg_map
|
|
134
|
+
)
|
|
135
|
+
if needs:
|
|
136
|
+
assert st.args, f"{s.id}: {st.tool} imported with no arguments"
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
@needs_data
|
|
140
|
+
def test_unclassified_tools_are_skipped_rather_than_imported_as_unknown():
|
|
141
|
+
known = {s.tool for s in RAS_EVAL_SPECS}
|
|
142
|
+
for s in ras_eval_scenarios():
|
|
143
|
+
for st in s.steps:
|
|
144
|
+
assert st.tool in known, f"{s.id} imported unclassified {st.tool}"
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
@needs_data
|
|
148
|
+
def test_imported_traffic_is_not_blocked():
|
|
149
|
+
"""The measurement this import exists for.
|
|
150
|
+
|
|
151
|
+
It found two real spec bugs on its first run — a 16.2% false-positive rate that
|
|
152
|
+
the hand-authored corpus could not have surfaced, because I wrote both the spec
|
|
153
|
+
and the scenario and used my invented argument names in each.
|
|
154
|
+
"""
|
|
155
|
+
results = Harness().run_all(ras_eval_scenarios())
|
|
156
|
+
fps = [r for r in results if r.outcome is Outcome.FALSE_POSITIVE]
|
|
157
|
+
assert not fps, [
|
|
158
|
+
(r.scenario.id, [s.reason for s in r.steps if not s.allowed]) for r in fps
|
|
159
|
+
]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
@needs_data
|
|
163
|
+
def test_including_the_import_reaches_a_production_like_ratio():
|
|
164
|
+
"""2.2:1 hand-authored, ~6:1 with the import — the ADR-Bench comparison."""
|
|
165
|
+
from revoco.bench import all_scenarios
|
|
166
|
+
|
|
167
|
+
combined = all_scenarios() + ras_eval_scenarios()
|
|
168
|
+
m = score(Harness().run_all(combined))
|
|
169
|
+
assert m.benign / m.malicious > 4.0
|
|
170
|
+
assert m.false_positive_rate == 0.0
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
@needs_data
|
|
174
|
+
def test_provenance_states_the_limits_rather_than_the_headline():
|
|
175
|
+
prov = provenance()
|
|
176
|
+
assert prov["license"].startswith("none declared")
|
|
177
|
+
assert "not the 640" in prov["note"]
|
|
178
|
+
assert "enterprise" in prov["note"]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|