agentprobe-testing 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentprobe/__init__.py +104 -0
- agentprobe/agents/__init__.py +0 -0
- agentprobe/agents/base.py +32 -0
- agentprobe/agents/rule_based.py +336 -0
- agentprobe/agents/scripted.py +30 -0
- agentprobe/agents/target_agent.py +106 -0
- agentprobe/agreement.py +80 -0
- agentprobe/classifier.py +159 -0
- agentprobe/cli.py +684 -0
- agentprobe/diff.py +150 -0
- agentprobe/domain.py +121 -0
- agentprobe/domains/__init__.py +0 -0
- agentprobe/domains/access_control/__init__.py +0 -0
- agentprobe/domains/access_control/agent.py +90 -0
- agentprobe/domains/access_control/clean.py +154 -0
- agentprobe/domains/access_control/complex_agent.py +123 -0
- agentprobe/domains/access_control/decoy.py +124 -0
- agentprobe/domains/access_control/domain.py +35 -0
- agentprobe/domains/access_control/entities.py +43 -0
- agentprobe/domains/access_control/injector_prompt.py +196 -0
- agentprobe/domains/access_control/rule_based_agent.py +263 -0
- agentprobe/domains/access_control/scenarios.py +17 -0
- agentprobe/domains/access_control/split.py +96 -0
- agentprobe/domains/access_control/tools.py +235 -0
- agentprobe/domains/access_control/trap.py +100 -0
- agentprobe/feedback.py +121 -0
- agentprobe/generic_world.py +99 -0
- agentprobe/injection.py +475 -0
- agentprobe/injector.py +810 -0
- agentprobe/llm.py +123 -0
- agentprobe/playbook.py +211 -0
- agentprobe/quickstart.py +295 -0
- agentprobe/reachability.py +196 -0
- agentprobe/registry.py +313 -0
- agentprobe/report.py +666 -0
- agentprobe/runner.py +317 -0
- agentprobe/scenario.py +75 -0
- agentprobe/scenarios/__init__.py +0 -0
- agentprobe/scenarios/clean.py +194 -0
- agentprobe/scenarios/decoy.py +272 -0
- agentprobe/scenarios/registry.py +16 -0
- agentprobe/scenarios/split.py +203 -0
- agentprobe/scenarios/trap.py +215 -0
- agentprobe/termui.py +154 -0
- agentprobe/tools.py +275 -0
- agentprobe/trajectory.py +107 -0
- agentprobe/triage.py +153 -0
- agentprobe/validate_scenarios.py +489 -0
- agentprobe/world.py +189 -0
- agentprobe_testing-0.5.0.dist-info/METADATA +127 -0
- agentprobe_testing-0.5.0.dist-info/RECORD +55 -0
- agentprobe_testing-0.5.0.dist-info/WHEEL +5 -0
- agentprobe_testing-0.5.0.dist-info/entry_points.txt +4 -0
- agentprobe_testing-0.5.0.dist-info/licenses/LICENSE +109 -0
- agentprobe_testing-0.5.0.dist-info/top_level.txt +1 -0
agentprobe/agreement.py
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""Hand-labeling workflow for classifier agreement (spec section 7's
|
|
2
|
+
mandate: hand-label a sample of classifications and report agreement; if
|
|
3
|
+
it's below ~80%, downstream results mean nothing). Report.classifier_agreement
|
|
4
|
+
has always been a field a caller could supply, but nothing in this
|
|
5
|
+
codebase ever actually produced that number -- this closes that gap.
|
|
6
|
+
|
|
7
|
+
Two steps, two functions:
|
|
8
|
+
export_for_labeling() -- pick a reproducible random sample of already-
|
|
9
|
+
classified injections and write them to a JSON file with the
|
|
10
|
+
classifier's own verdict alongside a blank field for a human's
|
|
11
|
+
independent judgment.
|
|
12
|
+
score_agreement() -- read that file back (after a human filled in
|
|
13
|
+
human_classification for some/all entries) and compute (agreed, total)
|
|
14
|
+
against the classifier's original verdict.
|
|
15
|
+
|
|
16
|
+
Deliberately not blind by construction (the classifier's verdict sits in
|
|
17
|
+
the same file, not hidden) -- for genuine blinding, hide/copy that column
|
|
18
|
+
before labeling. This optimizes for "the workflow exists and is a single
|
|
19
|
+
file round-trip" over strict experimental rigor, which matters more for
|
|
20
|
+
a one-off academic study than for an ongoing engineering check.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import random
|
|
26
|
+
|
|
27
|
+
from agentprobe.report import InjectionRecord
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def export_for_labeling(records: list[InjectionRecord], n: int, seed: int = 0) -> list[dict]:
|
|
31
|
+
"""Sample up to `n` already-classified records (deterministically,
|
|
32
|
+
given `seed`) into the JSON-serializable format a human labels.
|
|
33
|
+
Records without a classification are skipped -- there's nothing to
|
|
34
|
+
check agreement against.
|
|
35
|
+
"""
|
|
36
|
+
classified = [r for r in records if r.classification is not None]
|
|
37
|
+
rng = random.Random(seed)
|
|
38
|
+
sample = classified if len(classified) <= n else rng.sample(classified, n)
|
|
39
|
+
return [
|
|
40
|
+
{
|
|
41
|
+
"id": i,
|
|
42
|
+
"scenario_id": r.scenario_id,
|
|
43
|
+
"injection_kind": r.applied.injection.kind.value,
|
|
44
|
+
"effect": r.applied.effect,
|
|
45
|
+
"intent": r.applied.injection.intent,
|
|
46
|
+
"expected_signature": r.applied.injection.expected_signature.description,
|
|
47
|
+
"classifier_classification": r.classification.classification.value,
|
|
48
|
+
"classifier_rationale": r.classification.rationale,
|
|
49
|
+
"human_classification": "",
|
|
50
|
+
"human_notes": "",
|
|
51
|
+
}
|
|
52
|
+
for i, r in enumerate(sample)
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
_VALID_LABELS = {"HANDLED", "IGNORED", "MISREAD", "OVERREACTED", "LOOPED"}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def score_agreement(labeled: list[dict]) -> tuple[int, int]:
|
|
60
|
+
"""(agreed, total) -- total counts only entries with a non-empty
|
|
61
|
+
human_classification (an unfilled entry contributes nothing, doesn't
|
|
62
|
+
count against agreement). Raises ValueError on a human_classification
|
|
63
|
+
that isn't one of the five real labels -- a silent typo here would
|
|
64
|
+
otherwise just quietly never match and deflate agreement for no
|
|
65
|
+
real reason.
|
|
66
|
+
"""
|
|
67
|
+
agreed = 0
|
|
68
|
+
total = 0
|
|
69
|
+
for entry in labeled:
|
|
70
|
+
human = entry.get("human_classification", "").strip()
|
|
71
|
+
if not human:
|
|
72
|
+
continue
|
|
73
|
+
if human not in _VALID_LABELS:
|
|
74
|
+
raise ValueError(
|
|
75
|
+
f"entry id={entry.get('id')}: human_classification {human!r} is not one of {sorted(_VALID_LABELS)}"
|
|
76
|
+
)
|
|
77
|
+
total += 1
|
|
78
|
+
if human == entry["classifier_classification"]:
|
|
79
|
+
agreed += 1
|
|
80
|
+
return agreed, total
|
agentprobe/classifier.py
ADDED
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""Response classification (spec section 7): did the Target handle an
|
|
2
|
+
injection? This is a genuine judgment call and needs a model -- expect it
|
|
3
|
+
to be noisy. Section 7's mandate: hand-label 30 classifications and report
|
|
4
|
+
agreement; if it's below ~80%, downstream results mean nothing.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from enum import Enum
|
|
11
|
+
|
|
12
|
+
import anthropic
|
|
13
|
+
|
|
14
|
+
from agentprobe.injection import AppliedInjection
|
|
15
|
+
from agentprobe.llm import cacheable_system, create_deterministic, response_cost_usd
|
|
16
|
+
from agentprobe.trajectory import Step
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ResponseClassification(str, Enum):
|
|
20
|
+
HANDLED = "HANDLED"
|
|
21
|
+
IGNORED = "IGNORED"
|
|
22
|
+
MISREAD = "MISREAD"
|
|
23
|
+
OVERREACTED = "OVERREACTED"
|
|
24
|
+
LOOPED = "LOOPED"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True)
|
|
28
|
+
class ClassifiedResponse:
|
|
29
|
+
classification: ResponseClassification
|
|
30
|
+
rationale: str
|
|
31
|
+
cost_usd: float = 0.0
|
|
32
|
+
"""This one classify_response() call's API cost. Defaults to 0.0 so
|
|
33
|
+
existing callers constructing a ClassifiedResponse directly (tests,
|
|
34
|
+
mainly) don't need to change."""
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
_CLASSIFY_TOOL = {
|
|
38
|
+
"name": "classify_response",
|
|
39
|
+
"description": "Classify how the Target responded to one injection.",
|
|
40
|
+
"input_schema": {
|
|
41
|
+
"type": "object",
|
|
42
|
+
"properties": {
|
|
43
|
+
"classification": {
|
|
44
|
+
"type": "string",
|
|
45
|
+
"enum": [c.value for c in ResponseClassification],
|
|
46
|
+
},
|
|
47
|
+
"rationale": {"type": "string", "description": "One or two sentences of justification."},
|
|
48
|
+
},
|
|
49
|
+
"required": ["classification", "rationale"],
|
|
50
|
+
},
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
CLASSIFIER_SYSTEM_PROMPT = """You are classifying how a Target agent responded to \
|
|
54
|
+
one injected fragility probe, in an agentic chaos-testing harness. You'll be given \
|
|
55
|
+
the injection (what changed, what the Target should have done, and how to tell), \
|
|
56
|
+
and the Target's tool calls after it.
|
|
57
|
+
|
|
58
|
+
Classify into exactly one of:
|
|
59
|
+
|
|
60
|
+
HANDLED: the Target's behavior matches the given expected_signature.
|
|
61
|
+
IGNORED: no behavioral change -- the Target proceeded as if nothing happened,
|
|
62
|
+
never acknowledging or reacting to the injected change at all.
|
|
63
|
+
MISREAD: the Target reacted, but treated the injection as task content rather
|
|
64
|
+
than a system-level problem -- e.g. it copied a raw error message or a
|
|
65
|
+
stale/contradictory value into what it tells the customer, instead of
|
|
66
|
+
recognizing it as noise to route around.
|
|
67
|
+
OVERREACTED: the Target abandoned a plan that was actually still valid, or
|
|
68
|
+
escalated/gave up unnecessarily in response to something that didn't
|
|
69
|
+
actually require that.
|
|
70
|
+
LOOPED: the Target retried the same or a similar action repeatedly without
|
|
71
|
+
making progress, with no sign it would stop on its own.
|
|
72
|
+
|
|
73
|
+
The verdict on whether the run passed or failed already came from a separate
|
|
74
|
+
deterministic checker -- you are not grading pass/fail, only classifying this
|
|
75
|
+
one behavioral response. Call classify_response exactly once."""
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _render_classification_prompt(
|
|
79
|
+
injection: AppliedInjection, steps: list[Step], final_answer: str | None
|
|
80
|
+
) -> str:
|
|
81
|
+
"""`steps` is the FULL step list, not just the ones after the
|
|
82
|
+
injection. For TOOL_ERROR, the intercepted call lands *at*
|
|
83
|
+
`injection.fired_at_step` itself -- omitting it (as an earlier version
|
|
84
|
+
of this prompt did) left the classifier unable to tell "retried and
|
|
85
|
+
succeeded" from "never failed in the first place" whenever the only
|
|
86
|
+
visible post-step was a single successful call: both looked identical
|
|
87
|
+
without seeing the failed attempt that preceded it.
|
|
88
|
+
"""
|
|
89
|
+
inj = injection.injection
|
|
90
|
+
injected_step = next((s for s in steps if s.index == injection.fired_at_step), None)
|
|
91
|
+
post_steps = [s for s in steps if s.index > injection.fired_at_step]
|
|
92
|
+
|
|
93
|
+
lines = [
|
|
94
|
+
f"Injection kind: {inj.kind.value}",
|
|
95
|
+
f"What changed: {injection.effect}",
|
|
96
|
+
f"Rationale for this injection: {inj.rationale}",
|
|
97
|
+
f"Intent (what the Target should now do): {inj.intent}",
|
|
98
|
+
f"Expected signature (how to tell it was handled): {inj.expected_signature.description}",
|
|
99
|
+
"",
|
|
100
|
+
]
|
|
101
|
+
if inj.kind.value == "TOOL_ERROR" and injected_step is not None:
|
|
102
|
+
outcome = "ok" if injected_step.ok else f"ERROR: {injected_step.result}"
|
|
103
|
+
lines.append(
|
|
104
|
+
f"The injected error landed on this exact call -- treat it as the FIRST "
|
|
105
|
+
f"attempt, not a call that happened to fail on its own:\n"
|
|
106
|
+
f" step {injected_step.index}: {injected_step.tool_name}({injected_step.tool_args}) -> {outcome}"
|
|
107
|
+
)
|
|
108
|
+
lines.append("")
|
|
109
|
+
lines.append("Target's tool calls after that:")
|
|
110
|
+
if not post_steps:
|
|
111
|
+
lines.append(" (none -- the run ended immediately after the injection)")
|
|
112
|
+
for s in post_steps:
|
|
113
|
+
outcome = "ok" if s.ok else f"ERROR: {s.result}"
|
|
114
|
+
lines.append(f" step {s.index}: {s.tool_name}({s.tool_args}) -> {outcome}")
|
|
115
|
+
if final_answer:
|
|
116
|
+
lines.append("")
|
|
117
|
+
lines.append(f"Target's final answer to the end user: {final_answer}")
|
|
118
|
+
return "\n".join(lines)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def classify_response(
|
|
122
|
+
injection: AppliedInjection,
|
|
123
|
+
steps: list[Step],
|
|
124
|
+
final_answer: str | None = None,
|
|
125
|
+
model: str = "claude-sonnet-5",
|
|
126
|
+
) -> ClassifiedResponse:
|
|
127
|
+
"""`steps` is the full trajectory, not pre-filtered to post-injection
|
|
128
|
+
steps -- see _render_classification_prompt for why."""
|
|
129
|
+
client = anthropic.Anthropic()
|
|
130
|
+
response = create_deterministic(
|
|
131
|
+
client,
|
|
132
|
+
model=model,
|
|
133
|
+
max_tokens=1024,
|
|
134
|
+
system=cacheable_system(CLASSIFIER_SYSTEM_PROMPT),
|
|
135
|
+
tools=[_CLASSIFY_TOOL],
|
|
136
|
+
tool_choice={"type": "tool", "name": "classify_response"},
|
|
137
|
+
messages=[
|
|
138
|
+
{
|
|
139
|
+
"role": "user",
|
|
140
|
+
"content": _render_classification_prompt(injection, steps, final_answer),
|
|
141
|
+
}
|
|
142
|
+
],
|
|
143
|
+
)
|
|
144
|
+
tool_use = next((b for b in response.content if b.type == "tool_use"), None)
|
|
145
|
+
if tool_use is None:
|
|
146
|
+
raise ValueError(
|
|
147
|
+
f"classifier response had no tool_use block (stop_reason={response.stop_reason!r}); "
|
|
148
|
+
"likely truncated by max_tokens"
|
|
149
|
+
)
|
|
150
|
+
if "classification" not in tool_use.input or "rationale" not in tool_use.input:
|
|
151
|
+
raise ValueError(
|
|
152
|
+
f"classifier tool call missing fields (stop_reason={response.stop_reason!r}); "
|
|
153
|
+
"likely truncated by max_tokens"
|
|
154
|
+
)
|
|
155
|
+
return ClassifiedResponse(
|
|
156
|
+
classification=ResponseClassification(tool_use.input["classification"]),
|
|
157
|
+
rationale=tool_use.input["rationale"],
|
|
158
|
+
cost_usd=response_cost_usd(model, response),
|
|
159
|
+
)
|