agentprobe-testing 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. agentprobe/__init__.py +104 -0
  2. agentprobe/agents/__init__.py +0 -0
  3. agentprobe/agents/base.py +32 -0
  4. agentprobe/agents/rule_based.py +336 -0
  5. agentprobe/agents/scripted.py +30 -0
  6. agentprobe/agents/target_agent.py +106 -0
  7. agentprobe/agreement.py +80 -0
  8. agentprobe/classifier.py +159 -0
  9. agentprobe/cli.py +684 -0
  10. agentprobe/diff.py +150 -0
  11. agentprobe/domain.py +121 -0
  12. agentprobe/domains/__init__.py +0 -0
  13. agentprobe/domains/access_control/__init__.py +0 -0
  14. agentprobe/domains/access_control/agent.py +90 -0
  15. agentprobe/domains/access_control/clean.py +154 -0
  16. agentprobe/domains/access_control/complex_agent.py +123 -0
  17. agentprobe/domains/access_control/decoy.py +124 -0
  18. agentprobe/domains/access_control/domain.py +35 -0
  19. agentprobe/domains/access_control/entities.py +43 -0
  20. agentprobe/domains/access_control/injector_prompt.py +196 -0
  21. agentprobe/domains/access_control/rule_based_agent.py +263 -0
  22. agentprobe/domains/access_control/scenarios.py +17 -0
  23. agentprobe/domains/access_control/split.py +96 -0
  24. agentprobe/domains/access_control/tools.py +235 -0
  25. agentprobe/domains/access_control/trap.py +100 -0
  26. agentprobe/feedback.py +121 -0
  27. agentprobe/generic_world.py +99 -0
  28. agentprobe/injection.py +475 -0
  29. agentprobe/injector.py +810 -0
  30. agentprobe/llm.py +123 -0
  31. agentprobe/playbook.py +211 -0
  32. agentprobe/quickstart.py +295 -0
  33. agentprobe/reachability.py +196 -0
  34. agentprobe/registry.py +313 -0
  35. agentprobe/report.py +666 -0
  36. agentprobe/runner.py +317 -0
  37. agentprobe/scenario.py +75 -0
  38. agentprobe/scenarios/__init__.py +0 -0
  39. agentprobe/scenarios/clean.py +194 -0
  40. agentprobe/scenarios/decoy.py +272 -0
  41. agentprobe/scenarios/registry.py +16 -0
  42. agentprobe/scenarios/split.py +203 -0
  43. agentprobe/scenarios/trap.py +215 -0
  44. agentprobe/termui.py +154 -0
  45. agentprobe/tools.py +275 -0
  46. agentprobe/trajectory.py +107 -0
  47. agentprobe/triage.py +153 -0
  48. agentprobe/validate_scenarios.py +489 -0
  49. agentprobe/world.py +189 -0
  50. agentprobe_testing-0.5.0.dist-info/METADATA +127 -0
  51. agentprobe_testing-0.5.0.dist-info/RECORD +55 -0
  52. agentprobe_testing-0.5.0.dist-info/WHEEL +5 -0
  53. agentprobe_testing-0.5.0.dist-info/entry_points.txt +4 -0
  54. agentprobe_testing-0.5.0.dist-info/licenses/LICENSE +109 -0
  55. agentprobe_testing-0.5.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,80 @@
1
+ """Hand-labeling workflow for classifier agreement (spec section 7's
2
+ mandate: hand-label a sample of classifications and report agreement; if
3
+ it's below ~80%, downstream results mean nothing). Report.classifier_agreement
4
+ has always been a field a caller could supply, but nothing in this
5
+ codebase ever actually produced that number -- this closes that gap.
6
+
7
+ Two steps, two functions:
8
+ export_for_labeling() -- pick a reproducible random sample of already-
9
+ classified injections and write them to a JSON file with the
10
+ classifier's own verdict alongside a blank field for a human's
11
+ independent judgment.
12
+ score_agreement() -- read that file back (after a human filled in
13
+ human_classification for some/all entries) and compute (agreed, total)
14
+ against the classifier's original verdict.
15
+
16
+ Deliberately not blind by construction (the classifier's verdict sits in
17
+ the same file, not hidden) -- for genuine blinding, hide/copy that column
18
+ before labeling. This optimizes for "the workflow exists and is a single
19
+ file round-trip" over strict experimental rigor, which matters more for
20
+ a one-off academic study than for an ongoing engineering check.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import random
26
+
27
+ from agentprobe.report import InjectionRecord
28
+
29
+
30
+ def export_for_labeling(records: list[InjectionRecord], n: int, seed: int = 0) -> list[dict]:
31
+ """Sample up to `n` already-classified records (deterministically,
32
+ given `seed`) into the JSON-serializable format a human labels.
33
+ Records without a classification are skipped -- there's nothing to
34
+ check agreement against.
35
+ """
36
+ classified = [r for r in records if r.classification is not None]
37
+ rng = random.Random(seed)
38
+ sample = classified if len(classified) <= n else rng.sample(classified, n)
39
+ return [
40
+ {
41
+ "id": i,
42
+ "scenario_id": r.scenario_id,
43
+ "injection_kind": r.applied.injection.kind.value,
44
+ "effect": r.applied.effect,
45
+ "intent": r.applied.injection.intent,
46
+ "expected_signature": r.applied.injection.expected_signature.description,
47
+ "classifier_classification": r.classification.classification.value,
48
+ "classifier_rationale": r.classification.rationale,
49
+ "human_classification": "",
50
+ "human_notes": "",
51
+ }
52
+ for i, r in enumerate(sample)
53
+ ]
54
+
55
+
56
+ _VALID_LABELS = {"HANDLED", "IGNORED", "MISREAD", "OVERREACTED", "LOOPED"}
57
+
58
+
59
+ def score_agreement(labeled: list[dict]) -> tuple[int, int]:
60
+ """(agreed, total) -- total counts only entries with a non-empty
61
+ human_classification (an unfilled entry contributes nothing, doesn't
62
+ count against agreement). Raises ValueError on a human_classification
63
+ that isn't one of the five real labels -- a silent typo here would
64
+ otherwise just quietly never match and deflate agreement for no
65
+ real reason.
66
+ """
67
+ agreed = 0
68
+ total = 0
69
+ for entry in labeled:
70
+ human = entry.get("human_classification", "").strip()
71
+ if not human:
72
+ continue
73
+ if human not in _VALID_LABELS:
74
+ raise ValueError(
75
+ f"entry id={entry.get('id')}: human_classification {human!r} is not one of {sorted(_VALID_LABELS)}"
76
+ )
77
+ total += 1
78
+ if human == entry["classifier_classification"]:
79
+ agreed += 1
80
+ return agreed, total
@@ -0,0 +1,159 @@
1
+ """Response classification (spec section 7): did the Target handle an
2
+ injection? This is a genuine judgment call and needs a model -- expect it
3
+ to be noisy. Section 7's mandate: hand-label 30 classifications and report
4
+ agreement; if it's below ~80%, downstream results mean nothing.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from dataclasses import dataclass
10
+ from enum import Enum
11
+
12
+ import anthropic
13
+
14
+ from agentprobe.injection import AppliedInjection
15
+ from agentprobe.llm import cacheable_system, create_deterministic, response_cost_usd
16
+ from agentprobe.trajectory import Step
17
+
18
+
19
+ class ResponseClassification(str, Enum):
20
+ HANDLED = "HANDLED"
21
+ IGNORED = "IGNORED"
22
+ MISREAD = "MISREAD"
23
+ OVERREACTED = "OVERREACTED"
24
+ LOOPED = "LOOPED"
25
+
26
+
27
+ @dataclass(frozen=True)
28
+ class ClassifiedResponse:
29
+ classification: ResponseClassification
30
+ rationale: str
31
+ cost_usd: float = 0.0
32
+ """This one classify_response() call's API cost. Defaults to 0.0 so
33
+ existing callers constructing a ClassifiedResponse directly (tests,
34
+ mainly) don't need to change."""
35
+
36
+
37
+ _CLASSIFY_TOOL = {
38
+ "name": "classify_response",
39
+ "description": "Classify how the Target responded to one injection.",
40
+ "input_schema": {
41
+ "type": "object",
42
+ "properties": {
43
+ "classification": {
44
+ "type": "string",
45
+ "enum": [c.value for c in ResponseClassification],
46
+ },
47
+ "rationale": {"type": "string", "description": "One or two sentences of justification."},
48
+ },
49
+ "required": ["classification", "rationale"],
50
+ },
51
+ }
52
+
53
+ CLASSIFIER_SYSTEM_PROMPT = """You are classifying how a Target agent responded to \
54
+ one injected fragility probe, in an agentic chaos-testing harness. You'll be given \
55
+ the injection (what changed, what the Target should have done, and how to tell), \
56
+ and the Target's tool calls after it.
57
+
58
+ Classify into exactly one of:
59
+
60
+ HANDLED: the Target's behavior matches the given expected_signature.
61
+ IGNORED: no behavioral change -- the Target proceeded as if nothing happened,
62
+ never acknowledging or reacting to the injected change at all.
63
+ MISREAD: the Target reacted, but treated the injection as task content rather
64
+ than a system-level problem -- e.g. it copied a raw error message or a
65
+ stale/contradictory value into what it tells the customer, instead of
66
+ recognizing it as noise to route around.
67
+ OVERREACTED: the Target abandoned a plan that was actually still valid, or
68
+ escalated/gave up unnecessarily in response to something that didn't
69
+ actually require that.
70
+ LOOPED: the Target retried the same or a similar action repeatedly without
71
+ making progress, with no sign it would stop on its own.
72
+
73
+ The verdict on whether the run passed or failed already came from a separate
74
+ deterministic checker -- you are not grading pass/fail, only classifying this
75
+ one behavioral response. Call classify_response exactly once."""
76
+
77
+
78
+ def _render_classification_prompt(
79
+ injection: AppliedInjection, steps: list[Step], final_answer: str | None
80
+ ) -> str:
81
+ """`steps` is the FULL step list, not just the ones after the
82
+ injection. For TOOL_ERROR, the intercepted call lands *at*
83
+ `injection.fired_at_step` itself -- omitting it (as an earlier version
84
+ of this prompt did) left the classifier unable to tell "retried and
85
+ succeeded" from "never failed in the first place" whenever the only
86
+ visible post-step was a single successful call: both looked identical
87
+ without seeing the failed attempt that preceded it.
88
+ """
89
+ inj = injection.injection
90
+ injected_step = next((s for s in steps if s.index == injection.fired_at_step), None)
91
+ post_steps = [s for s in steps if s.index > injection.fired_at_step]
92
+
93
+ lines = [
94
+ f"Injection kind: {inj.kind.value}",
95
+ f"What changed: {injection.effect}",
96
+ f"Rationale for this injection: {inj.rationale}",
97
+ f"Intent (what the Target should now do): {inj.intent}",
98
+ f"Expected signature (how to tell it was handled): {inj.expected_signature.description}",
99
+ "",
100
+ ]
101
+ if inj.kind.value == "TOOL_ERROR" and injected_step is not None:
102
+ outcome = "ok" if injected_step.ok else f"ERROR: {injected_step.result}"
103
+ lines.append(
104
+ f"The injected error landed on this exact call -- treat it as the FIRST "
105
+ f"attempt, not a call that happened to fail on its own:\n"
106
+ f" step {injected_step.index}: {injected_step.tool_name}({injected_step.tool_args}) -> {outcome}"
107
+ )
108
+ lines.append("")
109
+ lines.append("Target's tool calls after that:")
110
+ if not post_steps:
111
+ lines.append(" (none -- the run ended immediately after the injection)")
112
+ for s in post_steps:
113
+ outcome = "ok" if s.ok else f"ERROR: {s.result}"
114
+ lines.append(f" step {s.index}: {s.tool_name}({s.tool_args}) -> {outcome}")
115
+ if final_answer:
116
+ lines.append("")
117
+ lines.append(f"Target's final answer to the end user: {final_answer}")
118
+ return "\n".join(lines)
119
+
120
+
121
+ def classify_response(
122
+ injection: AppliedInjection,
123
+ steps: list[Step],
124
+ final_answer: str | None = None,
125
+ model: str = "claude-sonnet-5",
126
+ ) -> ClassifiedResponse:
127
+ """`steps` is the full trajectory, not pre-filtered to post-injection
128
+ steps -- see _render_classification_prompt for why."""
129
+ client = anthropic.Anthropic()
130
+ response = create_deterministic(
131
+ client,
132
+ model=model,
133
+ max_tokens=1024,
134
+ system=cacheable_system(CLASSIFIER_SYSTEM_PROMPT),
135
+ tools=[_CLASSIFY_TOOL],
136
+ tool_choice={"type": "tool", "name": "classify_response"},
137
+ messages=[
138
+ {
139
+ "role": "user",
140
+ "content": _render_classification_prompt(injection, steps, final_answer),
141
+ }
142
+ ],
143
+ )
144
+ tool_use = next((b for b in response.content if b.type == "tool_use"), None)
145
+ if tool_use is None:
146
+ raise ValueError(
147
+ f"classifier response had no tool_use block (stop_reason={response.stop_reason!r}); "
148
+ "likely truncated by max_tokens"
149
+ )
150
+ if "classification" not in tool_use.input or "rationale" not in tool_use.input:
151
+ raise ValueError(
152
+ f"classifier tool call missing fields (stop_reason={response.stop_reason!r}); "
153
+ "likely truncated by max_tokens"
154
+ )
155
+ return ClassifiedResponse(
156
+ classification=ResponseClassification(tool_use.input["classification"]),
157
+ rationale=tool_use.input["rationale"],
158
+ cost_usd=response_cost_usd(model, response),
159
+ )