agentprobe-testing 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentprobe/__init__.py +104 -0
- agentprobe/agents/__init__.py +0 -0
- agentprobe/agents/base.py +32 -0
- agentprobe/agents/rule_based.py +336 -0
- agentprobe/agents/scripted.py +30 -0
- agentprobe/agents/target_agent.py +106 -0
- agentprobe/agreement.py +80 -0
- agentprobe/classifier.py +159 -0
- agentprobe/cli.py +684 -0
- agentprobe/diff.py +150 -0
- agentprobe/domain.py +121 -0
- agentprobe/domains/__init__.py +0 -0
- agentprobe/domains/access_control/__init__.py +0 -0
- agentprobe/domains/access_control/agent.py +90 -0
- agentprobe/domains/access_control/clean.py +154 -0
- agentprobe/domains/access_control/complex_agent.py +123 -0
- agentprobe/domains/access_control/decoy.py +124 -0
- agentprobe/domains/access_control/domain.py +35 -0
- agentprobe/domains/access_control/entities.py +43 -0
- agentprobe/domains/access_control/injector_prompt.py +196 -0
- agentprobe/domains/access_control/rule_based_agent.py +263 -0
- agentprobe/domains/access_control/scenarios.py +17 -0
- agentprobe/domains/access_control/split.py +96 -0
- agentprobe/domains/access_control/tools.py +235 -0
- agentprobe/domains/access_control/trap.py +100 -0
- agentprobe/feedback.py +121 -0
- agentprobe/generic_world.py +99 -0
- agentprobe/injection.py +475 -0
- agentprobe/injector.py +810 -0
- agentprobe/llm.py +123 -0
- agentprobe/playbook.py +211 -0
- agentprobe/quickstart.py +295 -0
- agentprobe/reachability.py +196 -0
- agentprobe/registry.py +313 -0
- agentprobe/report.py +666 -0
- agentprobe/runner.py +317 -0
- agentprobe/scenario.py +75 -0
- agentprobe/scenarios/__init__.py +0 -0
- agentprobe/scenarios/clean.py +194 -0
- agentprobe/scenarios/decoy.py +272 -0
- agentprobe/scenarios/registry.py +16 -0
- agentprobe/scenarios/split.py +203 -0
- agentprobe/scenarios/trap.py +215 -0
- agentprobe/termui.py +154 -0
- agentprobe/tools.py +275 -0
- agentprobe/trajectory.py +107 -0
- agentprobe/triage.py +153 -0
- agentprobe/validate_scenarios.py +489 -0
- agentprobe/world.py +189 -0
- agentprobe_testing-0.5.0.dist-info/METADATA +127 -0
- agentprobe_testing-0.5.0.dist-info/RECORD +55 -0
- agentprobe_testing-0.5.0.dist-info/WHEEL +5 -0
- agentprobe_testing-0.5.0.dist-info/entry_points.txt +4 -0
- agentprobe_testing-0.5.0.dist-info/licenses/LICENSE +109 -0
- agentprobe_testing-0.5.0.dist-info/top_level.txt +1 -0
agentprobe/injection.py
ADDED
|
@@ -0,0 +1,475 @@
|
|
|
1
|
+
"""The injection catalog (spec section 6) and the runtime that applies an
|
|
2
|
+
Injection to a live world (any domain's, via the generic entity protocol
|
|
3
|
+
-- see domain.py's EntityWorld). Application is pure bookkeeping -- no LLM
|
|
4
|
+
here. The judgment (which injection, when) lives in injector.py; the
|
|
5
|
+
judgment of whether it worked lives in classifier.py.
|
|
6
|
+
|
|
7
|
+
Injections are armed, not fired (repair Task 3): the Injector attaches a
|
|
8
|
+
Trigger -- a condition to wait for -- rather than guessing a step number.
|
|
9
|
+
`trigger_matches_pre_dispatch`/`trigger_matches_post_dispatch` are checked
|
|
10
|
+
by the runner every step; when one matches, THAT is when apply_injection()
|
|
11
|
+
actually runs. An ArmedInjection that never matches by run end is EXPIRED
|
|
12
|
+
(a bad prediction), which is a different failure from INVALID (a fired
|
|
13
|
+
injection that made the goal unreachable, i.e. unfair).
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import dataclasses
|
|
19
|
+
from dataclasses import dataclass, field
|
|
20
|
+
from enum import Enum
|
|
21
|
+
from typing import TYPE_CHECKING, Any, Optional
|
|
22
|
+
|
|
23
|
+
from agentprobe.scenario import CommitPattern
|
|
24
|
+
from agentprobe.tools import COMMIT_TOOLS
|
|
25
|
+
|
|
26
|
+
if TYPE_CHECKING:
|
|
27
|
+
from agentprobe.domain import EntityWorld
|
|
28
|
+
from agentprobe.trajectory import CallRecord, Step
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class InjectionKind(str, Enum):
|
|
32
|
+
TOOL_ERROR = "TOOL_ERROR"
|
|
33
|
+
STALE_READ = "STALE_READ"
|
|
34
|
+
CONTRADICTION = "CONTRADICTION"
|
|
35
|
+
LATE_INFO = "LATE_INFO"
|
|
36
|
+
AMBIGUITY = "AMBIGUITY"
|
|
37
|
+
PROMPT_INJECTION = "PROMPT_INJECTION"
|
|
38
|
+
"""An adversarial instruction embedded in ordinary-looking ticket
|
|
39
|
+
content (e.g. a note that reads like a customer message but actually
|
|
40
|
+
commands the Target to take an unauthorized action). Distinct from
|
|
41
|
+
CONTRADICTION: that kind tests whether the Target notices a changed
|
|
42
|
+
FACT; this kind tests whether the Target can be manipulated into an
|
|
43
|
+
ACTION by data it should treat as untrusted input, not instructions --
|
|
44
|
+
a real production-agent security failure mode, not just a robustness
|
|
45
|
+
one. Mechanically identical to CONTRADICTION/LATE_INFO (appends a
|
|
46
|
+
ticket note) -- the payload's text is what differs in kind, not the
|
|
47
|
+
plumbing."""
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass(frozen=True)
|
|
51
|
+
class ResponseSignature:
|
|
52
|
+
"""How to tell whether the Target's subsequent behavior counts as
|
|
53
|
+
having handled this injection. Free text, judged by the classifier --
|
|
54
|
+
but committed to *before* we see how the Target reacts."""
|
|
55
|
+
|
|
56
|
+
description: str
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass(frozen=True)
|
|
60
|
+
class Injection:
|
|
61
|
+
kind: InjectionKind
|
|
62
|
+
payload: dict[str, Any]
|
|
63
|
+
intent: str
|
|
64
|
+
"""What the Target SHOULD now do. Mandatory -- without it, a failure
|
|
65
|
+
can't be attributed to the target vs. an unfair injection."""
|
|
66
|
+
expected_signature: ResponseSignature
|
|
67
|
+
rationale: str
|
|
68
|
+
"""Why this moment -- for the report, and for auditing injector
|
|
69
|
+
gaming (spec section 12)."""
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class TriggerKind(str, Enum):
|
|
73
|
+
ON_TOOL_CALL = "on_tool_call"
|
|
74
|
+
ON_NTH_TOOL_CALL = "on_nth_tool_call"
|
|
75
|
+
ON_READ_OF = "on_read_of"
|
|
76
|
+
ON_READ_OF_ANY = "on_read_of_any"
|
|
77
|
+
AFTER_COMMIT = "after_commit"
|
|
78
|
+
ON_ANY_COMMIT = "on_any_commit"
|
|
79
|
+
AFTER_ANY_COMMIT = "after_any_commit"
|
|
80
|
+
ON_NEXT_ACTION = "on_next_action"
|
|
81
|
+
ON_STEP = "on_step"
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
@dataclass(frozen=True)
|
|
85
|
+
class Trigger:
|
|
86
|
+
"""A condition an ArmedInjection waits for. Construct via the factory
|
|
87
|
+
methods below, not the raw fields directly.
|
|
88
|
+
|
|
89
|
+
- on_tool_call(name): fires the next time the Target calls that tool,
|
|
90
|
+
whenever that comes -- the natural fit for TOOL_ERROR.
|
|
91
|
+
- on_nth_tool_call(name, n): fires on the Nth call to that tool.
|
|
92
|
+
- on_read_of(entity_id): fires right after the Target reads that
|
|
93
|
+
entity -- for STALE_READ, so the Target has a stale cached belief
|
|
94
|
+
*before* the value changes, not after (which would test nothing).
|
|
95
|
+
- on_read_of_any(entity_ids): fires right after the Target reads the
|
|
96
|
+
FIRST of several candidate entities -- same relaxation as
|
|
97
|
+
on_any_commit, one layer down. on_read_of forces a guess of which
|
|
98
|
+
single entity the Target reads next; a wrong guess just expires.
|
|
99
|
+
Use this whenever more than one entity is a plausible next read.
|
|
100
|
+
The fired AppliedInjection's payload entity_id is corrected to
|
|
101
|
+
whichever candidate was actually read, so the mutation always lands
|
|
102
|
+
on the right one regardless of which candidate matched.
|
|
103
|
+
- after_commit(pattern): fires right after a matching commit -- for
|
|
104
|
+
LATE_INFO, which is meaningless before a commit exists to be late
|
|
105
|
+
about.
|
|
106
|
+
- on_any_commit(): fires the next time the Target calls ANY commit
|
|
107
|
+
tool, whichever one it turns out to be. Repair Task 3 follow-up:
|
|
108
|
+
on_tool_call forced the Injector to guess which specific commit
|
|
109
|
+
tool (close_ticket vs issue_refund vs escalate) the Target would
|
|
110
|
+
call next, and a wrong guess meant a silent expiry -- most of the
|
|
111
|
+
measured wasted-armed-injection rate traced back to exactly this.
|
|
112
|
+
Use when what matters is "right before the Target commits to
|
|
113
|
+
anything," not which commit it is.
|
|
114
|
+
- after_any_commit(): same relaxation for after_commit -- fires right
|
|
115
|
+
after ANY commit tool succeeds, not a specifically predicted one.
|
|
116
|
+
- on_next_action(): fires unconditionally on the Target's very next
|
|
117
|
+
dispatched call, whichever tool it turns out to be. TOOL_ERROR is
|
|
118
|
+
checked pre-dispatch, so the actual tool name is already known at
|
|
119
|
+
fire time -- there was never a real need to pre-commit to one, and
|
|
120
|
+
guessing wrong (arming on_tool_call for a tool the Target doesn't
|
|
121
|
+
call again) was the single largest source of expired injections.
|
|
122
|
+
The fired AppliedInjection's payload tool_name is corrected to the
|
|
123
|
+
real intercepted tool at fire time regardless of what was guessed.
|
|
124
|
+
- on_step(n): fires at absolute step n. Kept for the hardcoded
|
|
125
|
+
injector and for tests that want exact, deterministic timing --
|
|
126
|
+
the ModelInjector is not offered this one, since re-introducing a
|
|
127
|
+
guessed step number is exactly what this repair removes.
|
|
128
|
+
"""
|
|
129
|
+
|
|
130
|
+
kind: TriggerKind
|
|
131
|
+
tool_name: Optional[str] = None
|
|
132
|
+
n: Optional[int] = None
|
|
133
|
+
entity_id: Optional[str] = None
|
|
134
|
+
entity_ids: Optional[tuple[str, ...]] = None
|
|
135
|
+
commit_pattern: Optional[CommitPattern] = None
|
|
136
|
+
step: Optional[int] = None
|
|
137
|
+
|
|
138
|
+
@staticmethod
|
|
139
|
+
def on_tool_call(name: str) -> "Trigger":
|
|
140
|
+
return Trigger(TriggerKind.ON_TOOL_CALL, tool_name=name)
|
|
141
|
+
|
|
142
|
+
@staticmethod
|
|
143
|
+
def on_nth_tool_call(name: str, n: int) -> "Trigger":
|
|
144
|
+
return Trigger(TriggerKind.ON_NTH_TOOL_CALL, tool_name=name, n=n)
|
|
145
|
+
|
|
146
|
+
@staticmethod
|
|
147
|
+
def on_read_of(entity_id: str) -> "Trigger":
|
|
148
|
+
return Trigger(TriggerKind.ON_READ_OF, entity_id=entity_id)
|
|
149
|
+
|
|
150
|
+
@staticmethod
|
|
151
|
+
def on_read_of_any(entity_ids: list[str]) -> "Trigger":
|
|
152
|
+
return Trigger(TriggerKind.ON_READ_OF_ANY, entity_ids=tuple(entity_ids))
|
|
153
|
+
|
|
154
|
+
@staticmethod
|
|
155
|
+
def after_commit(pattern: CommitPattern) -> "Trigger":
|
|
156
|
+
return Trigger(TriggerKind.AFTER_COMMIT, commit_pattern=pattern)
|
|
157
|
+
|
|
158
|
+
@staticmethod
|
|
159
|
+
def on_any_commit() -> "Trigger":
|
|
160
|
+
return Trigger(TriggerKind.ON_ANY_COMMIT)
|
|
161
|
+
|
|
162
|
+
@staticmethod
|
|
163
|
+
def after_any_commit() -> "Trigger":
|
|
164
|
+
return Trigger(TriggerKind.AFTER_ANY_COMMIT)
|
|
165
|
+
|
|
166
|
+
@staticmethod
|
|
167
|
+
def on_next_action() -> "Trigger":
|
|
168
|
+
return Trigger(TriggerKind.ON_NEXT_ACTION)
|
|
169
|
+
|
|
170
|
+
@staticmethod
|
|
171
|
+
def on_step(n: int) -> "Trigger":
|
|
172
|
+
return Trigger(TriggerKind.ON_STEP, step=n)
|
|
173
|
+
|
|
174
|
+
def describe(self) -> str:
|
|
175
|
+
if self.kind == TriggerKind.ON_TOOL_CALL:
|
|
176
|
+
return f"on_tool_call({self.tool_name})"
|
|
177
|
+
if self.kind == TriggerKind.ON_NTH_TOOL_CALL:
|
|
178
|
+
return f"on_nth_tool_call({self.tool_name}, {self.n})"
|
|
179
|
+
if self.kind == TriggerKind.ON_READ_OF:
|
|
180
|
+
return f"on_read_of({self.entity_id})"
|
|
181
|
+
if self.kind == TriggerKind.ON_READ_OF_ANY:
|
|
182
|
+
return f"on_read_of_any({self.entity_ids})"
|
|
183
|
+
if self.kind == TriggerKind.AFTER_COMMIT:
|
|
184
|
+
assert self.commit_pattern is not None
|
|
185
|
+
return f"after_commit({self.commit_pattern.tool}({self.commit_pattern.args}))"
|
|
186
|
+
if self.kind == TriggerKind.ON_ANY_COMMIT:
|
|
187
|
+
return "on_any_commit()"
|
|
188
|
+
if self.kind == TriggerKind.AFTER_ANY_COMMIT:
|
|
189
|
+
return "after_any_commit()"
|
|
190
|
+
if self.kind == TriggerKind.ON_NEXT_ACTION:
|
|
191
|
+
return "on_next_action()"
|
|
192
|
+
return f"on_step({self.step})"
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
@dataclass(frozen=True)
|
|
196
|
+
class ArmedInjection:
|
|
197
|
+
"""Decided but not yet applied -- waiting for `trigger` to match."""
|
|
198
|
+
|
|
199
|
+
injection: Injection
|
|
200
|
+
trigger: Trigger
|
|
201
|
+
armed_at_step: int
|
|
202
|
+
expires_after: Optional[int] = None
|
|
203
|
+
"""Steps after arming before this expires early even if the trigger
|
|
204
|
+
never matches. None = stays armed until it fires or the run ends."""
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
@dataclass(frozen=True)
|
|
208
|
+
class ExpiredInjection:
|
|
209
|
+
"""An ArmedInjection whose trigger never matched -- a bad prediction
|
|
210
|
+
by the Injector, not an unfair one (that's INVALID, which only applies
|
|
211
|
+
to injections that actually fired)."""
|
|
212
|
+
|
|
213
|
+
injection: Injection
|
|
214
|
+
trigger: Trigger
|
|
215
|
+
armed_at_step: int
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
@dataclass(frozen=True)
|
|
219
|
+
class AppliedInjection:
|
|
220
|
+
"""One injection that actually fired during a run, with its outcome."""
|
|
221
|
+
|
|
222
|
+
armed_at_step: int
|
|
223
|
+
fired_at_step: int
|
|
224
|
+
"""Logged separately from armed_at_step -- the gap between them is
|
|
225
|
+
diagnostic (how long the Injector's prediction took to pan out, if at
|
|
226
|
+
all)."""
|
|
227
|
+
injection: Injection
|
|
228
|
+
effect: str
|
|
229
|
+
"""What apply_injection actually did, in words -- for the report."""
|
|
230
|
+
valid: bool = True
|
|
231
|
+
"""False if the validity check found the goal unreachable right after
|
|
232
|
+
this fired (spec section 8) -- an INVALID_INJECTION."""
|
|
233
|
+
trigger_kind: str = ""
|
|
234
|
+
"""The TriggerKind.value that actually fired this, e.g. "on_read_of_any".
|
|
235
|
+
Defaults to "" for backward compatibility with existing call sites
|
|
236
|
+
that don't set it -- but runner.py's fire() always populates it, since
|
|
237
|
+
without this a fired injection can't be attributed to the trigger
|
|
238
|
+
choice that worked, unlike ExpiredInjection (which already carries
|
|
239
|
+
its trigger). Needed for playbook.py's outcome tracking."""
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
class InjectionError(Exception):
|
|
243
|
+
"""Raised when an injection's payload doesn't match live world state
|
|
244
|
+
(e.g. references an id that doesn't exist). Distinct from ToolError --
|
|
245
|
+
this is an injector mistake, not a target mistake."""
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def resolve_tool_error_target(injection: Injection, intercepted_tool_name: str) -> Injection:
|
|
249
|
+
"""TOOL_ERROR is always checked pre-dispatch, so the tool about to be
|
|
250
|
+
called is already known by fire time -- there's no need to trust the
|
|
251
|
+
Injector's payload tool_name (which may have been an approximate guess,
|
|
252
|
+
e.g. under on_next_action). Returns a copy of `injection` with
|
|
253
|
+
payload["tool_name"] corrected to the real intercepted tool; a no-op
|
|
254
|
+
for every other kind."""
|
|
255
|
+
if injection.kind != InjectionKind.TOOL_ERROR:
|
|
256
|
+
return injection
|
|
257
|
+
if injection.payload.get("tool_name") == intercepted_tool_name:
|
|
258
|
+
return injection
|
|
259
|
+
return dataclasses.replace(injection, payload={**injection.payload, "tool_name": intercepted_tool_name})
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def resolve_stale_read_target(injection: Injection, actual_entity_id: str) -> Injection:
|
|
263
|
+
"""STALE_READ fired via on_read_of_any doesn't know in advance which
|
|
264
|
+
of its candidate entities will actually get read -- correct
|
|
265
|
+
payload["entity_id"] to whichever one was, so the mutation always
|
|
266
|
+
lands on the entity the Target just perceived. A no-op for every
|
|
267
|
+
other kind, or if the payload already named the right one."""
|
|
268
|
+
if injection.kind != InjectionKind.STALE_READ:
|
|
269
|
+
return injection
|
|
270
|
+
if injection.payload.get("entity_id") == actual_entity_id:
|
|
271
|
+
return injection
|
|
272
|
+
return dataclasses.replace(injection, payload={**injection.payload, "entity_id": actual_entity_id})
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def apply_injection(world: "EntityWorld", injection: Injection, pending_tool_errors: dict[str, int]) -> str:
|
|
276
|
+
"""Applies one injection to the live world (or, for TOOL_ERROR, to the
|
|
277
|
+
pending-error registry the runner consults before dispatching a tool
|
|
278
|
+
call). Called at FIRE time, once the injection's trigger has matched --
|
|
279
|
+
never at arm time. Returns a human-readable description of the effect,
|
|
280
|
+
for the trajectory record and the report.
|
|
281
|
+
|
|
282
|
+
Every kind below reads its payload via direct p["key"] indexing --
|
|
283
|
+
correct for a well-formed payload, but a model-supplied JSON payload
|
|
284
|
+
missing a required key (or with a value of the wrong type, e.g. a
|
|
285
|
+
non-numeric TOOL_ERROR count) would otherwise raise a raw KeyError/
|
|
286
|
+
ValueError/TypeError, uncaught by the runner (which only catches
|
|
287
|
+
InjectionError) and crashing the whole run. This wrapper converts
|
|
288
|
+
that whole class of malformed-payload failure into one, instead of
|
|
289
|
+
needing a bespoke presence/type check before every single field
|
|
290
|
+
access across every kind.
|
|
291
|
+
"""
|
|
292
|
+
try:
|
|
293
|
+
return _apply_injection_impl(world, injection, pending_tool_errors)
|
|
294
|
+
except (KeyError, ValueError, TypeError) as e:
|
|
295
|
+
raise InjectionError(
|
|
296
|
+
f"{injection.kind.value} payload is malformed ({type(e).__name__}: {e}) -- payload was {injection.payload}"
|
|
297
|
+
) from e
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _apply_injection_impl(world: "EntityWorld", injection: Injection, pending_tool_errors: dict[str, int]) -> str:
|
|
301
|
+
"""Written against the generic entity protocol (entity_store/
|
|
302
|
+
get_entity/set_entity_field/clone_entity/append_note -- see domain.py's
|
|
303
|
+
EntityWorld) rather than world.tickets/orders/customers by name -- so
|
|
304
|
+
the same dispatch logic here works against a structurally different
|
|
305
|
+
domain's own world class, as long as it implements that same
|
|
306
|
+
five-method protocol.
|
|
307
|
+
"""
|
|
308
|
+
p = injection.payload
|
|
309
|
+
kind = injection.kind
|
|
310
|
+
|
|
311
|
+
if kind == InjectionKind.TOOL_ERROR:
|
|
312
|
+
tool_name = p["tool_name"]
|
|
313
|
+
count = int(p.get("count", 1))
|
|
314
|
+
pending_tool_errors[tool_name] = pending_tool_errors.get(tool_name, 0) + count
|
|
315
|
+
return f"next {count} call(s) to {tool_name} will raise a server error"
|
|
316
|
+
|
|
317
|
+
if kind == InjectionKind.STALE_READ:
|
|
318
|
+
entity_type, entity_id, field_name, new_value = (
|
|
319
|
+
p["entity_type"],
|
|
320
|
+
p["entity_id"],
|
|
321
|
+
p["field"],
|
|
322
|
+
p["new_value"],
|
|
323
|
+
)
|
|
324
|
+
if not world.entity_exists(entity_type, entity_id):
|
|
325
|
+
raise InjectionError(f"STALE_READ target {entity_type}:{entity_id} does not exist")
|
|
326
|
+
entity = world.get_entity(entity_type, entity_id)
|
|
327
|
+
if not hasattr(entity, field_name):
|
|
328
|
+
# A hallucinated/invalid field name would otherwise raise a
|
|
329
|
+
# raw AttributeError here -- uncaught by the runner (which
|
|
330
|
+
# only catches InjectionError), crashing the whole run
|
|
331
|
+
# instead of just failing this one injection.
|
|
332
|
+
raise InjectionError(f"STALE_READ field {field_name!r} does not exist on {entity_type} {entity_id}")
|
|
333
|
+
old_value = getattr(entity, field_name)
|
|
334
|
+
world.set_entity_field(entity_type, entity_id, field_name, new_value)
|
|
335
|
+
return f"{entity_type} {entity_id}.{field_name} silently changed {old_value!r} -> {new_value!r}"
|
|
336
|
+
|
|
337
|
+
if kind in (InjectionKind.CONTRADICTION, InjectionKind.LATE_INFO, InjectionKind.PROMPT_INJECTION):
|
|
338
|
+
# "ticket_id" is the payload key every current Injector prompt
|
|
339
|
+
# actually emits (ticket-domain vocabulary); "entity_type"/
|
|
340
|
+
# "entity_id" is the generic form a different domain's Injector
|
|
341
|
+
# prompt can emit instead. Both are accepted so this dispatch
|
|
342
|
+
# logic doesn't need to change when a new domain starts using it.
|
|
343
|
+
entity_type = p.get("entity_type", "ticket")
|
|
344
|
+
entity_id = p["entity_id"] if "entity_id" in p else p["ticket_id"]
|
|
345
|
+
note = p["note"]
|
|
346
|
+
if not world.entity_exists(entity_type, entity_id):
|
|
347
|
+
raise InjectionError(f"{kind.value} target {entity_type} {entity_id} does not exist")
|
|
348
|
+
world.append_note(entity_type, entity_id, note)
|
|
349
|
+
return f"new note appended to {entity_type} {entity_id}: {note!r}"
|
|
350
|
+
|
|
351
|
+
if kind == InjectionKind.AMBIGUITY:
|
|
352
|
+
entity_type, source_id, new_id = p["entity_type"], p["source_id"], p["new_id"]
|
|
353
|
+
overrides = dict(p.get("overrides", {}))
|
|
354
|
+
# "id" is redundant with new_id (and, if a model supplies both,
|
|
355
|
+
# conflicts with it as a duplicate replace() keyword argument) --
|
|
356
|
+
# new_id is authoritative.
|
|
357
|
+
overrides.pop("id", None)
|
|
358
|
+
if not world.entity_exists(entity_type, source_id):
|
|
359
|
+
raise InjectionError(f"AMBIGUITY source {entity_type}:{source_id} does not exist")
|
|
360
|
+
source_entity = world.get_entity(entity_type, source_id)
|
|
361
|
+
for key, value in list(overrides.items()):
|
|
362
|
+
# Model-supplied payload came through JSON -- lists, not
|
|
363
|
+
# whatever tuple-typed field this entity's notes/history lives
|
|
364
|
+
# in (Ticket.notes here). dataclasses.replace() doesn't
|
|
365
|
+
# enforce the field's type annotation at runtime, so left
|
|
366
|
+
# uncoerced this plants an entity that crashes the next
|
|
367
|
+
# append_note. Checked against the *current value's* actual
|
|
368
|
+
# runtime type, not a string-parsed type annotation, so this
|
|
369
|
+
# holds for any entity type/field, not just this domain's.
|
|
370
|
+
if isinstance(value, list) and isinstance(getattr(source_entity, key, None), tuple):
|
|
371
|
+
overrides[key] = tuple(value)
|
|
372
|
+
if world.entity_exists(entity_type, new_id):
|
|
373
|
+
# clone_entity has no idea this id is already taken -- it
|
|
374
|
+
# would silently overwrite that existing entity (e.g. a
|
|
375
|
+
# colliding id in a decoy scenario's cluster of similarly-
|
|
376
|
+
# numbered tickets), corrupting world state without any
|
|
377
|
+
# error. The model's own payload contract says new_id must
|
|
378
|
+
# be fresh; enforce it here rather than trust it.
|
|
379
|
+
raise InjectionError(
|
|
380
|
+
f"AMBIGUITY new_id {entity_type}:{new_id} already exists -- would silently overwrite it"
|
|
381
|
+
)
|
|
382
|
+
valid_fields = {f.name for f in dataclasses.fields(source_entity)}
|
|
383
|
+
bogus_fields = set(overrides) - valid_fields
|
|
384
|
+
if bogus_fields:
|
|
385
|
+
# A hallucinated override key would otherwise raise a raw
|
|
386
|
+
# TypeError from dataclasses.replace() -- uncaught by the
|
|
387
|
+
# runner, crashing the whole run instead of failing just this
|
|
388
|
+
# one injection.
|
|
389
|
+
raise InjectionError(
|
|
390
|
+
f"AMBIGUITY overrides for {entity_type} reference nonexistent field(s) {sorted(bogus_fields)}"
|
|
391
|
+
)
|
|
392
|
+
world.clone_entity(entity_type, source_id, new_id, overrides)
|
|
393
|
+
return f"planted near-duplicate {entity_type} {new_id} (cloned from {source_id}, overrides={overrides})"
|
|
394
|
+
|
|
395
|
+
raise InjectionError(f"unknown injection kind {kind!r}")
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def trigger_matches_pre_dispatch(
|
|
399
|
+
trigger: Trigger, tool_name: str, step_index: int, tool_call_counts: dict[str, int], commit_tools=COMMIT_TOOLS
|
|
400
|
+
) -> bool:
|
|
401
|
+
"""Checked BEFORE the Target's chosen action executes -- only these
|
|
402
|
+
three kinds can intercept a call outright (turn it into an error).
|
|
403
|
+
`tool_call_counts` holds successful-call counts so far, NOT including
|
|
404
|
+
the about-to-happen call. `commit_tools` defaults to the ticket
|
|
405
|
+
domain's for backward compatibility -- callers driving a different
|
|
406
|
+
domain must pass that domain's own commit_tools (see domain.py's
|
|
407
|
+
Domain.commit_tools), or ON_ANY_COMMIT will only ever match by
|
|
408
|
+
coincidence, on whichever tool names happen to also exist in the
|
|
409
|
+
ticket domain.
|
|
410
|
+
"""
|
|
411
|
+
if trigger.kind == TriggerKind.ON_TOOL_CALL:
|
|
412
|
+
return trigger.tool_name == tool_name
|
|
413
|
+
if trigger.kind == TriggerKind.ON_NTH_TOOL_CALL:
|
|
414
|
+
return trigger.tool_name == tool_name and tool_call_counts.get(tool_name, 0) + 1 == trigger.n
|
|
415
|
+
if trigger.kind == TriggerKind.ON_ANY_COMMIT:
|
|
416
|
+
return tool_name in commit_tools
|
|
417
|
+
if trigger.kind == TriggerKind.ON_NEXT_ACTION:
|
|
418
|
+
return True
|
|
419
|
+
if trigger.kind == TriggerKind.ON_STEP:
|
|
420
|
+
return trigger.step == step_index
|
|
421
|
+
return False
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def trigger_matches_post_dispatch(trigger: Trigger, call: "CallRecord") -> bool:
|
|
425
|
+
"""Checked AFTER a successful call -- these kinds react to something
|
|
426
|
+
the Target just did, rather than intercepting it."""
|
|
427
|
+
if trigger.kind == TriggerKind.ON_READ_OF:
|
|
428
|
+
return not call.is_commit and trigger.entity_id in call.args.values()
|
|
429
|
+
if trigger.kind == TriggerKind.ON_READ_OF_ANY:
|
|
430
|
+
assert trigger.entity_ids is not None
|
|
431
|
+
return not call.is_commit and any(eid in call.args.values() for eid in trigger.entity_ids)
|
|
432
|
+
if trigger.kind == TriggerKind.AFTER_COMMIT:
|
|
433
|
+
assert trigger.commit_pattern is not None
|
|
434
|
+
return call.is_commit and trigger.commit_pattern.matches(call.tool_name, call.args)
|
|
435
|
+
if trigger.kind == TriggerKind.AFTER_ANY_COMMIT:
|
|
436
|
+
return call.is_commit
|
|
437
|
+
return False
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def matched_entity_id(trigger: Trigger, call: "CallRecord") -> Optional[str]:
|
|
441
|
+
"""For ON_READ_OF_ANY: which of the trigger's candidate entities was
|
|
442
|
+
actually present in this read's args. None for every other trigger
|
|
443
|
+
kind, or if none of the candidates match (shouldn't happen if this is
|
|
444
|
+
only called after trigger_matches_post_dispatch returned True)."""
|
|
445
|
+
if trigger.kind != TriggerKind.ON_READ_OF_ANY:
|
|
446
|
+
return None
|
|
447
|
+
assert trigger.entity_ids is not None
|
|
448
|
+
return next((v for v in call.args.values() if v in trigger.entity_ids), None)
|
|
449
|
+
|
|
450
|
+
|
|
451
|
+
def injection_was_triggered(applied: AppliedInjection, steps: list["Step"]) -> bool:
|
|
452
|
+
"""True if the Target had a chance to perceive this FIRED injection's
|
|
453
|
+
effect.
|
|
454
|
+
|
|
455
|
+
TOOL_ERROR now only ever fires via interception (the arm/trigger model
|
|
456
|
+
guarantees this -- see trigger_matches_pre_dispatch), so any TOOL_ERROR
|
|
457
|
+
in the fired list was, by construction, actually encountered.
|
|
458
|
+
World-mutation kinds still need at least one more tool call after
|
|
459
|
+
firing to be observable at all -- the Target only perceives world
|
|
460
|
+
state through tool calls, it has no push channel.
|
|
461
|
+
"""
|
|
462
|
+
if applied.injection.kind == InjectionKind.TOOL_ERROR:
|
|
463
|
+
return True
|
|
464
|
+
return any(s.index > applied.fired_at_step for s in steps)
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def check_pending_tool_error(pending_tool_errors: dict[str, int], tool_name: str) -> str | None:
|
|
468
|
+
"""Consumes one pending injected error for `tool_name`, if any. Returns
|
|
469
|
+
the synthetic error message, or None if the call should proceed
|
|
470
|
+
normally."""
|
|
471
|
+
remaining = pending_tool_errors.get(tool_name, 0)
|
|
472
|
+
if remaining <= 0:
|
|
473
|
+
return None
|
|
474
|
+
pending_tool_errors[tool_name] = remaining - 1
|
|
475
|
+
return "HTTP 500 Internal Server Error: upstream service temporarily unavailable, please retry"
|