agentprobe-testing 0.8.1__tar.gz → 0.8.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/PKG-INFO +1 -1
  2. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/__init__.py +1 -1
  3. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/report.py +86 -9
  4. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe_testing.egg-info/PKG-INFO +1 -1
  5. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/pyproject.toml +1 -1
  6. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_report.py +106 -0
  7. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/LICENSE +0 -0
  8. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/README.md +0 -0
  9. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/agents/__init__.py +0 -0
  10. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/agents/base.py +0 -0
  11. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/agents/rule_based.py +0 -0
  12. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/agents/scripted.py +0 -0
  13. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/agents/target_agent.py +0 -0
  14. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/agreement.py +0 -0
  15. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/classifier.py +0 -0
  16. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/cli.py +0 -0
  17. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/diff.py +0 -0
  18. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domain.py +0 -0
  19. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/__init__.py +0 -0
  20. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/__init__.py +0 -0
  21. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/agent.py +0 -0
  22. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/clean.py +0 -0
  23. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/complex_agent.py +0 -0
  24. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/decoy.py +0 -0
  25. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/domain.py +0 -0
  26. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/entities.py +0 -0
  27. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/injector_prompt.py +0 -0
  28. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/rule_based_agent.py +0 -0
  29. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/scenarios.py +0 -0
  30. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/split.py +0 -0
  31. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/tools.py +0 -0
  32. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/domains/access_control/trap.py +0 -0
  33. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/feedback.py +0 -0
  34. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/generic_world.py +0 -0
  35. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/injection.py +0 -0
  36. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/injector.py +0 -0
  37. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/llm.py +0 -0
  38. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/playbook.py +0 -0
  39. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/quickstart.py +0 -0
  40. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/reachability.py +0 -0
  41. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/registry.py +0 -0
  42. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/runner.py +0 -0
  43. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/scenario.py +0 -0
  44. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/scenarios/__init__.py +0 -0
  45. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/scenarios/clean.py +0 -0
  46. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/scenarios/decoy.py +0 -0
  47. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/scenarios/registry.py +0 -0
  48. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/scenarios/split.py +0 -0
  49. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/scenarios/trap.py +0 -0
  50. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/termui.py +0 -0
  51. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/tools.py +0 -0
  52. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/trajectory.py +0 -0
  53. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/triage.py +0 -0
  54. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/validate_scenarios.py +0 -0
  55. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe/world.py +0 -0
  56. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe_testing.egg-info/SOURCES.txt +0 -0
  57. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe_testing.egg-info/dependency_links.txt +0 -0
  58. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe_testing.egg-info/entry_points.txt +0 -0
  59. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe_testing.egg-info/requires.txt +0 -0
  60. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/agentprobe_testing.egg-info/top_level.txt +0 -0
  61. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/setup.cfg +0 -0
  62. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_access_control_agent.py +0 -0
  63. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_access_control_domain.py +0 -0
  64. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_access_control_rule_based_agent.py +0 -0
  65. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_access_control_scenarios.py +0 -0
  66. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_access_control_tools.py +0 -0
  67. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_agreement.py +0 -0
  68. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_classifier.py +0 -0
  69. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_cli.py +0 -0
  70. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_complex_access_control_agent.py +0 -0
  71. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_diff.py +0 -0
  72. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_domain.py +0 -0
  73. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_feedback.py +0 -0
  74. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_generic_world.py +0 -0
  75. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_injection.py +0 -0
  76. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_injector.py +0 -0
  77. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_llm.py +0 -0
  78. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_package_api.py +0 -0
  79. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_playbook.py +0 -0
  80. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_quickstart.py +0 -0
  81. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_reachability.py +0 -0
  82. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_registry.py +0 -0
  83. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_rule_based_agent.py +0 -0
  84. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_runner.py +0 -0
  85. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_scenarios.py +0 -0
  86. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_target_agent.py +0 -0
  87. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_termui.py +0 -0
  88. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_tools.py +0 -0
  89. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_trajectory.py +0 -0
  90. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_triage.py +0 -0
  91. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_validate_scenarios.py +0 -0
  92. {agentprobe_testing-0.8.1 → agentprobe_testing-0.8.2}/tests/test_world.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentprobe-testing
3
- Version: 0.8.1
3
+ Version: 0.8.2
4
4
  Summary: Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance.
5
5
  License: Business Source License 1.1
6
6
 
@@ -60,7 +60,7 @@ from agentprobe.scenario import CommitPattern, FactPattern, GoalSpec, Scenario
60
60
  from agentprobe.scenarios.registry import ALL_SCENARIOS as TICKET_SCENARIOS
61
61
  from agentprobe.scenarios.registry import BY_ID as TICKET_SCENARIOS_BY_ID
62
62
 
63
- __version__ = "0.8.1"
63
+ __version__ = "0.8.2"
64
64
 
65
65
  __all__ = [
66
66
  "Agent",
@@ -118,10 +118,18 @@ _REPORT_CSS = """
118
118
  .pass { color: var(--green); font-weight: 600; }
119
119
  .fail { color: var(--red); font-weight: 600; }
120
120
  .warn { color: var(--amber); font-weight: 600; }
121
+ .injected { color: var(--amber); font-weight: 600; }
122
+ tr.step-injected { background: var(--amber-bg); }
123
+ tr.step-injected td { border-bottom-color: var(--amber); }
121
124
  .injection-note { font-size: 0.83rem; margin-top: 0.5rem; padding: 0.55rem 0.7rem; border-radius: 6px;
122
125
  border-left: 3px solid var(--amber); background: var(--amber-bg); }
123
126
  .injection-note--invalid { border-left-color: var(--red); background: var(--red-bg); }
124
127
  .injection-note--expired { border-left-color: var(--border); background: var(--bg-alt); }
128
+ .injector-error-banner { margin: 1rem 0 1.5rem; padding: 0.85rem 1rem; border-radius: 10px;
129
+ border: 1px solid var(--amber); background: var(--amber-bg); font-size: 0.85rem; }
130
+ .injector-error-banner strong { display: block; margin-bottom: 0.35rem; }
131
+ .injector-error-banner ul { margin: 0.35rem 0 0; padding-left: 1.25rem; }
132
+ .injector-error-banner li { margin-top: 0.2rem; font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: 0.8em; }
125
133
  footer { margin-top: 3rem; padding-top: 1.25rem; border-top: 1px solid var(--border);
126
134
  font-size: 0.78rem; color: var(--text-dim); }
127
135
  @media print {
@@ -136,20 +144,41 @@ def _format_args(args: dict) -> str:
136
144
  return ", ".join(f"{k}={v!r}" for k, v in args.items())
137
145
 
138
146
 
147
+ def _is_injected_outcome(step) -> bool:
148
+ """True for any of the four ways a step's perceived outcome can be
149
+ chaos-injected -- not just injected_error. unverified_outcome in
150
+ particular leaves `ok=True` (an ambiguous, non-committal response, not
151
+ a clean failure), so a check that only looked at `not s.ok` would
152
+ silently drop it from a report entirely."""
153
+ return step.injected_error or step.phantom_success or step.unverified_outcome or step.freeform_override
154
+
155
+
139
156
  def _notable_steps(steps: list) -> list:
140
157
  """The steps worth showing in a human-facing report: irreversible
141
158
  actions and anything that went wrong. A successful read (get_ticket,
142
159
  get_order, ...) is the agent looking something up -- noise in a report
143
160
  meant to answer "what did it actually do," not a trace of its
144
- reasoning. Commits, errors, and injected failures are the actual
161
+ reasoning. Commits, errors, and injected outcomes are the actual
145
162
  story."""
146
- return [s for s in steps if s.is_commit or not s.ok or s.injected_error]
163
+ return [s for s in steps if s.is_commit or not s.ok or _is_injected_outcome(s)]
147
164
 
148
165
 
149
166
  def _format_step_line(step) -> str:
150
167
  call = f"{step.tool_name}({_format_args(step.tool_args)})"
151
168
  tag = " [COMMIT]" if step.is_commit else ""
152
- if step.injected_error:
169
+ if step.phantom_success:
170
+ # The call genuinely succeeded -- the world already changed -- but
171
+ # the Target was shown a fabricated failure. Labeling this "error"
172
+ # would itself misreport what actually happened.
173
+ status = yellow("secretly succeeded (told it failed)")
174
+ detail = f" {dim(_truncate(str(step.result)))}"
175
+ elif step.unverified_outcome:
176
+ status = yellow("unverified (ambiguous response)")
177
+ detail = f" {dim(_truncate(str(step.result)))}"
178
+ elif step.freeform_override:
179
+ status = yellow("injected (freeform)")
180
+ detail = f" {dim(_truncate(str(step.result)))}"
181
+ elif step.injected_error:
153
182
  status = red("injected error")
154
183
  detail = f" {dim(_truncate(str(step.result)))}"
155
184
  elif step.ok:
@@ -545,6 +574,36 @@ class Report:
545
574
 
546
575
  scenario_detail_html = self._scenario_detail_html(esc)
547
576
 
577
+ # injector.decide() raising (a bad API key, a rate limit, anything)
578
+ # is caught by runner.py and silently treated as "no injection this
579
+ # step" -- indistinguishable, from the stats above alone, from the
580
+ # Injector genuinely choosing not to intervene. Surface it loudly
581
+ # instead of leaving a failing Injector looking identical to a quiet
582
+ # one (confirmed live: an auto-revoked BYOK key produced exactly
583
+ # this silent-looking "0 fired, 0 expired" result across an entire
584
+ # run before this was traced back to the real cause).
585
+ injector_error_lines = [
586
+ f"{esc(t.scenario_id)}: {esc(err)}"
587
+ for t in self.chaos_trajectories
588
+ for err in t.injector_errors
589
+ ]
590
+ injector_error_banner = ""
591
+ if injector_error_lines:
592
+ shown = injector_error_lines[:10]
593
+ more = len(injector_error_lines) - len(shown)
594
+ items = "".join(f"<li>{line}</li>" for line in shown)
595
+ if more > 0:
596
+ items += f"<li>&hellip; and {more} more</li>"
597
+ injector_error_banner = (
598
+ "<div class='injector-error-banner'>"
599
+ f"<strong>The Injector failed on {len(injector_error_lines)} step"
600
+ f"{'s' if len(injector_error_lines) != 1 else ''} during this run</strong>"
601
+ "Each failure (a bad API key, a rate limit, anything raised by injector.decide()) was silently "
602
+ "treated as “no injection this step” -- the stats above cannot tell that apart from the "
603
+ "Injector genuinely choosing not to intervene. Injections fired/expired counts below are likely "
604
+ f"undercounted.<ul>{items}</ul></div>"
605
+ )
606
+
548
607
  return f"""<!doctype html>
549
608
  <html><head><meta charset="utf-8"><title>AgentProbe report</title>
550
609
  <style>
@@ -556,6 +615,7 @@ class Report:
556
615
  <div class="title-row"><span class="health-dot health-dot--{health}"></span><h1>AgentProbe report</h1></div>
557
616
  <div class="subtitle">mode: {esc(self.mode)} &middot; injector: {esc(self.injector_model)} &middot; target: {esc(self.target_model)} &middot; {esc(c['n'])} scenarios</div>
558
617
  </header>
618
+ {injector_error_banner}
559
619
  <div class="stats">
560
620
  {"".join(stats_html)}
561
621
  </div>
@@ -583,8 +643,24 @@ class Report:
583
643
  def step_rows(steps) -> str:
584
644
  rows = []
585
645
  for s in steps:
586
- if s.injected_error:
587
- status = "<span class='fail'>injected error</span>"
646
+ # Four distinct ways a step's PERCEIVED outcome can be
647
+ # chaos-injected (see trajectory.py's Step) -- rendering
648
+ # them all as plain "ok"/"error" would hide exactly the
649
+ # steps a chaos report exists to show. phantom_success in
650
+ # particular actually SUCCEEDED (the world changed) despite
651
+ # ok=False here, so labeling it "error" would itself
652
+ # misreport what happened -- distinct from injected_error,
653
+ # where nothing really executed.
654
+ is_injected = s.injected_error or s.phantom_success or s.unverified_outcome or s.freeform_override
655
+ row_class = " class='step-injected'" if is_injected else ""
656
+ if s.phantom_success:
657
+ status = "<span class='injected'>&#9889; secretly succeeded (told it failed)</span>"
658
+ elif s.unverified_outcome:
659
+ status = "<span class='injected'>&#9889; unverified (ambiguous response)</span>"
660
+ elif s.freeform_override:
661
+ status = "<span class='injected'>&#9889; injected (freeform)</span>"
662
+ elif s.injected_error:
663
+ status = "<span class='injected'>&#9889; injected error</span>"
588
664
  elif s.ok:
589
665
  status = "<span class='pass'>ok</span>"
590
666
  else:
@@ -592,11 +668,12 @@ class Report:
592
668
  tag = " <span class='muted'>[commit]</span>" if s.is_commit else ""
593
669
  args = ", ".join(f"{k}={v!r}" for k, v in s.tool_args.items())
594
670
  # only show the result payload for something that went
595
- # wrong -- a successful call's full JSON dump is noise, not
596
- # signal, in a report meant to show what happened
597
- result = "" if s.ok and not s.injected_error else esc(_truncate(str(s.result), 200))
671
+ # wrong or was injected -- a successful call's full JSON
672
+ # dump is noise, not signal, in a report meant to show what
673
+ # happened
674
+ result = "" if s.ok and not is_injected else esc(_truncate(str(s.result), 200))
598
675
  rows.append(
599
- f"<tr><td class='num'>{s.index}</td>"
676
+ f"<tr{row_class}><td class='num'>{s.index}</td>"
600
677
  f"<td><code>{esc(s.tool_name)}({esc(args)})</code>{tag}</td>"
601
678
  f"<td>{status}</td><td class='muted'>{result}</td></tr>"
602
679
  )
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentprobe-testing
3
- Version: 0.8.1
3
+ Version: 0.8.2
4
4
  Summary: Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance.
5
5
  License: Business Source License 1.1
6
6
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "agentprobe-testing"
3
- version = "0.8.1"
3
+ version = "0.8.2"
4
4
  description = "Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance."
5
5
  requires-python = ">=3.11"
6
6
  dependencies = ["anthropic>=1.0.0", "python-dotenv>=1.0.0"]
@@ -553,6 +553,73 @@ def test_report_text_shows_injected_error_and_plain_error_step_statuses():
553
553
  assert "close_ticket" in text and "error" in text
554
554
 
555
555
 
556
+ def test_render_html_marks_a_phantom_success_step_distinctly_from_a_plain_error():
557
+ # phantom_success means the call genuinely succeeded (the world
558
+ # changed) despite ok=False here -- rendering it as a plain "error"
559
+ # would misreport what actually happened, and it must not be silently
560
+ # dropped from the notable-steps list either.
561
+ step = Step(
562
+ index=0, tool_name="issue_refund", tool_args={"amount": 75}, is_commit=True,
563
+ result={"error": "HTTP 500"}, ok=False, world_hash="h",
564
+ reachability=Reachability(status=ReachabilityStatus.ACHIEVED), progress=0,
565
+ phantom_success=True,
566
+ )
567
+ chaos = [Trajectory(scenario_id="a", steps=[step], final_status=ReachabilityStatus.ACHIEVED)]
568
+ report = Report(mode="robustness", injector_model="m", target_model="m", clean_trajectories=[], chaos_trajectories=chaos)
569
+ out = report.render_html()
570
+
571
+ assert "class='step-injected'" in out
572
+ assert "secretly succeeded" in out
573
+ assert "HTTP 500" in out # the fabricated result the Target actually saw
574
+
575
+
576
+ def test_render_html_marks_an_unverified_outcome_step_despite_ok_being_true():
577
+ # unverified_outcome is the one flag that leaves ok=True (an ambiguous,
578
+ # non-committal response, not a clean failure) -- a check that only
579
+ # looked at `not ok` would silently drop this step from the report.
580
+ step = Step(
581
+ index=0, tool_name="issue_refund", tool_args={"amount": 75}, is_commit=True,
582
+ result={"status": "pending confirmation"}, ok=True, world_hash="h",
583
+ reachability=Reachability(status=ReachabilityStatus.ACHIEVED), progress=0,
584
+ unverified_outcome=True,
585
+ )
586
+ chaos = [Trajectory(scenario_id="a", steps=[step], final_status=ReachabilityStatus.ACHIEVED)]
587
+ report = Report(mode="robustness", injector_model="m", target_model="m", clean_trajectories=[], chaos_trajectories=chaos)
588
+ out = report.render_html()
589
+
590
+ assert "class='step-injected'" in out
591
+ assert "unverified (ambiguous response)" in out
592
+ assert "pending confirmation" in out
593
+
594
+
595
+ def test_render_html_marks_a_freeform_override_step():
596
+ step = Step(
597
+ index=0, tool_name="get_policy", tool_args={"name": "refund"}, is_commit=False,
598
+ result={"text": "fabricated policy text"}, ok=True, world_hash="h",
599
+ reachability=Reachability(status=ReachabilityStatus.ACHIEVED), progress=0,
600
+ freeform_override=True,
601
+ )
602
+ chaos = [Trajectory(scenario_id="a", steps=[step], final_status=ReachabilityStatus.ACHIEVED)]
603
+ report = Report(mode="robustness", injector_model="m", target_model="m", clean_trajectories=[], chaos_trajectories=chaos)
604
+ out = report.render_html()
605
+
606
+ assert "class='step-injected'" in out
607
+ assert "injected (freeform)" in out
608
+
609
+
610
+ def test_report_text_marks_a_phantom_success_step_distinctly():
611
+ step = Step(
612
+ index=0, tool_name="issue_refund", tool_args={}, is_commit=True, result={"error": "HTTP 500"},
613
+ ok=False, world_hash="h", reachability=Reachability(status=ReachabilityStatus.ACHIEVED),
614
+ progress=0, phantom_success=True,
615
+ )
616
+ chaos = [Trajectory(scenario_id="a", steps=[step], final_status=ReachabilityStatus.ACHIEVED)]
617
+ report = Report(mode="robustness", injector_model="m", target_model="m", clean_trajectories=[], chaos_trajectories=chaos)
618
+ text = report.render_scenario_detail()
619
+
620
+ assert "secretly succeeded" in text
621
+
622
+
556
623
  def test_report_text_shows_a_discarded_note_for_an_invalid_injection():
557
624
  applied = make_applied(InjectionKind.TOOL_ERROR, valid=False)
558
625
  chaos = [make_trajectory_with_injection("a", "chaos", True, applied=applied)]
@@ -602,3 +669,42 @@ def test_report_text_notes_unobservable_injections():
602
669
  )
603
670
  text = report.render()
604
671
  assert "of which unobservable: 1" in text
672
+
673
+
674
+ def test_render_html_shows_a_warning_banner_when_the_injector_raised():
675
+ # injector.decide() raising (a bad API key, a rate limit, ...) is caught
676
+ # by runner.py and silently treated as "no injection this step" -- from
677
+ # the stats alone that's indistinguishable from the Injector genuinely
678
+ # choosing not to intervene. The report must say so explicitly instead.
679
+ chaos = make_trajectory("a", "chaos", True)
680
+ chaos.injector_errors = ["step 3: injector.decide() raised AuthenticationError(...)"]
681
+ report = Report(
682
+ mode="robustness", injector_model="m", target_model="m",
683
+ clean_trajectories=[], chaos_trajectories=[chaos],
684
+ )
685
+ out = report.render_html()
686
+ assert "<div class='injector-error-banner'>" in out
687
+ assert "The Injector failed on 1 step" in out
688
+ assert "AuthenticationError" in out
689
+
690
+
691
+ def test_render_html_omits_the_injector_error_banner_when_nothing_failed():
692
+ chaos = [make_trajectory("a", "chaos", True)]
693
+ report = Report(
694
+ mode="robustness", injector_model="m", target_model="m",
695
+ clean_trajectories=[], chaos_trajectories=chaos,
696
+ )
697
+ out = report.render_html()
698
+ assert "<div class='injector-error-banner'>" not in out
699
+
700
+
701
+ def test_render_html_pluralizes_the_injector_error_banner_and_caps_the_listed_lines():
702
+ chaos = make_trajectory("a", "chaos", True)
703
+ chaos.injector_errors = [f"step {i}: boom" for i in range(12)]
704
+ report = Report(
705
+ mode="robustness", injector_model="m", target_model="m",
706
+ clean_trajectories=[], chaos_trajectories=[chaos],
707
+ )
708
+ out = report.render_html()
709
+ assert "The Injector failed on 12 steps" in out
710
+ assert "and 2 more" in out # 12 total, only the first 10 are listed individually