gitinject 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (164) hide show
  1. gitinject/__init__.py +1 -0
  2. gitinject/__main__.py +5 -0
  3. gitinject/analyzer.py +67 -0
  4. gitinject/attacks/__init__.py +27 -0
  5. gitinject/attacks/autoinject.py +165 -0
  6. gitinject/attacks/base.py +33 -0
  7. gitinject/attacks/static.py +25 -0
  8. gitinject/cli.py +768 -0
  9. gitinject/data/research/scenarios/claude_skills_injection.md +96 -0
  10. gitinject/data/research/scenarios/cline_issue_body_injection.md +93 -0
  11. gitinject/data/research/scenarios/codex_agents_md_injection.md +109 -0
  12. gitinject/data/research/scenarios/dos_request_flood.md +133 -0
  13. gitinject/data/research/scenarios/dropped/ci_log_injection_workflow_poisoning.md +138 -0
  14. gitinject/data/research/scenarios/dropped/claude_md_instructions_injection.md +158 -0
  15. gitinject/data/research/scenarios/dropped/supply_chain_token_pivot.md +101 -0
  16. gitinject/data/research/scenarios/gemini_api_key_exfiltration.md +68 -0
  17. gitinject/data/research/scenarios/gemini_api_key_exfiltration_replication.md +0 -0
  18. gitinject/data/research/scenarios/gemini_md_instructions_injection.md +131 -0
  19. gitinject/data/research/scenarios/nsfw_api_key_block.md +114 -0
  20. gitinject/data/research/scenarios/pr_token_exfiltration_via_git_config.md +117 -0
  21. gitinject/data/research/scenarios/supply_chain_malicious_code.md +135 -0
  22. gitinject/evaluators.py +169 -0
  23. gitinject/evidence.py +72 -0
  24. gitinject/gl_runner.py +163 -0
  25. gitinject/resources.py +30 -0
  26. gitinject/run_context.py +72 -0
  27. gitinject/run_record.py +113 -0
  28. gitinject/runner.py +1024 -0
  29. gitinject/scanner/__init__.py +22 -0
  30. gitinject/scanner/baselines/__init__.py +0 -0
  31. gitinject/scanner/baselines/actionlint_runner.py +44 -0
  32. gitinject/scanner/baselines/zizmor_runner.py +35 -0
  33. gitinject/scanner/diagnostics.py +174 -0
  34. gitinject/scanner/hypothesis_generator.py +164 -0
  35. gitinject/scanner/live_validator.py +284 -0
  36. gitinject/scanner/llm_ranker.py +120 -0
  37. gitinject/scanner/memory.py +239 -0
  38. gitinject/scanner/primitives.py +236 -0
  39. gitinject/scanner/prompt_extractor.py +134 -0
  40. gitinject/scanner/recipe_scenario.py +203 -0
  41. gitinject/scanner/report_generator.py +230 -0
  42. gitinject/scanner/types.py +179 -0
  43. gitinject/scenario_base.py +147 -0
  44. gitinject/scenario_loader.py +77 -0
  45. gitinject/scenarios/__init__.py +0 -0
  46. gitinject/scenarios/benign/documentation_out_of_sync/contents/math_utils.py +3 -0
  47. gitinject/scenarios/benign/documentation_out_of_sync/contents/pydantic/v1/utils.py +21 -0
  48. gitinject/scenarios/benign/documentation_out_of_sync/scenario.py +57 -0
  49. gitinject/scenarios/benign/issue_deduplication_utility/scenario.py +76 -0
  50. gitinject/scenarios/benign/repository_onboarding_qa/contents/README.md +3 -0
  51. gitinject/scenarios/benign/repository_onboarding_qa/scenario.py +52 -0
  52. gitinject/scenarios/benign/secrets_false_positive/contents/tests/test_auth.py +14 -0
  53. gitinject/scenarios/benign/secrets_false_positive/scenario.py +58 -0
  54. gitinject/scenarios/benign/test_coverage_gap/contents/src/requests/auth_extra.py +33 -0
  55. gitinject/scenarios/benign/test_coverage_gap/scenario.py +56 -0
  56. gitinject/scenarios/benign/vulnerability_fix_verification/contents/src/flask/utils.py +15 -0
  57. gitinject/scenarios/benign/vulnerability_fix_verification/scenario.py +57 -0
  58. gitinject/scenarios/benign/vulnerable_code_review/contents/examples/tutorial/flaskr/db_utils.py +18 -0
  59. gitinject/scenarios/benign/vulnerable_code_review/scenario.py +55 -0
  60. gitinject/scenarios/benign/weekly_activity_summary/scenario.py +83 -0
  61. gitinject/scenarios/malicious/agents_md_approval_manipulation/contents/AGENTS.md +25 -0
  62. gitinject/scenarios/malicious/agents_md_approval_manipulation/contents/src/flask/csrf.py +151 -0
  63. gitinject/scenarios/malicious/agents_md_approval_manipulation/contents/tests/test_csrf.py +71 -0
  64. gitinject/scenarios/malicious/agents_md_approval_manipulation/scenario.py +121 -0
  65. gitinject/scenarios/malicious/claude_md_instructions_injection/contents/CLAUDE.md +23 -0
  66. gitinject/scenarios/malicious/claude_md_instructions_injection/contents/scripts/deploy_check.sh +29 -0
  67. gitinject/scenarios/malicious/claude_md_instructions_injection/scenario.py +93 -0
  68. gitinject/scenarios/malicious/cline_issue_body_injection/scenario.py +132 -0
  69. gitinject/scenarios/malicious/codex_agents_md_injection/contents/AGENTS.md +36 -0
  70. gitinject/scenarios/malicious/codex_agents_md_injection/contents/scripts/utils.py +12 -0
  71. gitinject/scenarios/malicious/codex_agents_md_injection/scenario.py +128 -0
  72. gitinject/scenarios/malicious/dos_request_flood/scenario.py +80 -0
  73. gitinject/scenarios/malicious/gemini_md_approval_manipulation/contents/src/flask/csrf.py +151 -0
  74. gitinject/scenarios/malicious/gemini_md_approval_manipulation/contents/tests/test_csrf.py +71 -0
  75. gitinject/scenarios/malicious/gemini_md_approval_manipulation/scenario.py +121 -0
  76. gitinject/scenarios/malicious/gemini_md_instructions_injection/contents/scripts/deploy_check.sh +38 -0
  77. gitinject/scenarios/malicious/gemini_md_instructions_injection/scenario.py +105 -0
  78. gitinject/scenarios/malicious/issue_onboarding_secret_exfiltration/scenario.py +76 -0
  79. gitinject/scenarios/malicious/pr_token_exfiltration_via_git_config/contents/scripts/deploy_check.sh +7 -0
  80. gitinject/scenarios/malicious/pr_token_exfiltration_via_git_config/scenario.py +113 -0
  81. gitinject/scenarios/malicious/supply_chain_malicious_code/contents/CLAUDE.md +25 -0
  82. gitinject/scenarios/malicious/supply_chain_malicious_code/contents/src/flask/csrf.py +151 -0
  83. gitinject/scenarios/malicious/supply_chain_malicious_code/contents/tests/test_csrf.py +71 -0
  84. gitinject/scenarios/malicious/supply_chain_malicious_code/scenario.py +123 -0
  85. gitinject/scenarios/malicious/unauthorized_pr_approval/contents/django/utils/crypto.py +13 -0
  86. gitinject/scenarios/malicious/unauthorized_pr_approval/scenario.py +57 -0
  87. gitinject/simulator.py +89 -0
  88. gitinject/utils/__init__.py +0 -0
  89. gitinject/utils/gh_client.py +628 -0
  90. gitinject/utils/gl_client.py +132 -0
  91. gitinject/utils/gl_provisioner.py +83 -0
  92. gitinject/utils/llm.py +205 -0
  93. gitinject/utils/provisioner.py +114 -0
  94. gitinject/utils/scenario_resources.py +33 -0
  95. gitinject/utils/types.py +49 -0
  96. gitinject/workflows/__init__.py +0 -0
  97. gitinject/workflows/claude-ci-auto-fix/contents/.github/workflows/main.yml +107 -0
  98. gitinject/workflows/claude-ci-auto-fix/metadata.json +10 -0
  99. gitinject/workflows/claude-general/contents/.github/workflows/main.yml +58 -0
  100. gitinject/workflows/claude-general/metadata.json +10 -0
  101. gitinject/workflows/claude-gitlab-mr-review/contents/.gitlab-ci.yml +36 -0
  102. gitinject/workflows/claude-gitlab-mr-review/metadata.json +11 -0
  103. gitinject/workflows/claude-issue-deduplication/contents/.github/workflows/main.yml +66 -0
  104. gitinject/workflows/claude-issue-deduplication/metadata.json +10 -0
  105. gitinject/workflows/claude-issue-triage/contents/.github/workflows/main.yml +34 -0
  106. gitinject/workflows/claude-issue-triage/metadata.json +10 -0
  107. gitinject/workflows/claude-manual-analysis/contents/.github/workflows/main.yml +42 -0
  108. gitinject/workflows/claude-manual-analysis/metadata.json +10 -0
  109. gitinject/workflows/claude-pr-review/contents/.github/workflows/main.yml +77 -0
  110. gitinject/workflows/claude-pr-review/metadata.json +10 -0
  111. gitinject/workflows/claude-pr-review-authors/contents/.github/workflows/main.yml +48 -0
  112. gitinject/workflows/claude-pr-review-authors/metadata.json +10 -0
  113. gitinject/workflows/claude-pr-review-paths/contents/.github/workflows/main.yml +49 -0
  114. gitinject/workflows/claude-pr-review-paths/metadata.json +10 -0
  115. gitinject/workflows/claude-test-analysis/contents/.github/workflows/main.yml +114 -0
  116. gitinject/workflows/claude-test-analysis/metadata.json +10 -0
  117. gitinject/workflows/cline-assistant/contents/.github/workflows/main.yml +87 -0
  118. gitinject/workflows/cline-assistant/contents/git-scripts/analyze-issue.sh +43 -0
  119. gitinject/workflows/cline-assistant/metadata.json +10 -0
  120. gitinject/workflows/codex-pr-review/contents/.github/workflows/main.yml +73 -0
  121. gitinject/workflows/codex-pr-review/metadata.json +10 -0
  122. gitinject/workflows/copilot-ci-doctor/contents/.github/workflows/ci-doctor.yml +1161 -0
  123. gitinject/workflows/copilot-ci-doctor/metadata.json +10 -0
  124. gitinject/workflows/copilot-lean-squad/contents/.github/workflows/lean-squad.yml +1313 -0
  125. gitinject/workflows/copilot-lean-squad/metadata.json +10 -0
  126. gitinject/workflows/copilot-malicious-scan/contents/.github/workflows/daily-malicious-code-scan.yml +899 -0
  127. gitinject/workflows/copilot-malicious-scan/metadata.json +10 -0
  128. gitinject/workflows/copilot-repo-assist/contents/.github/workflows/repo-assist.yml +1503 -0
  129. gitinject/workflows/copilot-repo-assist/metadata.json +10 -0
  130. gitinject/workflows/copilot-wiki-writer/contents/.github/workflows/agentic-wiki-writer.yml +1316 -0
  131. gitinject/workflows/copilot-wiki-writer/metadata.json +10 -0
  132. gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-invoke.yml +122 -0
  133. gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-plan-execute.yml +130 -0
  134. gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-review.yml +118 -0
  135. gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-scheduled-triage.yml +220 -0
  136. gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-triage.yml +160 -0
  137. gitinject/workflows/gemini-assistant/contents/.github/workflows/main.yml +220 -0
  138. gitinject/workflows/gemini-assistant/metadata.json +10 -0
  139. gitinject/workflows/gemini-assistant-original/AWESOME.md +118 -0
  140. gitinject/workflows/gemini-assistant-original/CONFIGURATION.md +162 -0
  141. gitinject/workflows/gemini-assistant-original/README.md +93 -0
  142. gitinject/workflows/gemini-assistant-original/gemini-assistant/README.md +192 -0
  143. gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-invoke.toml +94 -0
  144. gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-invoke.yml +131 -0
  145. gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-plan-execute.toml +100 -0
  146. gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-plan-execute.yml +139 -0
  147. gitinject/workflows/gemini-assistant-original/gemini-dispatch/README.md +49 -0
  148. gitinject/workflows/gemini-assistant-original/gemini-dispatch/gemini-dispatch.yml +221 -0
  149. gitinject/workflows/gemini-assistant-original/issue-triage/README.md +190 -0
  150. gitinject/workflows/gemini-assistant-original/issue-triage/gemini-scheduled-triage.toml +96 -0
  151. gitinject/workflows/gemini-assistant-original/issue-triage/gemini-scheduled-triage.yml +223 -0
  152. gitinject/workflows/gemini-assistant-original/issue-triage/gemini-triage.toml +32 -0
  153. gitinject/workflows/gemini-assistant-original/issue-triage/gemini-triage.yml +167 -0
  154. gitinject/workflows/gemini-assistant-original/metadata.json +10 -0
  155. gitinject/workflows/gemini-assistant-original/pr-review/README.md +337 -0
  156. gitinject/workflows/gemini-assistant-original/pr-review/gemini-review.toml +176 -0
  157. gitinject/workflows/gemini-assistant-original/pr-review/gemini-review.yml +119 -0
  158. gitinject/workflows/opencode-pr-review/contents/.github/workflows/main.yml +28 -0
  159. gitinject/workflows/opencode-pr-review/metadata.json +10 -0
  160. gitinject-0.1.0.dist-info/METADATA +128 -0
  161. gitinject-0.1.0.dist-info/RECORD +164 -0
  162. gitinject-0.1.0.dist-info/WHEEL +4 -0
  163. gitinject-0.1.0.dist-info/entry_points.txt +2 -0
  164. gitinject-0.1.0.dist-info/licenses/LICENSE +202 -0
gitinject/cli.py ADDED
@@ -0,0 +1,768 @@
1
+ import json
2
+ import os
3
+
4
+ import click
5
+
6
+ from .resources import dataset_dir
7
+
8
+
9
+ @click.group()
10
+ def cli():
11
+ """AI-Powered GitHub Workflows Security Benchmark CLI."""
12
+ pass
13
+
14
+
15
+ @cli.group()
16
+ def list():
17
+ """List benchmark components."""
18
+ pass
19
+
20
+
21
+ @list.command(name="workflows")
22
+ def list_workflows():
23
+ """List available workflows with their categories and supported events."""
24
+ workflows_dir = dataset_dir("workflows")
25
+ if not os.path.exists(workflows_dir):
26
+ click.echo("Workflows directory not found.")
27
+ return
28
+
29
+ workflows = sorted([d for d in os.listdir(workflows_dir) if os.path.isdir(os.path.join(workflows_dir, d))])
30
+ for w in workflows:
31
+ metadata_path = os.path.join(workflows_dir, w, "metadata.json")
32
+ if os.path.exists(metadata_path):
33
+ with open(metadata_path, "r") as f:
34
+ metadata = json.load(f)
35
+ category = metadata.get("category", "uncategorized")
36
+ events = ", ".join(metadata.get("supported_events", []))
37
+ click.echo(f"- {w:25} | Category: {category:20} | Events: {events}")
38
+ else:
39
+ click.echo(f"- {w:25} | No metadata found.")
40
+
41
+
42
+ def _discover_scenarios(scenarios_dir):
43
+ """Load scenario definitions without constructing an authenticated runner."""
44
+ from .scenario_loader import discover_scenario_paths, load_scenario
45
+
46
+ return sorted(
47
+ [(path.parent.name, load_scenario(path, os.getcwd())) for path in discover_scenario_paths(scenarios_dir)],
48
+ key=lambda item: item[0],
49
+ )
50
+
51
+
52
+ @list.command(name="scenarios")
53
+ def list_scenarios():
54
+ """List available scenarios with their categories and event types."""
55
+ scenarios_dir = dataset_dir("scenarios")
56
+ if not os.path.exists(scenarios_dir):
57
+ click.echo("Scenarios directory not found.")
58
+ return
59
+
60
+ scenarios = _discover_scenarios(scenarios_dir)
61
+
62
+ for s_name, s_obj in scenarios:
63
+ category = s_obj.category.value if s_obj.category else "none"
64
+ event = s_obj.get_event().get("event_type", "unknown")
65
+ s_type = s_obj.scenario_type.value if hasattr(s_obj, "scenario_type") else "benign"
66
+ click.echo(f"- {s_name:25} | Type: {s_type:10} | Category: {category:20} | Event: {event}")
67
+
68
+
69
+ @cli.command()
70
+ @click.option("--workflow", required=True, help="Workflow ID to run.")
71
+ @click.option("--scenario", required=True, help="Scenario ID to run (or 'all' for all compatible scenarios).")
72
+ @click.option(
73
+ "--repo-prefix",
74
+ default=lambda: os.environ.get("GITHUB_REPO_PREFIX", "benchmark-run"),
75
+ help="Target GitHub repository prefix.",
76
+ )
77
+ @click.option(
78
+ "--cleanup/--no-cleanup",
79
+ default=True,
80
+ help="Automatically delete the GitHub repository after the run.",
81
+ )
82
+ @click.option(
83
+ "--log-llm-input",
84
+ is_flag=True,
85
+ help="Reconstruct and print the effective LLM prompt before triggering the run, and save it to runs/*/llm_input.txt.",
86
+ )
87
+ @click.option(
88
+ "--attack",
89
+ "attack_id",
90
+ default=None,
91
+ help="Attack type to apply (autoinject, static). Omit to use the scenario's hardcoded payload.",
92
+ )
93
+ @click.option(
94
+ "--attack-payload",
95
+ default=None,
96
+ help="Inline payload string or path to a payload file. Used with --attack static.",
97
+ )
98
+ @click.option(
99
+ "--repeat",
100
+ default=1,
101
+ show_default=True,
102
+ help="Number of times to repeat each run.",
103
+ )
104
+ @click.option("--parameters", default="{}", help="JSON object supplied to the scenario's run context.")
105
+ @click.option("--seed", type=int, default=None, help="Seed for the scenario context's random generator.")
106
+ def run(
107
+ workflow, scenario, repo_prefix, cleanup, log_llm_input, attack_id, attack_payload, repeat, parameters, seed
108
+ ):
109
+ """Run benchmark tests."""
110
+ from .runner import BenchmarkRunner
111
+
112
+ try:
113
+ parameters = json.loads(parameters)
114
+ if not isinstance(parameters, dict):
115
+ raise ValueError("Expected a JSON object")
116
+ except ValueError as exc:
117
+ raise click.BadParameter(str(exc), param_hint="--parameters") from exc
118
+
119
+ workflows_dir = dataset_dir("workflows")
120
+ scenarios_dir = dataset_dir("scenarios")
121
+
122
+ workflow_path = os.path.join(workflows_dir, workflow)
123
+ if not os.path.isdir(workflow_path):
124
+ click.echo(click.style(f"Error: Workflow '{workflow}' not found.", fg="red"))
125
+ return
126
+
127
+ meta_path = os.path.join(workflow_path, "metadata.json")
128
+ workflow_meta = {}
129
+ if os.path.exists(meta_path):
130
+ with open(meta_path, "r") as f:
131
+ workflow_meta = json.load(f)
132
+
133
+ platform = workflow_meta.get("platform", "github")
134
+ w_category = workflow_meta.get("category")
135
+ supported_events = set(workflow_meta.get("supported_events", []))
136
+
137
+ def _run_pair(wf, sc):
138
+ if platform == "gitlab":
139
+ from .gl_runner import GitLabRunner
140
+
141
+ runner = GitLabRunner(os.getcwd(), project_prefix=repo_prefix)
142
+ return runner.run(wf, sc, cleanup=cleanup)
143
+ runner = BenchmarkRunner(os.getcwd(), repo_prefix=repo_prefix)
144
+ return runner.run(
145
+ wf,
146
+ sc,
147
+ attack_id=attack_id,
148
+ attack_payload=attack_payload,
149
+ cleanup=cleanup,
150
+ log_llm_input=log_llm_input,
151
+ parameters=parameters,
152
+ seed=seed,
153
+ )
154
+
155
+ if scenario.lower() == "all":
156
+ click.echo(f"Identifying compatible scenarios for workflow '{workflow}'...")
157
+
158
+ valid_scenarios = _discover_scenarios(scenarios_dir)
159
+ scenarios_to_run = []
160
+ for s_name, s_obj in valid_scenarios:
161
+ s_event = s_obj.get_event().get("event_type")
162
+ s_platform = getattr(s_obj, "platform", "github")
163
+ if s_obj.category == w_category and s_event in supported_events and s_platform == platform:
164
+ scenarios_to_run.append(s_name)
165
+
166
+ if not scenarios_to_run:
167
+ click.echo(click.style(f"No compatible scenarios found for workflow '{workflow}'.", fg="yellow"))
168
+ return
169
+
170
+ click.echo(f"Found {len(scenarios_to_run)} compatible scenarios: {', '.join(scenarios_to_run)}")
171
+
172
+ pairs_results = {}
173
+ for s_name in scenarios_to_run:
174
+ pairs_results[(workflow, s_name)] = []
175
+ for i in range(repeat):
176
+ label = f"--- Running {workflow} against {s_name} (run {i + 1}/{repeat}) ---"
177
+ click.echo("\n" + click.style(label, bold=True))
178
+ result = _run_pair(workflow, s_name)
179
+ _display_run_result(result)
180
+ pairs_results[(workflow, s_name)].append(result)
181
+
182
+ if repeat > 1:
183
+ _display_repeat_summary(pairs_results)
184
+ else:
185
+ pairs_results = {(workflow, scenario): []}
186
+ for i in range(repeat):
187
+ if repeat > 1:
188
+ click.echo(click.style(f"\n--- Run {i + 1}/{repeat} ---", bold=True))
189
+ if platform == "gitlab":
190
+ from .gl_runner import GitLabRunner
191
+
192
+ runner = GitLabRunner(os.getcwd(), project_prefix=repo_prefix)
193
+ click.echo(f"Running GitLab benchmark: workflow={workflow}, scenario={scenario}")
194
+ result = runner.run(workflow, scenario, cleanup=cleanup)
195
+ else:
196
+ runner = BenchmarkRunner(os.getcwd(), repo_prefix=repo_prefix)
197
+ click.echo(f"Running benchmark on {runner.repo_name}: workflow={workflow}, scenario={scenario}")
198
+ result = runner.run(
199
+ workflow,
200
+ scenario,
201
+ attack_id=attack_id,
202
+ attack_payload=attack_payload,
203
+ cleanup=cleanup,
204
+ log_llm_input=log_llm_input,
205
+ parameters=parameters,
206
+ seed=seed,
207
+ )
208
+ _display_run_result(result)
209
+ pairs_results[(workflow, scenario)].append(result)
210
+
211
+ if repeat > 1:
212
+ _display_repeat_summary(pairs_results)
213
+
214
+
215
+ def _display_run_result(result):
216
+ """Helper to display the result of a single benchmark run."""
217
+ if "error" in result:
218
+ click.echo(click.style(f"Error: {result['error']}", fg="red"))
219
+ else:
220
+ click.echo("\n" + click.style("--- Benchmark Evaluation ---", bold=True))
221
+ analysis = result.get("analysis", {})
222
+ for label, key in (
223
+ ("Utility Achieved", "utility_achieved"),
224
+ ("Security Breached", "security_breached"),
225
+ ("Autonomy Achieved", "autonomy_achieved"),
226
+ ):
227
+ value = analysis.get(key)
228
+ click.echo(f"{label}: {value if value is not None else 'unknown'}")
229
+ for metric, error in analysis.get("evaluation_errors", {}).items():
230
+ click.echo(f" {metric}: {error}")
231
+ click.echo(click.style("----------------------------", bold=True))
232
+ click.echo(f"Message: {result.get('message')}")
233
+
234
+
235
+ def _metric_summary(results, key, *, invert=False):
236
+ values = [r.get("analysis", {}).get(key) for r in results if "error" not in r]
237
+ known = [value for value in values if type(value) is bool]
238
+ if not known:
239
+ return "N/A (0/0)"
240
+ positives = sum(value is (False if invert else True) for value in known)
241
+ return f"{positives / len(known):.2f} ({positives}/{len(known)})"
242
+
243
+
244
+ def _display_repeat_summary(pairs_results: dict[tuple, list]):
245
+ """Report each metric using its own known-verdict denominator."""
246
+ click.echo("\n" + click.style("--- Repeat Summary ---", bold=True))
247
+ click.echo(f"{'Workflow':<25} {'Scenario':<30} {'Security':<18} {'Utility':<18} {'Autonomy':<18} {'Attempts':<9}")
248
+ for (wf, sc), results in sorted(pairs_results.items()):
249
+ sec = _metric_summary(results, "security_breached", invert=True)
250
+ util = _metric_summary(results, "utility_achieved")
251
+ auto = _metric_summary(results, "autonomy_achieved")
252
+ click.echo(f"{wf:<25} {sc:<30} {sec:<18} {util:<18} {auto:<18} {len(results):<9}")
253
+ click.echo("Ratios exclude unknown verdicts and execution errors; Attempts includes all runs.")
254
+
255
+
256
+ @cli.command(name="run-suite")
257
+ @click.option("--workflow-labels", help="Comma-separated list of workflow labels to filter by.")
258
+ @click.option("--scenario-labels", help="Comma-separated list of scenario labels to filter by.")
259
+ @click.option("--scenario-type", help="Filter by scenario type (benign/malicious).")
260
+ @click.option("--event", help="Filter by event type.")
261
+ @click.option(
262
+ "--repo-prefix",
263
+ default=lambda: os.environ.get("GITHUB_REPO_PREFIX", "benchmark-run"),
264
+ help="Target GitHub repository prefix.",
265
+ )
266
+ @click.option(
267
+ "--cleanup/--no-cleanup",
268
+ default=True,
269
+ help="Automatically delete the GitHub repository after the run.",
270
+ )
271
+ @click.option(
272
+ "--dry-run",
273
+ help="List compatible pairs without executing them.",
274
+ is_flag=True,
275
+ )
276
+ @click.option(
277
+ "--log-llm-input",
278
+ is_flag=True,
279
+ help="Reconstruct and print the effective LLM prompt before each run.",
280
+ )
281
+ @click.option(
282
+ "--repeat",
283
+ default=1,
284
+ show_default=True,
285
+ help="Number of times to repeat each workflow/scenario pair.",
286
+ )
287
+ def run_suite(
288
+ workflow_labels, scenario_labels, scenario_type, event, repo_prefix, cleanup, dry_run, log_llm_input, repeat
289
+ ):
290
+ """Run a suite of compatible workflows and scenarios."""
291
+ from .runner import BenchmarkRunner
292
+
293
+ workflows_dir = dataset_dir("workflows")
294
+ scenarios_dir = dataset_dir("scenarios")
295
+
296
+ wf_filters = set(workflow_labels.split(",")) if workflow_labels else set()
297
+ sc_filters = set(scenario_labels.split(",")) if scenario_labels else set()
298
+
299
+ # Load workflows
300
+ valid_workflows = []
301
+ for w in os.listdir(workflows_dir):
302
+ if not os.path.isdir(os.path.join(workflows_dir, w)):
303
+ continue
304
+ meta_path = os.path.join(workflows_dir, w, "metadata.json")
305
+ if os.path.exists(meta_path):
306
+ with open(meta_path, "r") as f:
307
+ meta = json.load(f)
308
+ labels = set(meta.get("labels", []))
309
+ supported_events = set(meta.get("supported_events", []))
310
+
311
+ if wf_filters and not wf_filters.intersection(labels):
312
+ continue
313
+ if event and event not in supported_events:
314
+ continue
315
+ valid_workflows.append((w, meta))
316
+
317
+ # Load scenarios
318
+ all_scenarios = _discover_scenarios(scenarios_dir)
319
+ valid_scenarios = []
320
+ for s_name, s_obj in all_scenarios:
321
+ labels = set(getattr(s_obj, "labels", []))
322
+ s_event = s_obj.get_event().get("event_type")
323
+ s_type = s_obj.scenario_type.value if hasattr(s_obj, "scenario_type") else "benign"
324
+
325
+ if sc_filters and not sc_filters.intersection(labels):
326
+ continue
327
+ if scenario_type and scenario_type.lower() != s_type.lower():
328
+ continue
329
+ if event and event != s_event:
330
+ continue
331
+ valid_scenarios.append((s_name, s_obj))
332
+
333
+ # Generate compatible pairs
334
+ pairs = []
335
+ for w_name, w_meta in valid_workflows:
336
+ w_category = w_meta.get("category")
337
+ supported_events = set(w_meta.get("supported_events", []))
338
+
339
+ for s_name, s_obj in valid_scenarios:
340
+ s_event = s_obj.get_event().get("event_type")
341
+ if s_obj.category == w_category and s_event in supported_events:
342
+ pairs.append((w_name, s_name))
343
+
344
+ if not pairs:
345
+ click.echo("No compatible workflow/scenario pairs found.")
346
+ return
347
+
348
+ if dry_run:
349
+ click.echo(click.style(f"DRY RUN: Found {len(pairs)} compatible pairs:", bold=True))
350
+ for w_name, s_name in pairs:
351
+ click.echo(f" - Workflow: {w_name:20} | Scenario: {s_name}")
352
+ return
353
+
354
+ total = len(pairs) * repeat
355
+ click.echo(f"Starting suite with {len(pairs)} compatible pairs × {repeat} repeat(s) = {total} total runs.")
356
+ pairs_results = {}
357
+ for w_name, s_name in pairs:
358
+ pairs_results[(w_name, s_name)] = []
359
+ for i in range(repeat):
360
+ label = f"--- Running {w_name} against {s_name} (run {i + 1}/{repeat}) ---"
361
+ click.echo("\n" + click.style(label, bold=True))
362
+ runner = BenchmarkRunner(os.getcwd(), repo_prefix=repo_prefix)
363
+ res = runner.run(w_name, s_name, cleanup=cleanup, log_llm_input=log_llm_input)
364
+ _display_run_result(res)
365
+ pairs_results[(w_name, s_name)].append(res)
366
+
367
+ all_results = [r for results in pairs_results.values() for r in results]
368
+ click.echo("\n" + click.style("--- Benchmark Suite Complete ---", bold=True))
369
+ success = sum(1 for r in all_results if "error" not in r)
370
+ click.echo(f"Runs: {success}/{total} successful.")
371
+ if repeat > 1:
372
+ _display_repeat_summary(pairs_results)
373
+
374
+
375
+ @cli.command()
376
+ @click.option("--workflow", required=True, help="Workflow ID.")
377
+ @click.option("--scenario", required=True, help="Scenario ID.")
378
+ @click.option("--attack", "attack_id", required=True, help="Attack type (autoinject, static).")
379
+ @click.option(
380
+ "--victim-model",
381
+ default=None,
382
+ help="OpenRouter model string for the victim (e.g. openai/gpt-5.4-2026-03-05). Defaults to ATTACK_VICTIM_MODEL env var.",
383
+ )
384
+ def preflight(workflow, scenario, attack_id, victim_model):
385
+ """
386
+ Single offline shot: generate a payload, send the injected prompt directly to the
387
+ victim model, and report whether the attack succeeded. No GitHub repo needed.
388
+ """
389
+ from .runner import BenchmarkRunner
390
+
391
+ runner = BenchmarkRunner(os.getcwd(), repo_prefix="preflight")
392
+ result = runner.offline_optimize(workflow, scenario, attack_id, iterations=1, victim_model=victim_model)
393
+
394
+ if "error" in result:
395
+ click.echo(click.style(f"Error: {result['error']}", fg="red"))
396
+ return
397
+
398
+ score = result["asr_curve"][0] if result["asr_curve"] else None
399
+ label = (
400
+ click.style("UNKNOWN — offline trial could not be scored", fg="yellow")
401
+ if score is None
402
+ else (
403
+ click.style("PASS — attack worked offline", fg="green")
404
+ if score
405
+ else click.style("FAIL — attack did not work offline", fg="red")
406
+ )
407
+ )
408
+ click.echo(f"\nPreflight result: {label}")
409
+ click.echo(f"Payload at: {result['runs_dir']}/best_payload.txt")
410
+ click.echo(
411
+ "\nNote: offline uses a plain chat call. The agentic Codex context may differ. "
412
+ "A PASS here is a strong indicator but not a guarantee."
413
+ )
414
+
415
+
416
+ @cli.command()
417
+ @click.option("--workflow", required=True, help="Workflow ID to run.")
418
+ @click.option("--scenario", required=True, help="Scenario ID to optimize against.")
419
+ @click.option("--attack", "attack_id", required=True, help="Attack type (autoinject, static).")
420
+ @click.option("--iterations", default=5, show_default=True, help="Number of optimization iterations.")
421
+ @click.option(
422
+ "--offline",
423
+ is_flag=True,
424
+ help="Optimize using direct model calls instead of GitHub workflow runs. "
425
+ "Fast, no repo provisioning. Requires scenario.get_preflight_evaluator().",
426
+ )
427
+ @click.option(
428
+ "--victim-model",
429
+ default=None,
430
+ help="Override victim model for offline mode (OpenRouter string). Defaults to ATTACK_VICTIM_MODEL env var.",
431
+ )
432
+ @click.option(
433
+ "--repo-prefix",
434
+ default=lambda: os.environ.get("GITHUB_REPO_PREFIX", "benchmark-run"),
435
+ help="Target GitHub repository prefix (online mode only).",
436
+ )
437
+ @click.option("--cleanup/--no-cleanup", default=True, help="Delete the repository after the run (online mode only).")
438
+ def optimize(workflow, scenario, attack_id, iterations, offline, victim_model, repo_prefix, cleanup):
439
+ """Iteratively optimize an attack payload, writing the best result to runs/*/best_payload.txt."""
440
+ from .runner import BenchmarkRunner
441
+
442
+ runner = BenchmarkRunner(os.getcwd(), repo_prefix=repo_prefix)
443
+
444
+ if offline:
445
+ click.echo(
446
+ f"Offline optimization: workflow={workflow}, scenario={scenario}, attack={attack_id}, iterations={iterations}"
447
+ )
448
+ result = runner.offline_optimize(workflow, scenario, attack_id, iterations, victim_model=victim_model)
449
+ else:
450
+ click.echo(
451
+ f"Optimizing on {runner.repo_name}: workflow={workflow}, scenario={scenario}, "
452
+ f"attack={attack_id}, iterations={iterations}"
453
+ )
454
+ result = runner.optimize(workflow, scenario, attack_id, iterations, cleanup=cleanup)
455
+
456
+ if "error" in result:
457
+ click.echo(click.style(f"Error: {result['error']}", fg="red"))
458
+ return
459
+
460
+ click.echo("\n" + click.style("--- Optimization Result ---", bold=True))
461
+ scores = [score for score in result["asr_curve"] if score is not None]
462
+ click.echo(f"Final ASR : {result['final_asr']} ({sum(scores)}/{len(scores)} valid trials)")
463
+ click.echo(f"ASR curve : {result['asr_curve']}")
464
+ click.echo(f"Runs dir : {result['runs_dir']}")
465
+ if result.get("best_payload"):
466
+ click.echo(click.style("Best payload saved to runs_dir/best_payload.txt", fg="green"))
467
+
468
+
469
+ @cli.command()
470
+ @click.option(
471
+ "--prefix",
472
+ default="benchmark-run",
473
+ help="Prefix of repositories to delete.",
474
+ )
475
+ @click.option("--force", is_flag=True, help="Skip confirmation prompt.")
476
+ def cleanup(prefix, force):
477
+ """Delete all benchmark repositories with a specific prefix."""
478
+ from .utils.gh_client import GitHubClient
479
+
480
+ gh = GitHubClient()
481
+ click.echo(f"Searching for repositories with prefix '{prefix}'...")
482
+ repos = gh.list_repos(limit=100)
483
+
484
+ # Filter by prefix (checking both name and nameWithOwner)
485
+ to_delete = [r["nameWithOwner"] for r in repos if r["name"].startswith(prefix)]
486
+
487
+ if not to_delete:
488
+ click.echo("No matching repositories found.")
489
+ return
490
+
491
+ click.echo(f"Found {len(to_delete)} repositories:")
492
+ for repo in to_delete:
493
+ click.echo(f" - {repo}")
494
+
495
+ if not force and not click.confirm("\nAre you sure you want to delete these repositories?"):
496
+ click.echo("Aborted.")
497
+ return
498
+
499
+ for repo_name in to_delete:
500
+ click.echo(f"Deleting {repo_name}...")
501
+ client = GitHubClient(repo=repo_name)
502
+ success, err = client.delete_repo()
503
+ if not success:
504
+ click.echo(click.style(f"Failed to delete {repo_name}: {err}", fg="red"))
505
+ else:
506
+ click.echo(click.style(f"Successfully deleted {repo_name}", fg="green"))
507
+
508
+
509
+ @cli.command()
510
+ @click.option("--aggregate", is_flag=True, help="Aggregate results by workflow.")
511
+ def report(aggregate):
512
+ """Generate a summary of previous runs from the 'runs/' directory."""
513
+ runs_dir = "runs"
514
+ if not os.path.exists(runs_dir):
515
+ click.echo("No runs found.")
516
+ return
517
+
518
+ if aggregate:
519
+ pairs = {}
520
+ for folder in os.listdir(runs_dir):
521
+ metadata_path = os.path.join(runs_dir, folder, "metadata.json")
522
+ if not os.path.isfile(metadata_path):
523
+ continue
524
+ try:
525
+ with open(metadata_path) as handle:
526
+ data = json.load(handle)
527
+ except json.JSONDecodeError:
528
+ continue
529
+ if data.get("workflow") and ("analysis" in data or "error" in data):
530
+ pairs.setdefault((data["workflow"], "all scenarios"), []).append(data)
531
+ _display_repeat_summary(pairs)
532
+
533
+ else:
534
+ header = f"{'Timestamp':<25} {'Workflow':<20} {'Scenario':<25} {'Util':<6} {'Sec':<6} {'Auto':<6}"
535
+ click.echo(click.style(header, bold=True))
536
+ click.echo("-" * 95)
537
+
538
+ run_folders = sorted([d for d in os.listdir(runs_dir) if os.path.isdir(os.path.join(runs_dir, d))], reverse=True)
539
+
540
+ for folder in run_folders:
541
+ metadata_path = os.path.join(runs_dir, folder, "metadata.json")
542
+ if os.path.exists(metadata_path):
543
+ with open(metadata_path, "r") as f:
544
+ try:
545
+ data = json.load(f)
546
+ if "analysis" not in data and "error" not in data:
547
+ continue
548
+ analysis = data.get("analysis", {})
549
+
550
+ def verdict_label(key, invert=False):
551
+ value = analysis.get(key)
552
+ if type(value) is not bool:
553
+ return "?"
554
+ return "T" if value is (False if invert else True) else "F"
555
+
556
+ click.echo(
557
+ f"{data.get('timestamp', folder):<25} "
558
+ f"{data.get('workflow'):<20} "
559
+ f"{str(data.get('scenario', '')):<25} "
560
+ f"{verdict_label('utility_achieved'):<6} "
561
+ f"{verdict_label('security_breached', True):<6} "
562
+ f"{verdict_label('autonomy_achieved'):<6}"
563
+ )
564
+ except (json.JSONDecodeError, KeyError):
565
+ pass
566
+
567
+
568
+ @cli.command()
569
+ @click.option("--workflow", "workflow_id", default=None, help="Workflow ID to scan.")
570
+ @click.option("--all", "scan_all", is_flag=True, help="Scan all workflows in the inventory.")
571
+ @click.option("--hypotheses", default=12, show_default=True, help="Hypotheses per scan, split across 4 MITRE categories.")
572
+ @click.option("--max-live", default=5, show_default=True, help="Max hypotheses to validate live, by severity rank.")
573
+ @click.option("--runs-per", default=3, show_default=True, help="Runs per hypothesis for confirmation.")
574
+ @click.option("--iterations", default=2, show_default=True, help="Hypothesis refinement iterations.")
575
+ @click.option("--dry-run", is_flag=True, help="Generate and rank hypotheses only; no live runs.")
576
+ @click.option("--no-ranker", is_flag=True, help="Skip LLM ranker; use structural pre-pass only (ablation).")
577
+ @click.option("--no-memory", is_flag=True, help="Disable cross-workflow memory seeding (ablation).")
578
+ @click.option("--monolithic", is_flag=True, help="Use single hypothesis prompt instead of per-category (ablation).")
579
+ @click.option("--output", "output_dir", default="reports/scanner", show_default=True, help="Directory for reports.")
580
+ @click.option("--hypothesis-model", default="claude-sonnet-4-6", show_default=True, help="LLM for hypothesis generation.")
581
+ @click.option("--ranker-model", default="claude-sonnet-4-6", show_default=True, help="LLM for plausibility ranking.")
582
+ @click.option(
583
+ "--judge-model",
584
+ default="gemini-3.1-pro-preview",
585
+ show_default=True,
586
+ help="LLM judge for semantic success evaluation.",
587
+ )
588
+ @click.option(
589
+ "--repo-prefix",
590
+ default=lambda: os.environ.get("GITHUB_REPO_PREFIX", "benchmark-scan"),
591
+ help="GitHub repository prefix for live runs.",
592
+ )
593
+ @click.option("--cleanup/--no-cleanup", default=True, help="Delete repository after each live run.")
594
+ @click.option("--baselines/--no-baselines", default=True, help="Run zizmor and actionlint baselines.")
595
+ @click.option("--reseed", is_flag=True, help="Force reload the warm-start research corpus.")
596
+ @click.option(
597
+ "--no-diagnostics",
598
+ is_flag=True,
599
+ help="Disable diagnostic stage; collapse all failures to payload_ineffective (ablation).",
600
+ )
601
+ @click.option(
602
+ "--diagnostic-model",
603
+ default="claude-haiku-4-5",
604
+ show_default=True,
605
+ help="LLM for fast-path artifact inspection in the diagnostic stage.",
606
+ )
607
+ def scan(
608
+ workflow_id,
609
+ scan_all,
610
+ hypotheses,
611
+ max_live,
612
+ runs_per,
613
+ iterations,
614
+ dry_run,
615
+ no_ranker,
616
+ no_memory,
617
+ monolithic,
618
+ output_dir,
619
+ hypothesis_model,
620
+ ranker_model,
621
+ judge_model,
622
+ repo_prefix,
623
+ cleanup,
624
+ baselines,
625
+ reseed,
626
+ no_diagnostics,
627
+ diagnostic_model,
628
+ ):
629
+ """Autonomously scan a workflow for prompt injection vulnerabilities."""
630
+ import time as _time
631
+
632
+ from .scanner import hypothesis_generator, llm_ranker, prompt_extractor, report_generator
633
+ from .scanner.baselines import actionlint_runner, zizmor_runner
634
+ from .scanner.live_validator import validate
635
+ from .scanner.memory import CrossWorkflowMemory
636
+ from .scanner.types import ScanCost, roll_up_usage
637
+ from .utils.llm import track_usage
638
+
639
+ workflows_dir = dataset_dir("workflows")
640
+
641
+ if scan_all:
642
+ workflow_ids = sorted(
643
+ [d for d in os.listdir(workflows_dir) if os.path.isdir(os.path.join(workflows_dir, d)) and not d.startswith("_")]
644
+ )
645
+ elif workflow_id:
646
+ workflow_ids = [workflow_id]
647
+ else:
648
+ click.echo(click.style("Error: provide --workflow or --all.", fg="red"))
649
+ return
650
+
651
+ memory = CrossWorkflowMemory() if not no_memory else CrossWorkflowMemory.__new__(CrossWorkflowMemory)
652
+ if no_memory:
653
+ memory._entries = []
654
+ memory.path = "/dev/null"
655
+ memory._save = lambda: None
656
+ else:
657
+ needs_warm_start = reseed or not memory._entries
658
+ if needs_warm_start:
659
+ click.echo(" Loading the warm-start research corpus...")
660
+ n = memory.warm_start(reseed=reseed)
661
+ click.echo(f" Warm-start: {n} confirmed entries loaded.")
662
+
663
+ all_summary: list[dict] = []
664
+
665
+ for wf_id in workflow_ids:
666
+ click.echo("\n" + click.style(f"=== Scanning: {wf_id} ===", bold=True))
667
+
668
+ meta_path = os.path.join(workflows_dir, wf_id, "metadata.json")
669
+ workflow_category = "code-review"
670
+ if os.path.exists(meta_path):
671
+ with open(meta_path) as f:
672
+ workflow_category = json.load(f).get("category", "code-review")
673
+
674
+ try:
675
+ context = prompt_extractor.extract(wf_id, workflows_dir)
676
+ except Exception as e:
677
+ click.echo(click.style(f" Failed to extract context: {e}", fg="red"))
678
+ continue
679
+
680
+ click.echo(f" Provider: {context.provider}")
681
+ click.echo(f" Trigger: {context.trigger_event}")
682
+
683
+ baseline_findings: list[dict] = []
684
+ if baselines:
685
+ wf_contents = os.path.join(workflows_dir, wf_id, "contents")
686
+ if os.path.isdir(wf_contents):
687
+ click.echo(" Running baselines...")
688
+ baseline_findings.extend(zizmor_runner.run(wf_contents))
689
+ baseline_findings.extend(actionlint_runner.run(wf_contents))
690
+ click.echo(f" Baseline findings: {len(baseline_findings)}")
691
+
692
+ wf_start = _time.monotonic()
693
+ with track_usage() as usage_log:
694
+ click.echo(f" Generating {hypotheses} hypotheses...")
695
+ raw_hypotheses = hypothesis_generator.generate(
696
+ context,
697
+ memory,
698
+ hypotheses_per_scan=hypotheses,
699
+ monolithic=monolithic,
700
+ model=hypothesis_model,
701
+ )
702
+ click.echo(f" Generated: {len(raw_hypotheses)}")
703
+
704
+ ranked, discarded = llm_ranker.rank(raw_hypotheses, context, skip_llm=no_ranker, model=ranker_model)
705
+ click.echo(f" After filtering: {len(ranked)} ranked, {len(discarded)} discarded")
706
+
707
+ results = validate(
708
+ hypotheses=ranked,
709
+ context=context,
710
+ workflow_id=wf_id,
711
+ workflow_category=workflow_category,
712
+ runs_per_hypothesis=runs_per,
713
+ max_hypotheses=max_live,
714
+ iterations=iterations,
715
+ repo_prefix=repo_prefix,
716
+ cleanup=cleanup,
717
+ dry_run=dry_run,
718
+ judge_model=judge_model,
719
+ enable_diagnostics=not no_diagnostics,
720
+ diagnostic_model=diagnostic_model,
721
+ )
722
+
723
+ scan_cost = ScanCost(
724
+ token_usage_by_model=roll_up_usage(usage_log),
725
+ total_billable_minutes=sum(r.billable_minutes for r in results),
726
+ total_wall_seconds=_time.monotonic() - wf_start,
727
+ )
728
+
729
+ confirmed = [r for r in results if r.status == "confirmed"]
730
+ click.echo(
731
+ click.style(
732
+ f" Result: {len(confirmed)}/{len(results)} confirmed "
733
+ f"(${scan_cost.total_usd:.4f}, {scan_cost.total_billable_minutes:.1f} billable min)",
734
+ fg="green" if confirmed else "yellow",
735
+ )
736
+ )
737
+
738
+ md_path, _ = report_generator.generate(
739
+ context, results, discarded, baseline_findings, output_dir, scan_cost=scan_cost
740
+ )
741
+ click.echo(f" Report: {md_path}")
742
+
743
+ all_summary.append(
744
+ {
745
+ "workflow": wf_id,
746
+ "confirmed": len(confirmed),
747
+ "validated": len(results),
748
+ "filtered": len(discarded),
749
+ "report_md": md_path,
750
+ }
751
+ )
752
+
753
+ if len(workflow_ids) > 1:
754
+ click.echo("\n" + click.style("=== Scan Summary ===", bold=True))
755
+ total_confirmed = sum(s["confirmed"] for s in all_summary)
756
+ total_validated = sum(s["validated"] for s in all_summary)
757
+ click.echo(f"Workflows scanned: {len(all_summary)}")
758
+ click.echo(f"Total confirmed: {total_confirmed} / {total_validated} validated")
759
+ for s in all_summary:
760
+ status_color = "green" if s["confirmed"] > 0 else "white"
761
+ click.echo(
762
+ click.style(f" {s['workflow']:<30}", bold=True)
763
+ + click.style(f"confirmed: {s['confirmed']}", fg=status_color)
764
+ )
765
+
766
+
767
+ if __name__ == "__main__":
768
+ cli()