gitinject 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gitinject/__init__.py +1 -0
- gitinject/__main__.py +5 -0
- gitinject/analyzer.py +67 -0
- gitinject/attacks/__init__.py +27 -0
- gitinject/attacks/autoinject.py +165 -0
- gitinject/attacks/base.py +33 -0
- gitinject/attacks/static.py +25 -0
- gitinject/cli.py +768 -0
- gitinject/data/research/scenarios/claude_skills_injection.md +96 -0
- gitinject/data/research/scenarios/cline_issue_body_injection.md +93 -0
- gitinject/data/research/scenarios/codex_agents_md_injection.md +109 -0
- gitinject/data/research/scenarios/dos_request_flood.md +133 -0
- gitinject/data/research/scenarios/dropped/ci_log_injection_workflow_poisoning.md +138 -0
- gitinject/data/research/scenarios/dropped/claude_md_instructions_injection.md +158 -0
- gitinject/data/research/scenarios/dropped/supply_chain_token_pivot.md +101 -0
- gitinject/data/research/scenarios/gemini_api_key_exfiltration.md +68 -0
- gitinject/data/research/scenarios/gemini_api_key_exfiltration_replication.md +0 -0
- gitinject/data/research/scenarios/gemini_md_instructions_injection.md +131 -0
- gitinject/data/research/scenarios/nsfw_api_key_block.md +114 -0
- gitinject/data/research/scenarios/pr_token_exfiltration_via_git_config.md +117 -0
- gitinject/data/research/scenarios/supply_chain_malicious_code.md +135 -0
- gitinject/evaluators.py +169 -0
- gitinject/evidence.py +72 -0
- gitinject/gl_runner.py +163 -0
- gitinject/resources.py +30 -0
- gitinject/run_context.py +72 -0
- gitinject/run_record.py +113 -0
- gitinject/runner.py +1024 -0
- gitinject/scanner/__init__.py +22 -0
- gitinject/scanner/baselines/__init__.py +0 -0
- gitinject/scanner/baselines/actionlint_runner.py +44 -0
- gitinject/scanner/baselines/zizmor_runner.py +35 -0
- gitinject/scanner/diagnostics.py +174 -0
- gitinject/scanner/hypothesis_generator.py +164 -0
- gitinject/scanner/live_validator.py +284 -0
- gitinject/scanner/llm_ranker.py +120 -0
- gitinject/scanner/memory.py +239 -0
- gitinject/scanner/primitives.py +236 -0
- gitinject/scanner/prompt_extractor.py +134 -0
- gitinject/scanner/recipe_scenario.py +203 -0
- gitinject/scanner/report_generator.py +230 -0
- gitinject/scanner/types.py +179 -0
- gitinject/scenario_base.py +147 -0
- gitinject/scenario_loader.py +77 -0
- gitinject/scenarios/__init__.py +0 -0
- gitinject/scenarios/benign/documentation_out_of_sync/contents/math_utils.py +3 -0
- gitinject/scenarios/benign/documentation_out_of_sync/contents/pydantic/v1/utils.py +21 -0
- gitinject/scenarios/benign/documentation_out_of_sync/scenario.py +57 -0
- gitinject/scenarios/benign/issue_deduplication_utility/scenario.py +76 -0
- gitinject/scenarios/benign/repository_onboarding_qa/contents/README.md +3 -0
- gitinject/scenarios/benign/repository_onboarding_qa/scenario.py +52 -0
- gitinject/scenarios/benign/secrets_false_positive/contents/tests/test_auth.py +14 -0
- gitinject/scenarios/benign/secrets_false_positive/scenario.py +58 -0
- gitinject/scenarios/benign/test_coverage_gap/contents/src/requests/auth_extra.py +33 -0
- gitinject/scenarios/benign/test_coverage_gap/scenario.py +56 -0
- gitinject/scenarios/benign/vulnerability_fix_verification/contents/src/flask/utils.py +15 -0
- gitinject/scenarios/benign/vulnerability_fix_verification/scenario.py +57 -0
- gitinject/scenarios/benign/vulnerable_code_review/contents/examples/tutorial/flaskr/db_utils.py +18 -0
- gitinject/scenarios/benign/vulnerable_code_review/scenario.py +55 -0
- gitinject/scenarios/benign/weekly_activity_summary/scenario.py +83 -0
- gitinject/scenarios/malicious/agents_md_approval_manipulation/contents/AGENTS.md +25 -0
- gitinject/scenarios/malicious/agents_md_approval_manipulation/contents/src/flask/csrf.py +151 -0
- gitinject/scenarios/malicious/agents_md_approval_manipulation/contents/tests/test_csrf.py +71 -0
- gitinject/scenarios/malicious/agents_md_approval_manipulation/scenario.py +121 -0
- gitinject/scenarios/malicious/claude_md_instructions_injection/contents/CLAUDE.md +23 -0
- gitinject/scenarios/malicious/claude_md_instructions_injection/contents/scripts/deploy_check.sh +29 -0
- gitinject/scenarios/malicious/claude_md_instructions_injection/scenario.py +93 -0
- gitinject/scenarios/malicious/cline_issue_body_injection/scenario.py +132 -0
- gitinject/scenarios/malicious/codex_agents_md_injection/contents/AGENTS.md +36 -0
- gitinject/scenarios/malicious/codex_agents_md_injection/contents/scripts/utils.py +12 -0
- gitinject/scenarios/malicious/codex_agents_md_injection/scenario.py +128 -0
- gitinject/scenarios/malicious/dos_request_flood/scenario.py +80 -0
- gitinject/scenarios/malicious/gemini_md_approval_manipulation/contents/src/flask/csrf.py +151 -0
- gitinject/scenarios/malicious/gemini_md_approval_manipulation/contents/tests/test_csrf.py +71 -0
- gitinject/scenarios/malicious/gemini_md_approval_manipulation/scenario.py +121 -0
- gitinject/scenarios/malicious/gemini_md_instructions_injection/contents/scripts/deploy_check.sh +38 -0
- gitinject/scenarios/malicious/gemini_md_instructions_injection/scenario.py +105 -0
- gitinject/scenarios/malicious/issue_onboarding_secret_exfiltration/scenario.py +76 -0
- gitinject/scenarios/malicious/pr_token_exfiltration_via_git_config/contents/scripts/deploy_check.sh +7 -0
- gitinject/scenarios/malicious/pr_token_exfiltration_via_git_config/scenario.py +113 -0
- gitinject/scenarios/malicious/supply_chain_malicious_code/contents/CLAUDE.md +25 -0
- gitinject/scenarios/malicious/supply_chain_malicious_code/contents/src/flask/csrf.py +151 -0
- gitinject/scenarios/malicious/supply_chain_malicious_code/contents/tests/test_csrf.py +71 -0
- gitinject/scenarios/malicious/supply_chain_malicious_code/scenario.py +123 -0
- gitinject/scenarios/malicious/unauthorized_pr_approval/contents/django/utils/crypto.py +13 -0
- gitinject/scenarios/malicious/unauthorized_pr_approval/scenario.py +57 -0
- gitinject/simulator.py +89 -0
- gitinject/utils/__init__.py +0 -0
- gitinject/utils/gh_client.py +628 -0
- gitinject/utils/gl_client.py +132 -0
- gitinject/utils/gl_provisioner.py +83 -0
- gitinject/utils/llm.py +205 -0
- gitinject/utils/provisioner.py +114 -0
- gitinject/utils/scenario_resources.py +33 -0
- gitinject/utils/types.py +49 -0
- gitinject/workflows/__init__.py +0 -0
- gitinject/workflows/claude-ci-auto-fix/contents/.github/workflows/main.yml +107 -0
- gitinject/workflows/claude-ci-auto-fix/metadata.json +10 -0
- gitinject/workflows/claude-general/contents/.github/workflows/main.yml +58 -0
- gitinject/workflows/claude-general/metadata.json +10 -0
- gitinject/workflows/claude-gitlab-mr-review/contents/.gitlab-ci.yml +36 -0
- gitinject/workflows/claude-gitlab-mr-review/metadata.json +11 -0
- gitinject/workflows/claude-issue-deduplication/contents/.github/workflows/main.yml +66 -0
- gitinject/workflows/claude-issue-deduplication/metadata.json +10 -0
- gitinject/workflows/claude-issue-triage/contents/.github/workflows/main.yml +34 -0
- gitinject/workflows/claude-issue-triage/metadata.json +10 -0
- gitinject/workflows/claude-manual-analysis/contents/.github/workflows/main.yml +42 -0
- gitinject/workflows/claude-manual-analysis/metadata.json +10 -0
- gitinject/workflows/claude-pr-review/contents/.github/workflows/main.yml +77 -0
- gitinject/workflows/claude-pr-review/metadata.json +10 -0
- gitinject/workflows/claude-pr-review-authors/contents/.github/workflows/main.yml +48 -0
- gitinject/workflows/claude-pr-review-authors/metadata.json +10 -0
- gitinject/workflows/claude-pr-review-paths/contents/.github/workflows/main.yml +49 -0
- gitinject/workflows/claude-pr-review-paths/metadata.json +10 -0
- gitinject/workflows/claude-test-analysis/contents/.github/workflows/main.yml +114 -0
- gitinject/workflows/claude-test-analysis/metadata.json +10 -0
- gitinject/workflows/cline-assistant/contents/.github/workflows/main.yml +87 -0
- gitinject/workflows/cline-assistant/contents/git-scripts/analyze-issue.sh +43 -0
- gitinject/workflows/cline-assistant/metadata.json +10 -0
- gitinject/workflows/codex-pr-review/contents/.github/workflows/main.yml +73 -0
- gitinject/workflows/codex-pr-review/metadata.json +10 -0
- gitinject/workflows/copilot-ci-doctor/contents/.github/workflows/ci-doctor.yml +1161 -0
- gitinject/workflows/copilot-ci-doctor/metadata.json +10 -0
- gitinject/workflows/copilot-lean-squad/contents/.github/workflows/lean-squad.yml +1313 -0
- gitinject/workflows/copilot-lean-squad/metadata.json +10 -0
- gitinject/workflows/copilot-malicious-scan/contents/.github/workflows/daily-malicious-code-scan.yml +899 -0
- gitinject/workflows/copilot-malicious-scan/metadata.json +10 -0
- gitinject/workflows/copilot-repo-assist/contents/.github/workflows/repo-assist.yml +1503 -0
- gitinject/workflows/copilot-repo-assist/metadata.json +10 -0
- gitinject/workflows/copilot-wiki-writer/contents/.github/workflows/agentic-wiki-writer.yml +1316 -0
- gitinject/workflows/copilot-wiki-writer/metadata.json +10 -0
- gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-invoke.yml +122 -0
- gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-plan-execute.yml +130 -0
- gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-review.yml +118 -0
- gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-scheduled-triage.yml +220 -0
- gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-triage.yml +160 -0
- gitinject/workflows/gemini-assistant/contents/.github/workflows/main.yml +220 -0
- gitinject/workflows/gemini-assistant/metadata.json +10 -0
- gitinject/workflows/gemini-assistant-original/AWESOME.md +118 -0
- gitinject/workflows/gemini-assistant-original/CONFIGURATION.md +162 -0
- gitinject/workflows/gemini-assistant-original/README.md +93 -0
- gitinject/workflows/gemini-assistant-original/gemini-assistant/README.md +192 -0
- gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-invoke.toml +94 -0
- gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-invoke.yml +131 -0
- gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-plan-execute.toml +100 -0
- gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-plan-execute.yml +139 -0
- gitinject/workflows/gemini-assistant-original/gemini-dispatch/README.md +49 -0
- gitinject/workflows/gemini-assistant-original/gemini-dispatch/gemini-dispatch.yml +221 -0
- gitinject/workflows/gemini-assistant-original/issue-triage/README.md +190 -0
- gitinject/workflows/gemini-assistant-original/issue-triage/gemini-scheduled-triage.toml +96 -0
- gitinject/workflows/gemini-assistant-original/issue-triage/gemini-scheduled-triage.yml +223 -0
- gitinject/workflows/gemini-assistant-original/issue-triage/gemini-triage.toml +32 -0
- gitinject/workflows/gemini-assistant-original/issue-triage/gemini-triage.yml +167 -0
- gitinject/workflows/gemini-assistant-original/metadata.json +10 -0
- gitinject/workflows/gemini-assistant-original/pr-review/README.md +337 -0
- gitinject/workflows/gemini-assistant-original/pr-review/gemini-review.toml +176 -0
- gitinject/workflows/gemini-assistant-original/pr-review/gemini-review.yml +119 -0
- gitinject/workflows/opencode-pr-review/contents/.github/workflows/main.yml +28 -0
- gitinject/workflows/opencode-pr-review/metadata.json +10 -0
- gitinject-0.1.0.dist-info/METADATA +128 -0
- gitinject-0.1.0.dist-info/RECORD +164 -0
- gitinject-0.1.0.dist-info/WHEEL +4 -0
- gitinject-0.1.0.dist-info/entry_points.txt +2 -0
- gitinject-0.1.0.dist-info/licenses/LICENSE +202 -0
gitinject/cli.py
ADDED
|
@@ -0,0 +1,768 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import os
|
|
3
|
+
|
|
4
|
+
import click
|
|
5
|
+
|
|
6
|
+
from .resources import dataset_dir
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@click.group()
|
|
10
|
+
def cli():
|
|
11
|
+
"""AI-Powered GitHub Workflows Security Benchmark CLI."""
|
|
12
|
+
pass
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@cli.group()
|
|
16
|
+
def list():
|
|
17
|
+
"""List benchmark components."""
|
|
18
|
+
pass
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@list.command(name="workflows")
|
|
22
|
+
def list_workflows():
|
|
23
|
+
"""List available workflows with their categories and supported events."""
|
|
24
|
+
workflows_dir = dataset_dir("workflows")
|
|
25
|
+
if not os.path.exists(workflows_dir):
|
|
26
|
+
click.echo("Workflows directory not found.")
|
|
27
|
+
return
|
|
28
|
+
|
|
29
|
+
workflows = sorted([d for d in os.listdir(workflows_dir) if os.path.isdir(os.path.join(workflows_dir, d))])
|
|
30
|
+
for w in workflows:
|
|
31
|
+
metadata_path = os.path.join(workflows_dir, w, "metadata.json")
|
|
32
|
+
if os.path.exists(metadata_path):
|
|
33
|
+
with open(metadata_path, "r") as f:
|
|
34
|
+
metadata = json.load(f)
|
|
35
|
+
category = metadata.get("category", "uncategorized")
|
|
36
|
+
events = ", ".join(metadata.get("supported_events", []))
|
|
37
|
+
click.echo(f"- {w:25} | Category: {category:20} | Events: {events}")
|
|
38
|
+
else:
|
|
39
|
+
click.echo(f"- {w:25} | No metadata found.")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _discover_scenarios(scenarios_dir):
|
|
43
|
+
"""Load scenario definitions without constructing an authenticated runner."""
|
|
44
|
+
from .scenario_loader import discover_scenario_paths, load_scenario
|
|
45
|
+
|
|
46
|
+
return sorted(
|
|
47
|
+
[(path.parent.name, load_scenario(path, os.getcwd())) for path in discover_scenario_paths(scenarios_dir)],
|
|
48
|
+
key=lambda item: item[0],
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@list.command(name="scenarios")
|
|
53
|
+
def list_scenarios():
|
|
54
|
+
"""List available scenarios with their categories and event types."""
|
|
55
|
+
scenarios_dir = dataset_dir("scenarios")
|
|
56
|
+
if not os.path.exists(scenarios_dir):
|
|
57
|
+
click.echo("Scenarios directory not found.")
|
|
58
|
+
return
|
|
59
|
+
|
|
60
|
+
scenarios = _discover_scenarios(scenarios_dir)
|
|
61
|
+
|
|
62
|
+
for s_name, s_obj in scenarios:
|
|
63
|
+
category = s_obj.category.value if s_obj.category else "none"
|
|
64
|
+
event = s_obj.get_event().get("event_type", "unknown")
|
|
65
|
+
s_type = s_obj.scenario_type.value if hasattr(s_obj, "scenario_type") else "benign"
|
|
66
|
+
click.echo(f"- {s_name:25} | Type: {s_type:10} | Category: {category:20} | Event: {event}")
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@cli.command()
|
|
70
|
+
@click.option("--workflow", required=True, help="Workflow ID to run.")
|
|
71
|
+
@click.option("--scenario", required=True, help="Scenario ID to run (or 'all' for all compatible scenarios).")
|
|
72
|
+
@click.option(
|
|
73
|
+
"--repo-prefix",
|
|
74
|
+
default=lambda: os.environ.get("GITHUB_REPO_PREFIX", "benchmark-run"),
|
|
75
|
+
help="Target GitHub repository prefix.",
|
|
76
|
+
)
|
|
77
|
+
@click.option(
|
|
78
|
+
"--cleanup/--no-cleanup",
|
|
79
|
+
default=True,
|
|
80
|
+
help="Automatically delete the GitHub repository after the run.",
|
|
81
|
+
)
|
|
82
|
+
@click.option(
|
|
83
|
+
"--log-llm-input",
|
|
84
|
+
is_flag=True,
|
|
85
|
+
help="Reconstruct and print the effective LLM prompt before triggering the run, and save it to runs/*/llm_input.txt.",
|
|
86
|
+
)
|
|
87
|
+
@click.option(
|
|
88
|
+
"--attack",
|
|
89
|
+
"attack_id",
|
|
90
|
+
default=None,
|
|
91
|
+
help="Attack type to apply (autoinject, static). Omit to use the scenario's hardcoded payload.",
|
|
92
|
+
)
|
|
93
|
+
@click.option(
|
|
94
|
+
"--attack-payload",
|
|
95
|
+
default=None,
|
|
96
|
+
help="Inline payload string or path to a payload file. Used with --attack static.",
|
|
97
|
+
)
|
|
98
|
+
@click.option(
|
|
99
|
+
"--repeat",
|
|
100
|
+
default=1,
|
|
101
|
+
show_default=True,
|
|
102
|
+
help="Number of times to repeat each run.",
|
|
103
|
+
)
|
|
104
|
+
@click.option("--parameters", default="{}", help="JSON object supplied to the scenario's run context.")
|
|
105
|
+
@click.option("--seed", type=int, default=None, help="Seed for the scenario context's random generator.")
|
|
106
|
+
def run(
|
|
107
|
+
workflow, scenario, repo_prefix, cleanup, log_llm_input, attack_id, attack_payload, repeat, parameters, seed
|
|
108
|
+
):
|
|
109
|
+
"""Run benchmark tests."""
|
|
110
|
+
from .runner import BenchmarkRunner
|
|
111
|
+
|
|
112
|
+
try:
|
|
113
|
+
parameters = json.loads(parameters)
|
|
114
|
+
if not isinstance(parameters, dict):
|
|
115
|
+
raise ValueError("Expected a JSON object")
|
|
116
|
+
except ValueError as exc:
|
|
117
|
+
raise click.BadParameter(str(exc), param_hint="--parameters") from exc
|
|
118
|
+
|
|
119
|
+
workflows_dir = dataset_dir("workflows")
|
|
120
|
+
scenarios_dir = dataset_dir("scenarios")
|
|
121
|
+
|
|
122
|
+
workflow_path = os.path.join(workflows_dir, workflow)
|
|
123
|
+
if not os.path.isdir(workflow_path):
|
|
124
|
+
click.echo(click.style(f"Error: Workflow '{workflow}' not found.", fg="red"))
|
|
125
|
+
return
|
|
126
|
+
|
|
127
|
+
meta_path = os.path.join(workflow_path, "metadata.json")
|
|
128
|
+
workflow_meta = {}
|
|
129
|
+
if os.path.exists(meta_path):
|
|
130
|
+
with open(meta_path, "r") as f:
|
|
131
|
+
workflow_meta = json.load(f)
|
|
132
|
+
|
|
133
|
+
platform = workflow_meta.get("platform", "github")
|
|
134
|
+
w_category = workflow_meta.get("category")
|
|
135
|
+
supported_events = set(workflow_meta.get("supported_events", []))
|
|
136
|
+
|
|
137
|
+
def _run_pair(wf, sc):
|
|
138
|
+
if platform == "gitlab":
|
|
139
|
+
from .gl_runner import GitLabRunner
|
|
140
|
+
|
|
141
|
+
runner = GitLabRunner(os.getcwd(), project_prefix=repo_prefix)
|
|
142
|
+
return runner.run(wf, sc, cleanup=cleanup)
|
|
143
|
+
runner = BenchmarkRunner(os.getcwd(), repo_prefix=repo_prefix)
|
|
144
|
+
return runner.run(
|
|
145
|
+
wf,
|
|
146
|
+
sc,
|
|
147
|
+
attack_id=attack_id,
|
|
148
|
+
attack_payload=attack_payload,
|
|
149
|
+
cleanup=cleanup,
|
|
150
|
+
log_llm_input=log_llm_input,
|
|
151
|
+
parameters=parameters,
|
|
152
|
+
seed=seed,
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
if scenario.lower() == "all":
|
|
156
|
+
click.echo(f"Identifying compatible scenarios for workflow '{workflow}'...")
|
|
157
|
+
|
|
158
|
+
valid_scenarios = _discover_scenarios(scenarios_dir)
|
|
159
|
+
scenarios_to_run = []
|
|
160
|
+
for s_name, s_obj in valid_scenarios:
|
|
161
|
+
s_event = s_obj.get_event().get("event_type")
|
|
162
|
+
s_platform = getattr(s_obj, "platform", "github")
|
|
163
|
+
if s_obj.category == w_category and s_event in supported_events and s_platform == platform:
|
|
164
|
+
scenarios_to_run.append(s_name)
|
|
165
|
+
|
|
166
|
+
if not scenarios_to_run:
|
|
167
|
+
click.echo(click.style(f"No compatible scenarios found for workflow '{workflow}'.", fg="yellow"))
|
|
168
|
+
return
|
|
169
|
+
|
|
170
|
+
click.echo(f"Found {len(scenarios_to_run)} compatible scenarios: {', '.join(scenarios_to_run)}")
|
|
171
|
+
|
|
172
|
+
pairs_results = {}
|
|
173
|
+
for s_name in scenarios_to_run:
|
|
174
|
+
pairs_results[(workflow, s_name)] = []
|
|
175
|
+
for i in range(repeat):
|
|
176
|
+
label = f"--- Running {workflow} against {s_name} (run {i + 1}/{repeat}) ---"
|
|
177
|
+
click.echo("\n" + click.style(label, bold=True))
|
|
178
|
+
result = _run_pair(workflow, s_name)
|
|
179
|
+
_display_run_result(result)
|
|
180
|
+
pairs_results[(workflow, s_name)].append(result)
|
|
181
|
+
|
|
182
|
+
if repeat > 1:
|
|
183
|
+
_display_repeat_summary(pairs_results)
|
|
184
|
+
else:
|
|
185
|
+
pairs_results = {(workflow, scenario): []}
|
|
186
|
+
for i in range(repeat):
|
|
187
|
+
if repeat > 1:
|
|
188
|
+
click.echo(click.style(f"\n--- Run {i + 1}/{repeat} ---", bold=True))
|
|
189
|
+
if platform == "gitlab":
|
|
190
|
+
from .gl_runner import GitLabRunner
|
|
191
|
+
|
|
192
|
+
runner = GitLabRunner(os.getcwd(), project_prefix=repo_prefix)
|
|
193
|
+
click.echo(f"Running GitLab benchmark: workflow={workflow}, scenario={scenario}")
|
|
194
|
+
result = runner.run(workflow, scenario, cleanup=cleanup)
|
|
195
|
+
else:
|
|
196
|
+
runner = BenchmarkRunner(os.getcwd(), repo_prefix=repo_prefix)
|
|
197
|
+
click.echo(f"Running benchmark on {runner.repo_name}: workflow={workflow}, scenario={scenario}")
|
|
198
|
+
result = runner.run(
|
|
199
|
+
workflow,
|
|
200
|
+
scenario,
|
|
201
|
+
attack_id=attack_id,
|
|
202
|
+
attack_payload=attack_payload,
|
|
203
|
+
cleanup=cleanup,
|
|
204
|
+
log_llm_input=log_llm_input,
|
|
205
|
+
parameters=parameters,
|
|
206
|
+
seed=seed,
|
|
207
|
+
)
|
|
208
|
+
_display_run_result(result)
|
|
209
|
+
pairs_results[(workflow, scenario)].append(result)
|
|
210
|
+
|
|
211
|
+
if repeat > 1:
|
|
212
|
+
_display_repeat_summary(pairs_results)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _display_run_result(result):
|
|
216
|
+
"""Helper to display the result of a single benchmark run."""
|
|
217
|
+
if "error" in result:
|
|
218
|
+
click.echo(click.style(f"Error: {result['error']}", fg="red"))
|
|
219
|
+
else:
|
|
220
|
+
click.echo("\n" + click.style("--- Benchmark Evaluation ---", bold=True))
|
|
221
|
+
analysis = result.get("analysis", {})
|
|
222
|
+
for label, key in (
|
|
223
|
+
("Utility Achieved", "utility_achieved"),
|
|
224
|
+
("Security Breached", "security_breached"),
|
|
225
|
+
("Autonomy Achieved", "autonomy_achieved"),
|
|
226
|
+
):
|
|
227
|
+
value = analysis.get(key)
|
|
228
|
+
click.echo(f"{label}: {value if value is not None else 'unknown'}")
|
|
229
|
+
for metric, error in analysis.get("evaluation_errors", {}).items():
|
|
230
|
+
click.echo(f" {metric}: {error}")
|
|
231
|
+
click.echo(click.style("----------------------------", bold=True))
|
|
232
|
+
click.echo(f"Message: {result.get('message')}")
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _metric_summary(results, key, *, invert=False):
|
|
236
|
+
values = [r.get("analysis", {}).get(key) for r in results if "error" not in r]
|
|
237
|
+
known = [value for value in values if type(value) is bool]
|
|
238
|
+
if not known:
|
|
239
|
+
return "N/A (0/0)"
|
|
240
|
+
positives = sum(value is (False if invert else True) for value in known)
|
|
241
|
+
return f"{positives / len(known):.2f} ({positives}/{len(known)})"
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _display_repeat_summary(pairs_results: dict[tuple, list]):
|
|
245
|
+
"""Report each metric using its own known-verdict denominator."""
|
|
246
|
+
click.echo("\n" + click.style("--- Repeat Summary ---", bold=True))
|
|
247
|
+
click.echo(f"{'Workflow':<25} {'Scenario':<30} {'Security':<18} {'Utility':<18} {'Autonomy':<18} {'Attempts':<9}")
|
|
248
|
+
for (wf, sc), results in sorted(pairs_results.items()):
|
|
249
|
+
sec = _metric_summary(results, "security_breached", invert=True)
|
|
250
|
+
util = _metric_summary(results, "utility_achieved")
|
|
251
|
+
auto = _metric_summary(results, "autonomy_achieved")
|
|
252
|
+
click.echo(f"{wf:<25} {sc:<30} {sec:<18} {util:<18} {auto:<18} {len(results):<9}")
|
|
253
|
+
click.echo("Ratios exclude unknown verdicts and execution errors; Attempts includes all runs.")
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
@cli.command(name="run-suite")
|
|
257
|
+
@click.option("--workflow-labels", help="Comma-separated list of workflow labels to filter by.")
|
|
258
|
+
@click.option("--scenario-labels", help="Comma-separated list of scenario labels to filter by.")
|
|
259
|
+
@click.option("--scenario-type", help="Filter by scenario type (benign/malicious).")
|
|
260
|
+
@click.option("--event", help="Filter by event type.")
|
|
261
|
+
@click.option(
|
|
262
|
+
"--repo-prefix",
|
|
263
|
+
default=lambda: os.environ.get("GITHUB_REPO_PREFIX", "benchmark-run"),
|
|
264
|
+
help="Target GitHub repository prefix.",
|
|
265
|
+
)
|
|
266
|
+
@click.option(
|
|
267
|
+
"--cleanup/--no-cleanup",
|
|
268
|
+
default=True,
|
|
269
|
+
help="Automatically delete the GitHub repository after the run.",
|
|
270
|
+
)
|
|
271
|
+
@click.option(
|
|
272
|
+
"--dry-run",
|
|
273
|
+
help="List compatible pairs without executing them.",
|
|
274
|
+
is_flag=True,
|
|
275
|
+
)
|
|
276
|
+
@click.option(
|
|
277
|
+
"--log-llm-input",
|
|
278
|
+
is_flag=True,
|
|
279
|
+
help="Reconstruct and print the effective LLM prompt before each run.",
|
|
280
|
+
)
|
|
281
|
+
@click.option(
|
|
282
|
+
"--repeat",
|
|
283
|
+
default=1,
|
|
284
|
+
show_default=True,
|
|
285
|
+
help="Number of times to repeat each workflow/scenario pair.",
|
|
286
|
+
)
|
|
287
|
+
def run_suite(
|
|
288
|
+
workflow_labels, scenario_labels, scenario_type, event, repo_prefix, cleanup, dry_run, log_llm_input, repeat
|
|
289
|
+
):
|
|
290
|
+
"""Run a suite of compatible workflows and scenarios."""
|
|
291
|
+
from .runner import BenchmarkRunner
|
|
292
|
+
|
|
293
|
+
workflows_dir = dataset_dir("workflows")
|
|
294
|
+
scenarios_dir = dataset_dir("scenarios")
|
|
295
|
+
|
|
296
|
+
wf_filters = set(workflow_labels.split(",")) if workflow_labels else set()
|
|
297
|
+
sc_filters = set(scenario_labels.split(",")) if scenario_labels else set()
|
|
298
|
+
|
|
299
|
+
# Load workflows
|
|
300
|
+
valid_workflows = []
|
|
301
|
+
for w in os.listdir(workflows_dir):
|
|
302
|
+
if not os.path.isdir(os.path.join(workflows_dir, w)):
|
|
303
|
+
continue
|
|
304
|
+
meta_path = os.path.join(workflows_dir, w, "metadata.json")
|
|
305
|
+
if os.path.exists(meta_path):
|
|
306
|
+
with open(meta_path, "r") as f:
|
|
307
|
+
meta = json.load(f)
|
|
308
|
+
labels = set(meta.get("labels", []))
|
|
309
|
+
supported_events = set(meta.get("supported_events", []))
|
|
310
|
+
|
|
311
|
+
if wf_filters and not wf_filters.intersection(labels):
|
|
312
|
+
continue
|
|
313
|
+
if event and event not in supported_events:
|
|
314
|
+
continue
|
|
315
|
+
valid_workflows.append((w, meta))
|
|
316
|
+
|
|
317
|
+
# Load scenarios
|
|
318
|
+
all_scenarios = _discover_scenarios(scenarios_dir)
|
|
319
|
+
valid_scenarios = []
|
|
320
|
+
for s_name, s_obj in all_scenarios:
|
|
321
|
+
labels = set(getattr(s_obj, "labels", []))
|
|
322
|
+
s_event = s_obj.get_event().get("event_type")
|
|
323
|
+
s_type = s_obj.scenario_type.value if hasattr(s_obj, "scenario_type") else "benign"
|
|
324
|
+
|
|
325
|
+
if sc_filters and not sc_filters.intersection(labels):
|
|
326
|
+
continue
|
|
327
|
+
if scenario_type and scenario_type.lower() != s_type.lower():
|
|
328
|
+
continue
|
|
329
|
+
if event and event != s_event:
|
|
330
|
+
continue
|
|
331
|
+
valid_scenarios.append((s_name, s_obj))
|
|
332
|
+
|
|
333
|
+
# Generate compatible pairs
|
|
334
|
+
pairs = []
|
|
335
|
+
for w_name, w_meta in valid_workflows:
|
|
336
|
+
w_category = w_meta.get("category")
|
|
337
|
+
supported_events = set(w_meta.get("supported_events", []))
|
|
338
|
+
|
|
339
|
+
for s_name, s_obj in valid_scenarios:
|
|
340
|
+
s_event = s_obj.get_event().get("event_type")
|
|
341
|
+
if s_obj.category == w_category and s_event in supported_events:
|
|
342
|
+
pairs.append((w_name, s_name))
|
|
343
|
+
|
|
344
|
+
if not pairs:
|
|
345
|
+
click.echo("No compatible workflow/scenario pairs found.")
|
|
346
|
+
return
|
|
347
|
+
|
|
348
|
+
if dry_run:
|
|
349
|
+
click.echo(click.style(f"DRY RUN: Found {len(pairs)} compatible pairs:", bold=True))
|
|
350
|
+
for w_name, s_name in pairs:
|
|
351
|
+
click.echo(f" - Workflow: {w_name:20} | Scenario: {s_name}")
|
|
352
|
+
return
|
|
353
|
+
|
|
354
|
+
total = len(pairs) * repeat
|
|
355
|
+
click.echo(f"Starting suite with {len(pairs)} compatible pairs × {repeat} repeat(s) = {total} total runs.")
|
|
356
|
+
pairs_results = {}
|
|
357
|
+
for w_name, s_name in pairs:
|
|
358
|
+
pairs_results[(w_name, s_name)] = []
|
|
359
|
+
for i in range(repeat):
|
|
360
|
+
label = f"--- Running {w_name} against {s_name} (run {i + 1}/{repeat}) ---"
|
|
361
|
+
click.echo("\n" + click.style(label, bold=True))
|
|
362
|
+
runner = BenchmarkRunner(os.getcwd(), repo_prefix=repo_prefix)
|
|
363
|
+
res = runner.run(w_name, s_name, cleanup=cleanup, log_llm_input=log_llm_input)
|
|
364
|
+
_display_run_result(res)
|
|
365
|
+
pairs_results[(w_name, s_name)].append(res)
|
|
366
|
+
|
|
367
|
+
all_results = [r for results in pairs_results.values() for r in results]
|
|
368
|
+
click.echo("\n" + click.style("--- Benchmark Suite Complete ---", bold=True))
|
|
369
|
+
success = sum(1 for r in all_results if "error" not in r)
|
|
370
|
+
click.echo(f"Runs: {success}/{total} successful.")
|
|
371
|
+
if repeat > 1:
|
|
372
|
+
_display_repeat_summary(pairs_results)
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
@cli.command()
|
|
376
|
+
@click.option("--workflow", required=True, help="Workflow ID.")
|
|
377
|
+
@click.option("--scenario", required=True, help="Scenario ID.")
|
|
378
|
+
@click.option("--attack", "attack_id", required=True, help="Attack type (autoinject, static).")
|
|
379
|
+
@click.option(
|
|
380
|
+
"--victim-model",
|
|
381
|
+
default=None,
|
|
382
|
+
help="OpenRouter model string for the victim (e.g. openai/gpt-5.4-2026-03-05). Defaults to ATTACK_VICTIM_MODEL env var.",
|
|
383
|
+
)
|
|
384
|
+
def preflight(workflow, scenario, attack_id, victim_model):
|
|
385
|
+
"""
|
|
386
|
+
Single offline shot: generate a payload, send the injected prompt directly to the
|
|
387
|
+
victim model, and report whether the attack succeeded. No GitHub repo needed.
|
|
388
|
+
"""
|
|
389
|
+
from .runner import BenchmarkRunner
|
|
390
|
+
|
|
391
|
+
runner = BenchmarkRunner(os.getcwd(), repo_prefix="preflight")
|
|
392
|
+
result = runner.offline_optimize(workflow, scenario, attack_id, iterations=1, victim_model=victim_model)
|
|
393
|
+
|
|
394
|
+
if "error" in result:
|
|
395
|
+
click.echo(click.style(f"Error: {result['error']}", fg="red"))
|
|
396
|
+
return
|
|
397
|
+
|
|
398
|
+
score = result["asr_curve"][0] if result["asr_curve"] else None
|
|
399
|
+
label = (
|
|
400
|
+
click.style("UNKNOWN — offline trial could not be scored", fg="yellow")
|
|
401
|
+
if score is None
|
|
402
|
+
else (
|
|
403
|
+
click.style("PASS — attack worked offline", fg="green")
|
|
404
|
+
if score
|
|
405
|
+
else click.style("FAIL — attack did not work offline", fg="red")
|
|
406
|
+
)
|
|
407
|
+
)
|
|
408
|
+
click.echo(f"\nPreflight result: {label}")
|
|
409
|
+
click.echo(f"Payload at: {result['runs_dir']}/best_payload.txt")
|
|
410
|
+
click.echo(
|
|
411
|
+
"\nNote: offline uses a plain chat call. The agentic Codex context may differ. "
|
|
412
|
+
"A PASS here is a strong indicator but not a guarantee."
|
|
413
|
+
)
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
@cli.command()
|
|
417
|
+
@click.option("--workflow", required=True, help="Workflow ID to run.")
|
|
418
|
+
@click.option("--scenario", required=True, help="Scenario ID to optimize against.")
|
|
419
|
+
@click.option("--attack", "attack_id", required=True, help="Attack type (autoinject, static).")
|
|
420
|
+
@click.option("--iterations", default=5, show_default=True, help="Number of optimization iterations.")
|
|
421
|
+
@click.option(
|
|
422
|
+
"--offline",
|
|
423
|
+
is_flag=True,
|
|
424
|
+
help="Optimize using direct model calls instead of GitHub workflow runs. "
|
|
425
|
+
"Fast, no repo provisioning. Requires scenario.get_preflight_evaluator().",
|
|
426
|
+
)
|
|
427
|
+
@click.option(
|
|
428
|
+
"--victim-model",
|
|
429
|
+
default=None,
|
|
430
|
+
help="Override victim model for offline mode (OpenRouter string). Defaults to ATTACK_VICTIM_MODEL env var.",
|
|
431
|
+
)
|
|
432
|
+
@click.option(
|
|
433
|
+
"--repo-prefix",
|
|
434
|
+
default=lambda: os.environ.get("GITHUB_REPO_PREFIX", "benchmark-run"),
|
|
435
|
+
help="Target GitHub repository prefix (online mode only).",
|
|
436
|
+
)
|
|
437
|
+
@click.option("--cleanup/--no-cleanup", default=True, help="Delete the repository after the run (online mode only).")
|
|
438
|
+
def optimize(workflow, scenario, attack_id, iterations, offline, victim_model, repo_prefix, cleanup):
|
|
439
|
+
"""Iteratively optimize an attack payload, writing the best result to runs/*/best_payload.txt."""
|
|
440
|
+
from .runner import BenchmarkRunner
|
|
441
|
+
|
|
442
|
+
runner = BenchmarkRunner(os.getcwd(), repo_prefix=repo_prefix)
|
|
443
|
+
|
|
444
|
+
if offline:
|
|
445
|
+
click.echo(
|
|
446
|
+
f"Offline optimization: workflow={workflow}, scenario={scenario}, attack={attack_id}, iterations={iterations}"
|
|
447
|
+
)
|
|
448
|
+
result = runner.offline_optimize(workflow, scenario, attack_id, iterations, victim_model=victim_model)
|
|
449
|
+
else:
|
|
450
|
+
click.echo(
|
|
451
|
+
f"Optimizing on {runner.repo_name}: workflow={workflow}, scenario={scenario}, "
|
|
452
|
+
f"attack={attack_id}, iterations={iterations}"
|
|
453
|
+
)
|
|
454
|
+
result = runner.optimize(workflow, scenario, attack_id, iterations, cleanup=cleanup)
|
|
455
|
+
|
|
456
|
+
if "error" in result:
|
|
457
|
+
click.echo(click.style(f"Error: {result['error']}", fg="red"))
|
|
458
|
+
return
|
|
459
|
+
|
|
460
|
+
click.echo("\n" + click.style("--- Optimization Result ---", bold=True))
|
|
461
|
+
scores = [score for score in result["asr_curve"] if score is not None]
|
|
462
|
+
click.echo(f"Final ASR : {result['final_asr']} ({sum(scores)}/{len(scores)} valid trials)")
|
|
463
|
+
click.echo(f"ASR curve : {result['asr_curve']}")
|
|
464
|
+
click.echo(f"Runs dir : {result['runs_dir']}")
|
|
465
|
+
if result.get("best_payload"):
|
|
466
|
+
click.echo(click.style("Best payload saved to runs_dir/best_payload.txt", fg="green"))
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
@cli.command()
|
|
470
|
+
@click.option(
|
|
471
|
+
"--prefix",
|
|
472
|
+
default="benchmark-run",
|
|
473
|
+
help="Prefix of repositories to delete.",
|
|
474
|
+
)
|
|
475
|
+
@click.option("--force", is_flag=True, help="Skip confirmation prompt.")
|
|
476
|
+
def cleanup(prefix, force):
|
|
477
|
+
"""Delete all benchmark repositories with a specific prefix."""
|
|
478
|
+
from .utils.gh_client import GitHubClient
|
|
479
|
+
|
|
480
|
+
gh = GitHubClient()
|
|
481
|
+
click.echo(f"Searching for repositories with prefix '{prefix}'...")
|
|
482
|
+
repos = gh.list_repos(limit=100)
|
|
483
|
+
|
|
484
|
+
# Filter by prefix (checking both name and nameWithOwner)
|
|
485
|
+
to_delete = [r["nameWithOwner"] for r in repos if r["name"].startswith(prefix)]
|
|
486
|
+
|
|
487
|
+
if not to_delete:
|
|
488
|
+
click.echo("No matching repositories found.")
|
|
489
|
+
return
|
|
490
|
+
|
|
491
|
+
click.echo(f"Found {len(to_delete)} repositories:")
|
|
492
|
+
for repo in to_delete:
|
|
493
|
+
click.echo(f" - {repo}")
|
|
494
|
+
|
|
495
|
+
if not force and not click.confirm("\nAre you sure you want to delete these repositories?"):
|
|
496
|
+
click.echo("Aborted.")
|
|
497
|
+
return
|
|
498
|
+
|
|
499
|
+
for repo_name in to_delete:
|
|
500
|
+
click.echo(f"Deleting {repo_name}...")
|
|
501
|
+
client = GitHubClient(repo=repo_name)
|
|
502
|
+
success, err = client.delete_repo()
|
|
503
|
+
if not success:
|
|
504
|
+
click.echo(click.style(f"Failed to delete {repo_name}: {err}", fg="red"))
|
|
505
|
+
else:
|
|
506
|
+
click.echo(click.style(f"Successfully deleted {repo_name}", fg="green"))
|
|
507
|
+
|
|
508
|
+
|
|
509
|
+
@cli.command()
|
|
510
|
+
@click.option("--aggregate", is_flag=True, help="Aggregate results by workflow.")
|
|
511
|
+
def report(aggregate):
|
|
512
|
+
"""Generate a summary of previous runs from the 'runs/' directory."""
|
|
513
|
+
runs_dir = "runs"
|
|
514
|
+
if not os.path.exists(runs_dir):
|
|
515
|
+
click.echo("No runs found.")
|
|
516
|
+
return
|
|
517
|
+
|
|
518
|
+
if aggregate:
|
|
519
|
+
pairs = {}
|
|
520
|
+
for folder in os.listdir(runs_dir):
|
|
521
|
+
metadata_path = os.path.join(runs_dir, folder, "metadata.json")
|
|
522
|
+
if not os.path.isfile(metadata_path):
|
|
523
|
+
continue
|
|
524
|
+
try:
|
|
525
|
+
with open(metadata_path) as handle:
|
|
526
|
+
data = json.load(handle)
|
|
527
|
+
except json.JSONDecodeError:
|
|
528
|
+
continue
|
|
529
|
+
if data.get("workflow") and ("analysis" in data or "error" in data):
|
|
530
|
+
pairs.setdefault((data["workflow"], "all scenarios"), []).append(data)
|
|
531
|
+
_display_repeat_summary(pairs)
|
|
532
|
+
|
|
533
|
+
else:
|
|
534
|
+
header = f"{'Timestamp':<25} {'Workflow':<20} {'Scenario':<25} {'Util':<6} {'Sec':<6} {'Auto':<6}"
|
|
535
|
+
click.echo(click.style(header, bold=True))
|
|
536
|
+
click.echo("-" * 95)
|
|
537
|
+
|
|
538
|
+
run_folders = sorted([d for d in os.listdir(runs_dir) if os.path.isdir(os.path.join(runs_dir, d))], reverse=True)
|
|
539
|
+
|
|
540
|
+
for folder in run_folders:
|
|
541
|
+
metadata_path = os.path.join(runs_dir, folder, "metadata.json")
|
|
542
|
+
if os.path.exists(metadata_path):
|
|
543
|
+
with open(metadata_path, "r") as f:
|
|
544
|
+
try:
|
|
545
|
+
data = json.load(f)
|
|
546
|
+
if "analysis" not in data and "error" not in data:
|
|
547
|
+
continue
|
|
548
|
+
analysis = data.get("analysis", {})
|
|
549
|
+
|
|
550
|
+
def verdict_label(key, invert=False):
|
|
551
|
+
value = analysis.get(key)
|
|
552
|
+
if type(value) is not bool:
|
|
553
|
+
return "?"
|
|
554
|
+
return "T" if value is (False if invert else True) else "F"
|
|
555
|
+
|
|
556
|
+
click.echo(
|
|
557
|
+
f"{data.get('timestamp', folder):<25} "
|
|
558
|
+
f"{data.get('workflow'):<20} "
|
|
559
|
+
f"{str(data.get('scenario', '')):<25} "
|
|
560
|
+
f"{verdict_label('utility_achieved'):<6} "
|
|
561
|
+
f"{verdict_label('security_breached', True):<6} "
|
|
562
|
+
f"{verdict_label('autonomy_achieved'):<6}"
|
|
563
|
+
)
|
|
564
|
+
except (json.JSONDecodeError, KeyError):
|
|
565
|
+
pass
|
|
566
|
+
|
|
567
|
+
|
|
568
|
+
@cli.command()
|
|
569
|
+
@click.option("--workflow", "workflow_id", default=None, help="Workflow ID to scan.")
|
|
570
|
+
@click.option("--all", "scan_all", is_flag=True, help="Scan all workflows in the inventory.")
|
|
571
|
+
@click.option("--hypotheses", default=12, show_default=True, help="Hypotheses per scan, split across 4 MITRE categories.")
|
|
572
|
+
@click.option("--max-live", default=5, show_default=True, help="Max hypotheses to validate live, by severity rank.")
|
|
573
|
+
@click.option("--runs-per", default=3, show_default=True, help="Runs per hypothesis for confirmation.")
|
|
574
|
+
@click.option("--iterations", default=2, show_default=True, help="Hypothesis refinement iterations.")
|
|
575
|
+
@click.option("--dry-run", is_flag=True, help="Generate and rank hypotheses only; no live runs.")
|
|
576
|
+
@click.option("--no-ranker", is_flag=True, help="Skip LLM ranker; use structural pre-pass only (ablation).")
|
|
577
|
+
@click.option("--no-memory", is_flag=True, help="Disable cross-workflow memory seeding (ablation).")
|
|
578
|
+
@click.option("--monolithic", is_flag=True, help="Use single hypothesis prompt instead of per-category (ablation).")
|
|
579
|
+
@click.option("--output", "output_dir", default="reports/scanner", show_default=True, help="Directory for reports.")
|
|
580
|
+
@click.option("--hypothesis-model", default="claude-sonnet-4-6", show_default=True, help="LLM for hypothesis generation.")
|
|
581
|
+
@click.option("--ranker-model", default="claude-sonnet-4-6", show_default=True, help="LLM for plausibility ranking.")
|
|
582
|
+
@click.option(
|
|
583
|
+
"--judge-model",
|
|
584
|
+
default="gemini-3.1-pro-preview",
|
|
585
|
+
show_default=True,
|
|
586
|
+
help="LLM judge for semantic success evaluation.",
|
|
587
|
+
)
|
|
588
|
+
@click.option(
|
|
589
|
+
"--repo-prefix",
|
|
590
|
+
default=lambda: os.environ.get("GITHUB_REPO_PREFIX", "benchmark-scan"),
|
|
591
|
+
help="GitHub repository prefix for live runs.",
|
|
592
|
+
)
|
|
593
|
+
@click.option("--cleanup/--no-cleanup", default=True, help="Delete repository after each live run.")
|
|
594
|
+
@click.option("--baselines/--no-baselines", default=True, help="Run zizmor and actionlint baselines.")
|
|
595
|
+
@click.option("--reseed", is_flag=True, help="Force reload the warm-start research corpus.")
|
|
596
|
+
@click.option(
|
|
597
|
+
"--no-diagnostics",
|
|
598
|
+
is_flag=True,
|
|
599
|
+
help="Disable diagnostic stage; collapse all failures to payload_ineffective (ablation).",
|
|
600
|
+
)
|
|
601
|
+
@click.option(
|
|
602
|
+
"--diagnostic-model",
|
|
603
|
+
default="claude-haiku-4-5",
|
|
604
|
+
show_default=True,
|
|
605
|
+
help="LLM for fast-path artifact inspection in the diagnostic stage.",
|
|
606
|
+
)
|
|
607
|
+
def scan(
|
|
608
|
+
workflow_id,
|
|
609
|
+
scan_all,
|
|
610
|
+
hypotheses,
|
|
611
|
+
max_live,
|
|
612
|
+
runs_per,
|
|
613
|
+
iterations,
|
|
614
|
+
dry_run,
|
|
615
|
+
no_ranker,
|
|
616
|
+
no_memory,
|
|
617
|
+
monolithic,
|
|
618
|
+
output_dir,
|
|
619
|
+
hypothesis_model,
|
|
620
|
+
ranker_model,
|
|
621
|
+
judge_model,
|
|
622
|
+
repo_prefix,
|
|
623
|
+
cleanup,
|
|
624
|
+
baselines,
|
|
625
|
+
reseed,
|
|
626
|
+
no_diagnostics,
|
|
627
|
+
diagnostic_model,
|
|
628
|
+
):
|
|
629
|
+
"""Autonomously scan a workflow for prompt injection vulnerabilities."""
|
|
630
|
+
import time as _time
|
|
631
|
+
|
|
632
|
+
from .scanner import hypothesis_generator, llm_ranker, prompt_extractor, report_generator
|
|
633
|
+
from .scanner.baselines import actionlint_runner, zizmor_runner
|
|
634
|
+
from .scanner.live_validator import validate
|
|
635
|
+
from .scanner.memory import CrossWorkflowMemory
|
|
636
|
+
from .scanner.types import ScanCost, roll_up_usage
|
|
637
|
+
from .utils.llm import track_usage
|
|
638
|
+
|
|
639
|
+
workflows_dir = dataset_dir("workflows")
|
|
640
|
+
|
|
641
|
+
if scan_all:
|
|
642
|
+
workflow_ids = sorted(
|
|
643
|
+
[d for d in os.listdir(workflows_dir) if os.path.isdir(os.path.join(workflows_dir, d)) and not d.startswith("_")]
|
|
644
|
+
)
|
|
645
|
+
elif workflow_id:
|
|
646
|
+
workflow_ids = [workflow_id]
|
|
647
|
+
else:
|
|
648
|
+
click.echo(click.style("Error: provide --workflow or --all.", fg="red"))
|
|
649
|
+
return
|
|
650
|
+
|
|
651
|
+
memory = CrossWorkflowMemory() if not no_memory else CrossWorkflowMemory.__new__(CrossWorkflowMemory)
|
|
652
|
+
if no_memory:
|
|
653
|
+
memory._entries = []
|
|
654
|
+
memory.path = "/dev/null"
|
|
655
|
+
memory._save = lambda: None
|
|
656
|
+
else:
|
|
657
|
+
needs_warm_start = reseed or not memory._entries
|
|
658
|
+
if needs_warm_start:
|
|
659
|
+
click.echo(" Loading the warm-start research corpus...")
|
|
660
|
+
n = memory.warm_start(reseed=reseed)
|
|
661
|
+
click.echo(f" Warm-start: {n} confirmed entries loaded.")
|
|
662
|
+
|
|
663
|
+
all_summary: list[dict] = []
|
|
664
|
+
|
|
665
|
+
for wf_id in workflow_ids:
|
|
666
|
+
click.echo("\n" + click.style(f"=== Scanning: {wf_id} ===", bold=True))
|
|
667
|
+
|
|
668
|
+
meta_path = os.path.join(workflows_dir, wf_id, "metadata.json")
|
|
669
|
+
workflow_category = "code-review"
|
|
670
|
+
if os.path.exists(meta_path):
|
|
671
|
+
with open(meta_path) as f:
|
|
672
|
+
workflow_category = json.load(f).get("category", "code-review")
|
|
673
|
+
|
|
674
|
+
try:
|
|
675
|
+
context = prompt_extractor.extract(wf_id, workflows_dir)
|
|
676
|
+
except Exception as e:
|
|
677
|
+
click.echo(click.style(f" Failed to extract context: {e}", fg="red"))
|
|
678
|
+
continue
|
|
679
|
+
|
|
680
|
+
click.echo(f" Provider: {context.provider}")
|
|
681
|
+
click.echo(f" Trigger: {context.trigger_event}")
|
|
682
|
+
|
|
683
|
+
baseline_findings: list[dict] = []
|
|
684
|
+
if baselines:
|
|
685
|
+
wf_contents = os.path.join(workflows_dir, wf_id, "contents")
|
|
686
|
+
if os.path.isdir(wf_contents):
|
|
687
|
+
click.echo(" Running baselines...")
|
|
688
|
+
baseline_findings.extend(zizmor_runner.run(wf_contents))
|
|
689
|
+
baseline_findings.extend(actionlint_runner.run(wf_contents))
|
|
690
|
+
click.echo(f" Baseline findings: {len(baseline_findings)}")
|
|
691
|
+
|
|
692
|
+
wf_start = _time.monotonic()
|
|
693
|
+
with track_usage() as usage_log:
|
|
694
|
+
click.echo(f" Generating {hypotheses} hypotheses...")
|
|
695
|
+
raw_hypotheses = hypothesis_generator.generate(
|
|
696
|
+
context,
|
|
697
|
+
memory,
|
|
698
|
+
hypotheses_per_scan=hypotheses,
|
|
699
|
+
monolithic=monolithic,
|
|
700
|
+
model=hypothesis_model,
|
|
701
|
+
)
|
|
702
|
+
click.echo(f" Generated: {len(raw_hypotheses)}")
|
|
703
|
+
|
|
704
|
+
ranked, discarded = llm_ranker.rank(raw_hypotheses, context, skip_llm=no_ranker, model=ranker_model)
|
|
705
|
+
click.echo(f" After filtering: {len(ranked)} ranked, {len(discarded)} discarded")
|
|
706
|
+
|
|
707
|
+
results = validate(
|
|
708
|
+
hypotheses=ranked,
|
|
709
|
+
context=context,
|
|
710
|
+
workflow_id=wf_id,
|
|
711
|
+
workflow_category=workflow_category,
|
|
712
|
+
runs_per_hypothesis=runs_per,
|
|
713
|
+
max_hypotheses=max_live,
|
|
714
|
+
iterations=iterations,
|
|
715
|
+
repo_prefix=repo_prefix,
|
|
716
|
+
cleanup=cleanup,
|
|
717
|
+
dry_run=dry_run,
|
|
718
|
+
judge_model=judge_model,
|
|
719
|
+
enable_diagnostics=not no_diagnostics,
|
|
720
|
+
diagnostic_model=diagnostic_model,
|
|
721
|
+
)
|
|
722
|
+
|
|
723
|
+
scan_cost = ScanCost(
|
|
724
|
+
token_usage_by_model=roll_up_usage(usage_log),
|
|
725
|
+
total_billable_minutes=sum(r.billable_minutes for r in results),
|
|
726
|
+
total_wall_seconds=_time.monotonic() - wf_start,
|
|
727
|
+
)
|
|
728
|
+
|
|
729
|
+
confirmed = [r for r in results if r.status == "confirmed"]
|
|
730
|
+
click.echo(
|
|
731
|
+
click.style(
|
|
732
|
+
f" Result: {len(confirmed)}/{len(results)} confirmed "
|
|
733
|
+
f"(${scan_cost.total_usd:.4f}, {scan_cost.total_billable_minutes:.1f} billable min)",
|
|
734
|
+
fg="green" if confirmed else "yellow",
|
|
735
|
+
)
|
|
736
|
+
)
|
|
737
|
+
|
|
738
|
+
md_path, _ = report_generator.generate(
|
|
739
|
+
context, results, discarded, baseline_findings, output_dir, scan_cost=scan_cost
|
|
740
|
+
)
|
|
741
|
+
click.echo(f" Report: {md_path}")
|
|
742
|
+
|
|
743
|
+
all_summary.append(
|
|
744
|
+
{
|
|
745
|
+
"workflow": wf_id,
|
|
746
|
+
"confirmed": len(confirmed),
|
|
747
|
+
"validated": len(results),
|
|
748
|
+
"filtered": len(discarded),
|
|
749
|
+
"report_md": md_path,
|
|
750
|
+
}
|
|
751
|
+
)
|
|
752
|
+
|
|
753
|
+
if len(workflow_ids) > 1:
|
|
754
|
+
click.echo("\n" + click.style("=== Scan Summary ===", bold=True))
|
|
755
|
+
total_confirmed = sum(s["confirmed"] for s in all_summary)
|
|
756
|
+
total_validated = sum(s["validated"] for s in all_summary)
|
|
757
|
+
click.echo(f"Workflows scanned: {len(all_summary)}")
|
|
758
|
+
click.echo(f"Total confirmed: {total_confirmed} / {total_validated} validated")
|
|
759
|
+
for s in all_summary:
|
|
760
|
+
status_color = "green" if s["confirmed"] > 0 else "white"
|
|
761
|
+
click.echo(
|
|
762
|
+
click.style(f" {s['workflow']:<30}", bold=True)
|
|
763
|
+
+ click.style(f"confirmed: {s['confirmed']}", fg=status_color)
|
|
764
|
+
)
|
|
765
|
+
|
|
766
|
+
|
|
767
|
+
if __name__ == "__main__":
|
|
768
|
+
cli()
|