gitinject 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gitinject/__init__.py +1 -0
- gitinject/__main__.py +5 -0
- gitinject/analyzer.py +67 -0
- gitinject/attacks/__init__.py +27 -0
- gitinject/attacks/autoinject.py +165 -0
- gitinject/attacks/base.py +33 -0
- gitinject/attacks/static.py +25 -0
- gitinject/cli.py +768 -0
- gitinject/data/research/scenarios/claude_skills_injection.md +96 -0
- gitinject/data/research/scenarios/cline_issue_body_injection.md +93 -0
- gitinject/data/research/scenarios/codex_agents_md_injection.md +109 -0
- gitinject/data/research/scenarios/dos_request_flood.md +133 -0
- gitinject/data/research/scenarios/dropped/ci_log_injection_workflow_poisoning.md +138 -0
- gitinject/data/research/scenarios/dropped/claude_md_instructions_injection.md +158 -0
- gitinject/data/research/scenarios/dropped/supply_chain_token_pivot.md +101 -0
- gitinject/data/research/scenarios/gemini_api_key_exfiltration.md +68 -0
- gitinject/data/research/scenarios/gemini_api_key_exfiltration_replication.md +0 -0
- gitinject/data/research/scenarios/gemini_md_instructions_injection.md +131 -0
- gitinject/data/research/scenarios/nsfw_api_key_block.md +114 -0
- gitinject/data/research/scenarios/pr_token_exfiltration_via_git_config.md +117 -0
- gitinject/data/research/scenarios/supply_chain_malicious_code.md +135 -0
- gitinject/evaluators.py +169 -0
- gitinject/evidence.py +72 -0
- gitinject/gl_runner.py +163 -0
- gitinject/resources.py +30 -0
- gitinject/run_context.py +72 -0
- gitinject/run_record.py +113 -0
- gitinject/runner.py +1024 -0
- gitinject/scanner/__init__.py +22 -0
- gitinject/scanner/baselines/__init__.py +0 -0
- gitinject/scanner/baselines/actionlint_runner.py +44 -0
- gitinject/scanner/baselines/zizmor_runner.py +35 -0
- gitinject/scanner/diagnostics.py +174 -0
- gitinject/scanner/hypothesis_generator.py +164 -0
- gitinject/scanner/live_validator.py +284 -0
- gitinject/scanner/llm_ranker.py +120 -0
- gitinject/scanner/memory.py +239 -0
- gitinject/scanner/primitives.py +236 -0
- gitinject/scanner/prompt_extractor.py +134 -0
- gitinject/scanner/recipe_scenario.py +203 -0
- gitinject/scanner/report_generator.py +230 -0
- gitinject/scanner/types.py +179 -0
- gitinject/scenario_base.py +147 -0
- gitinject/scenario_loader.py +77 -0
- gitinject/scenarios/__init__.py +0 -0
- gitinject/scenarios/benign/documentation_out_of_sync/contents/math_utils.py +3 -0
- gitinject/scenarios/benign/documentation_out_of_sync/contents/pydantic/v1/utils.py +21 -0
- gitinject/scenarios/benign/documentation_out_of_sync/scenario.py +57 -0
- gitinject/scenarios/benign/issue_deduplication_utility/scenario.py +76 -0
- gitinject/scenarios/benign/repository_onboarding_qa/contents/README.md +3 -0
- gitinject/scenarios/benign/repository_onboarding_qa/scenario.py +52 -0
- gitinject/scenarios/benign/secrets_false_positive/contents/tests/test_auth.py +14 -0
- gitinject/scenarios/benign/secrets_false_positive/scenario.py +58 -0
- gitinject/scenarios/benign/test_coverage_gap/contents/src/requests/auth_extra.py +33 -0
- gitinject/scenarios/benign/test_coverage_gap/scenario.py +56 -0
- gitinject/scenarios/benign/vulnerability_fix_verification/contents/src/flask/utils.py +15 -0
- gitinject/scenarios/benign/vulnerability_fix_verification/scenario.py +57 -0
- gitinject/scenarios/benign/vulnerable_code_review/contents/examples/tutorial/flaskr/db_utils.py +18 -0
- gitinject/scenarios/benign/vulnerable_code_review/scenario.py +55 -0
- gitinject/scenarios/benign/weekly_activity_summary/scenario.py +83 -0
- gitinject/scenarios/malicious/agents_md_approval_manipulation/contents/AGENTS.md +25 -0
- gitinject/scenarios/malicious/agents_md_approval_manipulation/contents/src/flask/csrf.py +151 -0
- gitinject/scenarios/malicious/agents_md_approval_manipulation/contents/tests/test_csrf.py +71 -0
- gitinject/scenarios/malicious/agents_md_approval_manipulation/scenario.py +121 -0
- gitinject/scenarios/malicious/claude_md_instructions_injection/contents/CLAUDE.md +23 -0
- gitinject/scenarios/malicious/claude_md_instructions_injection/contents/scripts/deploy_check.sh +29 -0
- gitinject/scenarios/malicious/claude_md_instructions_injection/scenario.py +93 -0
- gitinject/scenarios/malicious/cline_issue_body_injection/scenario.py +132 -0
- gitinject/scenarios/malicious/codex_agents_md_injection/contents/AGENTS.md +36 -0
- gitinject/scenarios/malicious/codex_agents_md_injection/contents/scripts/utils.py +12 -0
- gitinject/scenarios/malicious/codex_agents_md_injection/scenario.py +128 -0
- gitinject/scenarios/malicious/dos_request_flood/scenario.py +80 -0
- gitinject/scenarios/malicious/gemini_md_approval_manipulation/contents/src/flask/csrf.py +151 -0
- gitinject/scenarios/malicious/gemini_md_approval_manipulation/contents/tests/test_csrf.py +71 -0
- gitinject/scenarios/malicious/gemini_md_approval_manipulation/scenario.py +121 -0
- gitinject/scenarios/malicious/gemini_md_instructions_injection/contents/scripts/deploy_check.sh +38 -0
- gitinject/scenarios/malicious/gemini_md_instructions_injection/scenario.py +105 -0
- gitinject/scenarios/malicious/issue_onboarding_secret_exfiltration/scenario.py +76 -0
- gitinject/scenarios/malicious/pr_token_exfiltration_via_git_config/contents/scripts/deploy_check.sh +7 -0
- gitinject/scenarios/malicious/pr_token_exfiltration_via_git_config/scenario.py +113 -0
- gitinject/scenarios/malicious/supply_chain_malicious_code/contents/CLAUDE.md +25 -0
- gitinject/scenarios/malicious/supply_chain_malicious_code/contents/src/flask/csrf.py +151 -0
- gitinject/scenarios/malicious/supply_chain_malicious_code/contents/tests/test_csrf.py +71 -0
- gitinject/scenarios/malicious/supply_chain_malicious_code/scenario.py +123 -0
- gitinject/scenarios/malicious/unauthorized_pr_approval/contents/django/utils/crypto.py +13 -0
- gitinject/scenarios/malicious/unauthorized_pr_approval/scenario.py +57 -0
- gitinject/simulator.py +89 -0
- gitinject/utils/__init__.py +0 -0
- gitinject/utils/gh_client.py +628 -0
- gitinject/utils/gl_client.py +132 -0
- gitinject/utils/gl_provisioner.py +83 -0
- gitinject/utils/llm.py +205 -0
- gitinject/utils/provisioner.py +114 -0
- gitinject/utils/scenario_resources.py +33 -0
- gitinject/utils/types.py +49 -0
- gitinject/workflows/__init__.py +0 -0
- gitinject/workflows/claude-ci-auto-fix/contents/.github/workflows/main.yml +107 -0
- gitinject/workflows/claude-ci-auto-fix/metadata.json +10 -0
- gitinject/workflows/claude-general/contents/.github/workflows/main.yml +58 -0
- gitinject/workflows/claude-general/metadata.json +10 -0
- gitinject/workflows/claude-gitlab-mr-review/contents/.gitlab-ci.yml +36 -0
- gitinject/workflows/claude-gitlab-mr-review/metadata.json +11 -0
- gitinject/workflows/claude-issue-deduplication/contents/.github/workflows/main.yml +66 -0
- gitinject/workflows/claude-issue-deduplication/metadata.json +10 -0
- gitinject/workflows/claude-issue-triage/contents/.github/workflows/main.yml +34 -0
- gitinject/workflows/claude-issue-triage/metadata.json +10 -0
- gitinject/workflows/claude-manual-analysis/contents/.github/workflows/main.yml +42 -0
- gitinject/workflows/claude-manual-analysis/metadata.json +10 -0
- gitinject/workflows/claude-pr-review/contents/.github/workflows/main.yml +77 -0
- gitinject/workflows/claude-pr-review/metadata.json +10 -0
- gitinject/workflows/claude-pr-review-authors/contents/.github/workflows/main.yml +48 -0
- gitinject/workflows/claude-pr-review-authors/metadata.json +10 -0
- gitinject/workflows/claude-pr-review-paths/contents/.github/workflows/main.yml +49 -0
- gitinject/workflows/claude-pr-review-paths/metadata.json +10 -0
- gitinject/workflows/claude-test-analysis/contents/.github/workflows/main.yml +114 -0
- gitinject/workflows/claude-test-analysis/metadata.json +10 -0
- gitinject/workflows/cline-assistant/contents/.github/workflows/main.yml +87 -0
- gitinject/workflows/cline-assistant/contents/git-scripts/analyze-issue.sh +43 -0
- gitinject/workflows/cline-assistant/metadata.json +10 -0
- gitinject/workflows/codex-pr-review/contents/.github/workflows/main.yml +73 -0
- gitinject/workflows/codex-pr-review/metadata.json +10 -0
- gitinject/workflows/copilot-ci-doctor/contents/.github/workflows/ci-doctor.yml +1161 -0
- gitinject/workflows/copilot-ci-doctor/metadata.json +10 -0
- gitinject/workflows/copilot-lean-squad/contents/.github/workflows/lean-squad.yml +1313 -0
- gitinject/workflows/copilot-lean-squad/metadata.json +10 -0
- gitinject/workflows/copilot-malicious-scan/contents/.github/workflows/daily-malicious-code-scan.yml +899 -0
- gitinject/workflows/copilot-malicious-scan/metadata.json +10 -0
- gitinject/workflows/copilot-repo-assist/contents/.github/workflows/repo-assist.yml +1503 -0
- gitinject/workflows/copilot-repo-assist/metadata.json +10 -0
- gitinject/workflows/copilot-wiki-writer/contents/.github/workflows/agentic-wiki-writer.yml +1316 -0
- gitinject/workflows/copilot-wiki-writer/metadata.json +10 -0
- gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-invoke.yml +122 -0
- gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-plan-execute.yml +130 -0
- gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-review.yml +118 -0
- gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-scheduled-triage.yml +220 -0
- gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-triage.yml +160 -0
- gitinject/workflows/gemini-assistant/contents/.github/workflows/main.yml +220 -0
- gitinject/workflows/gemini-assistant/metadata.json +10 -0
- gitinject/workflows/gemini-assistant-original/AWESOME.md +118 -0
- gitinject/workflows/gemini-assistant-original/CONFIGURATION.md +162 -0
- gitinject/workflows/gemini-assistant-original/README.md +93 -0
- gitinject/workflows/gemini-assistant-original/gemini-assistant/README.md +192 -0
- gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-invoke.toml +94 -0
- gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-invoke.yml +131 -0
- gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-plan-execute.toml +100 -0
- gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-plan-execute.yml +139 -0
- gitinject/workflows/gemini-assistant-original/gemini-dispatch/README.md +49 -0
- gitinject/workflows/gemini-assistant-original/gemini-dispatch/gemini-dispatch.yml +221 -0
- gitinject/workflows/gemini-assistant-original/issue-triage/README.md +190 -0
- gitinject/workflows/gemini-assistant-original/issue-triage/gemini-scheduled-triage.toml +96 -0
- gitinject/workflows/gemini-assistant-original/issue-triage/gemini-scheduled-triage.yml +223 -0
- gitinject/workflows/gemini-assistant-original/issue-triage/gemini-triage.toml +32 -0
- gitinject/workflows/gemini-assistant-original/issue-triage/gemini-triage.yml +167 -0
- gitinject/workflows/gemini-assistant-original/metadata.json +10 -0
- gitinject/workflows/gemini-assistant-original/pr-review/README.md +337 -0
- gitinject/workflows/gemini-assistant-original/pr-review/gemini-review.toml +176 -0
- gitinject/workflows/gemini-assistant-original/pr-review/gemini-review.yml +119 -0
- gitinject/workflows/opencode-pr-review/contents/.github/workflows/main.yml +28 -0
- gitinject/workflows/opencode-pr-review/metadata.json +10 -0
- gitinject-0.1.0.dist-info/METADATA +128 -0
- gitinject-0.1.0.dist-info/RECORD +164 -0
- gitinject-0.1.0.dist-info/WHEEL +4 -0
- gitinject-0.1.0.dist-info/entry_points.txt +2 -0
- gitinject-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import time
|
|
3
|
+
|
|
4
|
+
import click
|
|
5
|
+
import gitlab
|
|
6
|
+
import gitlab.exceptions
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class GitLabClient:
|
|
10
|
+
"""Thin wrapper around python-gitlab for benchmark provisioning and evaluation."""
|
|
11
|
+
|
|
12
|
+
def __init__(self, token: str | None = None, url: str = "https://gitlab.com"):
|
|
13
|
+
token = token or os.environ.get("GITLAB_TOKEN")
|
|
14
|
+
if not token:
|
|
15
|
+
click.echo(
|
|
16
|
+
click.style("Error: GITLAB_TOKEN not set.", fg="red"),
|
|
17
|
+
err=True,
|
|
18
|
+
)
|
|
19
|
+
raise RuntimeError("Missing GitLab authentication")
|
|
20
|
+
self.gl = gitlab.Gitlab(url, private_token=token)
|
|
21
|
+
self.project = None
|
|
22
|
+
|
|
23
|
+
def create_project(
|
|
24
|
+
self,
|
|
25
|
+
name: str,
|
|
26
|
+
namespace_path: str | None = None,
|
|
27
|
+
visibility: str = "public",
|
|
28
|
+
):
|
|
29
|
+
kwargs: dict = {"name": name, "visibility": visibility, "initialize_with_readme": True}
|
|
30
|
+
if namespace_path:
|
|
31
|
+
results = self.gl.namespaces.list(search=namespace_path)
|
|
32
|
+
if results:
|
|
33
|
+
kwargs["namespace_id"] = results[0].id
|
|
34
|
+
self.project = self.gl.projects.create(kwargs)
|
|
35
|
+
return self.project
|
|
36
|
+
|
|
37
|
+
def delete_project(self) -> None:
|
|
38
|
+
if self.project:
|
|
39
|
+
try:
|
|
40
|
+
self.project.delete()
|
|
41
|
+
except gitlab.exceptions.GitlabDeleteError as e:
|
|
42
|
+
click.echo(click.style(f"Warning: failed to delete project: {e}", fg="yellow"))
|
|
43
|
+
self.project = None
|
|
44
|
+
|
|
45
|
+
def push_files(self, branch: str, files: dict[str, str], commit_message: str = "add files") -> None:
|
|
46
|
+
"""Batch-commit multiple files to a branch in a single commit."""
|
|
47
|
+
if not files:
|
|
48
|
+
return
|
|
49
|
+
existing = set()
|
|
50
|
+
try:
|
|
51
|
+
tree = self.project.repository_tree(ref=branch, recursive=True, all=True)
|
|
52
|
+
existing = {item["path"] for item in tree}
|
|
53
|
+
except gitlab.exceptions.GitlabGetError:
|
|
54
|
+
pass
|
|
55
|
+
|
|
56
|
+
actions = []
|
|
57
|
+
for path, content in files.items():
|
|
58
|
+
action = "update" if path in existing else "create"
|
|
59
|
+
actions.append({"action": action, "file_path": path, "content": content})
|
|
60
|
+
|
|
61
|
+
self.project.commits.create(
|
|
62
|
+
{
|
|
63
|
+
"branch": branch,
|
|
64
|
+
"commit_message": commit_message,
|
|
65
|
+
"actions": actions,
|
|
66
|
+
}
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
def create_branch(self, name: str, ref: str = "main") -> None:
|
|
70
|
+
try:
|
|
71
|
+
self.project.branches.create({"branch": name, "ref": ref})
|
|
72
|
+
except gitlab.exceptions.GitlabCreateError:
|
|
73
|
+
pass # Branch already exists
|
|
74
|
+
|
|
75
|
+
def create_merge_request(
|
|
76
|
+
self,
|
|
77
|
+
source_branch: str,
|
|
78
|
+
target_branch: str,
|
|
79
|
+
title: str,
|
|
80
|
+
description: str = "",
|
|
81
|
+
) -> int:
|
|
82
|
+
mr = self.project.mergerequests.create(
|
|
83
|
+
{
|
|
84
|
+
"source_branch": source_branch,
|
|
85
|
+
"target_branch": target_branch,
|
|
86
|
+
"title": title,
|
|
87
|
+
"description": description,
|
|
88
|
+
}
|
|
89
|
+
)
|
|
90
|
+
return mr.iid
|
|
91
|
+
|
|
92
|
+
def get_mr_notes(self, mr_iid: int) -> list[str]:
|
|
93
|
+
mr = self.project.mergerequests.get(mr_iid)
|
|
94
|
+
notes = mr.notes.list(all=True)
|
|
95
|
+
return [n.body for n in notes if not n.system]
|
|
96
|
+
|
|
97
|
+
def get_mr_pipelines(self, mr_iid: int) -> list:
|
|
98
|
+
mr = self.project.mergerequests.get(mr_iid)
|
|
99
|
+
return mr.pipelines.list()
|
|
100
|
+
|
|
101
|
+
def wait_for_pipeline(self, pipeline_id: int, timeout_s: int = 600, poll_interval: int = 15) -> str:
|
|
102
|
+
deadline = time.time() + timeout_s
|
|
103
|
+
while time.time() < deadline:
|
|
104
|
+
pipeline = self.project.pipelines.get(pipeline_id)
|
|
105
|
+
if pipeline.status not in ("created", "waiting_for_resource", "preparing", "pending", "running", "scheduled"):
|
|
106
|
+
return pipeline.status
|
|
107
|
+
click.echo(f" Pipeline {pipeline_id} status: {pipeline.status}...")
|
|
108
|
+
time.sleep(poll_interval)
|
|
109
|
+
return "timeout"
|
|
110
|
+
|
|
111
|
+
def set_variable(self, key: str, value: str, masked: bool = True, protected: bool = False) -> None:
|
|
112
|
+
try:
|
|
113
|
+
var = self.project.variables.get(key)
|
|
114
|
+
var.value = value
|
|
115
|
+
var.save()
|
|
116
|
+
except gitlab.exceptions.GitlabGetError:
|
|
117
|
+
self.project.variables.create(
|
|
118
|
+
{
|
|
119
|
+
"key": key,
|
|
120
|
+
"value": value,
|
|
121
|
+
"masked": masked,
|
|
122
|
+
"protected": protected,
|
|
123
|
+
}
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
def get_project_url(self) -> str | None:
|
|
127
|
+
return self.project.web_url if self.project else None
|
|
128
|
+
|
|
129
|
+
def get_default_branch(self) -> str:
|
|
130
|
+
if self.project:
|
|
131
|
+
return self.project.default_branch or "main"
|
|
132
|
+
return "main"
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import time
|
|
3
|
+
|
|
4
|
+
import click
|
|
5
|
+
|
|
6
|
+
from .gl_client import GitLabClient
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class GitLabProvisioner:
|
|
10
|
+
"""Handles lifecycle of a GitLab project for benchmarking."""
|
|
11
|
+
|
|
12
|
+
def __init__(self, gl_client: GitLabClient):
|
|
13
|
+
self.gl_client = gl_client
|
|
14
|
+
|
|
15
|
+
def provision(
|
|
16
|
+
self,
|
|
17
|
+
project_name: str,
|
|
18
|
+
workflow_dir: str,
|
|
19
|
+
required_files: dict | None = None,
|
|
20
|
+
branch: str | None = None,
|
|
21
|
+
variables: dict | None = None,
|
|
22
|
+
) -> None:
|
|
23
|
+
click.echo(f"Creating GitLab project {project_name}...")
|
|
24
|
+
self.gl_client.create_project(project_name)
|
|
25
|
+
|
|
26
|
+
# GitLab needs a moment after initialize_with_readme before branches are writable
|
|
27
|
+
time.sleep(5)
|
|
28
|
+
|
|
29
|
+
default_branch = self.gl_client.get_default_branch()
|
|
30
|
+
|
|
31
|
+
# Collect workflow files from contents/
|
|
32
|
+
workflow_files: dict[str, str] = {}
|
|
33
|
+
contents_dir = os.path.join(workflow_dir, "contents")
|
|
34
|
+
if os.path.isdir(contents_dir):
|
|
35
|
+
for root, _, filenames in os.walk(contents_dir):
|
|
36
|
+
for filename in filenames:
|
|
37
|
+
abs_path = os.path.join(root, filename)
|
|
38
|
+
rel_path = os.path.relpath(abs_path, contents_dir)
|
|
39
|
+
workflow_files[rel_path] = self._read_file(abs_path)
|
|
40
|
+
|
|
41
|
+
if workflow_files:
|
|
42
|
+
click.echo(f"Pushing workflow files to {default_branch}...")
|
|
43
|
+
self.gl_client.push_files(default_branch, workflow_files, "provision: add CI workflow")
|
|
44
|
+
|
|
45
|
+
# Create target branch for scenario files (if different from default)
|
|
46
|
+
target_branch = branch or default_branch
|
|
47
|
+
if target_branch != default_branch:
|
|
48
|
+
click.echo(f"Creating branch {target_branch}...")
|
|
49
|
+
self.gl_client.create_branch(target_branch, ref=default_branch)
|
|
50
|
+
|
|
51
|
+
# Push scenario-specific files to target branch
|
|
52
|
+
if required_files:
|
|
53
|
+
scenario_files: dict[str, str] = {}
|
|
54
|
+
for repo_path, content_or_path in required_files.items():
|
|
55
|
+
if isinstance(content_or_path, str) and os.path.exists(content_or_path):
|
|
56
|
+
scenario_files[repo_path] = self._read_file(content_or_path)
|
|
57
|
+
else:
|
|
58
|
+
scenario_files[repo_path] = str(content_or_path)
|
|
59
|
+
if scenario_files:
|
|
60
|
+
click.echo(f"Pushing scenario files to {target_branch}...")
|
|
61
|
+
self.gl_client.push_files(target_branch, scenario_files, "provision: add scenario files")
|
|
62
|
+
|
|
63
|
+
# Set CI/CD variables
|
|
64
|
+
if variables:
|
|
65
|
+
for key, value in variables.items():
|
|
66
|
+
if value:
|
|
67
|
+
click.echo(f"Setting CI/CD variable '{key}'...")
|
|
68
|
+
self.gl_client.set_variable(key, value, masked=True)
|
|
69
|
+
|
|
70
|
+
click.echo(f"Project ready: {self.gl_client.get_project_url()}")
|
|
71
|
+
|
|
72
|
+
def teardown(self) -> None:
|
|
73
|
+
click.echo("Deleting GitLab project...")
|
|
74
|
+
self.gl_client.delete_project()
|
|
75
|
+
click.echo("Project deleted.")
|
|
76
|
+
|
|
77
|
+
def _read_file(self, path: str) -> str:
|
|
78
|
+
with open(path, "rb") as f:
|
|
79
|
+
raw = f.read()
|
|
80
|
+
try:
|
|
81
|
+
return raw.decode("utf-8")
|
|
82
|
+
except UnicodeDecodeError:
|
|
83
|
+
return raw.decode("latin-1")
|
gitinject/utils/llm.py
ADDED
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from contextvars import ContextVar
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class LLMError(Exception):
|
|
9
|
+
pass
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass
|
|
13
|
+
class LLMResponse:
|
|
14
|
+
text: str
|
|
15
|
+
input_tokens: int
|
|
16
|
+
output_tokens: int
|
|
17
|
+
model: str
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
_usage_log: ContextVar[list[LLMResponse] | None] = ContextVar("usage_log", default=None)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class track_usage:
|
|
24
|
+
"""Context manager that captures every LLMResponse produced within its scope.
|
|
25
|
+
|
|
26
|
+
Usage:
|
|
27
|
+
with track_usage() as log:
|
|
28
|
+
call_llm(...)
|
|
29
|
+
call_llm(...)
|
|
30
|
+
# log is a list[LLMResponse]
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
def __enter__(self) -> list[LLMResponse]:
|
|
34
|
+
self._log: list[LLMResponse] = []
|
|
35
|
+
self._token = _usage_log.set(self._log)
|
|
36
|
+
return self._log
|
|
37
|
+
|
|
38
|
+
def __exit__(self, *exc):
|
|
39
|
+
_usage_log.reset(self._token)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _record(resp: LLMResponse) -> None:
|
|
43
|
+
log = _usage_log.get()
|
|
44
|
+
if log is not None:
|
|
45
|
+
log.append(resp)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _parse_model(model: str) -> tuple[str, str]:
|
|
49
|
+
"""Parse 'provider/model-name' into (provider, model_name).
|
|
50
|
+
|
|
51
|
+
Supported providers: anthropic, google, openai, openrouter.
|
|
52
|
+
Bare model names fall back to heuristic detection.
|
|
53
|
+
"""
|
|
54
|
+
if model.startswith("openrouter/"):
|
|
55
|
+
return "openrouter", model[len("openrouter/") :]
|
|
56
|
+
if "/" in model:
|
|
57
|
+
provider, model_name = model.split("/", 1)
|
|
58
|
+
return provider, model_name
|
|
59
|
+
if model.startswith("claude"):
|
|
60
|
+
return "anthropic", model
|
|
61
|
+
if model.startswith("gemini"):
|
|
62
|
+
return "google", model
|
|
63
|
+
return "openai", model
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def call_llm(
|
|
67
|
+
model: str,
|
|
68
|
+
system: str,
|
|
69
|
+
user: str,
|
|
70
|
+
max_tokens: int = 2048,
|
|
71
|
+
temperature: float | None = None,
|
|
72
|
+
) -> LLMResponse:
|
|
73
|
+
"""Call an LLM and return an LLMResponse with text and token usage.
|
|
74
|
+
|
|
75
|
+
model format: 'provider/model-name' e.g. 'anthropic/claude-sonnet-4-6',
|
|
76
|
+
'google/gemini-2.5-flash', 'openai/gpt-4o', 'openrouter/openai/gpt-4o'.
|
|
77
|
+
Bare model names are accepted with heuristic provider detection.
|
|
78
|
+
"""
|
|
79
|
+
provider, model_name = _parse_model(model)
|
|
80
|
+
|
|
81
|
+
try:
|
|
82
|
+
if provider == "anthropic":
|
|
83
|
+
resp = _call_anthropic(model_name, system, user, max_tokens, temperature)
|
|
84
|
+
elif provider in ("google", "gemini"):
|
|
85
|
+
resp = _call_google(model_name, system, user, max_tokens, temperature)
|
|
86
|
+
elif provider == "openrouter":
|
|
87
|
+
resp = _call_openai(
|
|
88
|
+
model_name,
|
|
89
|
+
system,
|
|
90
|
+
user,
|
|
91
|
+
max_tokens,
|
|
92
|
+
temperature,
|
|
93
|
+
base_url="https://openrouter.ai/api/v1",
|
|
94
|
+
api_key=os.environ.get("OPENROUTER_API_KEY", ""),
|
|
95
|
+
)
|
|
96
|
+
else:
|
|
97
|
+
resp = _call_openai(model_name, system, user, max_tokens, temperature)
|
|
98
|
+
except LLMError:
|
|
99
|
+
raise
|
|
100
|
+
except Exception as e:
|
|
101
|
+
raise LLMError(str(e)) from e
|
|
102
|
+
|
|
103
|
+
_record(resp)
|
|
104
|
+
return resp
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _call_anthropic(model: str, system: str, user: str, max_tokens: int, temperature: float | None) -> LLMResponse:
|
|
108
|
+
import anthropic
|
|
109
|
+
|
|
110
|
+
try:
|
|
111
|
+
client = anthropic.Anthropic()
|
|
112
|
+
kwargs: dict = dict(
|
|
113
|
+
model=model,
|
|
114
|
+
max_tokens=max_tokens,
|
|
115
|
+
messages=[{"role": "user", "content": user}],
|
|
116
|
+
)
|
|
117
|
+
if system:
|
|
118
|
+
kwargs["system"] = system
|
|
119
|
+
if temperature is not None:
|
|
120
|
+
kwargs["temperature"] = temperature
|
|
121
|
+
response = client.messages.create(**kwargs)
|
|
122
|
+
usage = getattr(response, "usage", None)
|
|
123
|
+
return LLMResponse(
|
|
124
|
+
text=response.content[0].text,
|
|
125
|
+
input_tokens=int(getattr(usage, "input_tokens", 0) or 0),
|
|
126
|
+
output_tokens=int(getattr(usage, "output_tokens", 0) or 0),
|
|
127
|
+
model=model,
|
|
128
|
+
)
|
|
129
|
+
except anthropic.AuthenticationError as e:
|
|
130
|
+
raise LLMError(f"Anthropic authentication failed: {e}") from e
|
|
131
|
+
except anthropic.PermissionDeniedError as e:
|
|
132
|
+
raise LLMError(f"Anthropic access denied (credits exhausted?): {e}") from e
|
|
133
|
+
except anthropic.RateLimitError as e:
|
|
134
|
+
raise LLMError(f"Anthropic rate limit: {e}") from e
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _call_google(model: str, system: str, user: str, max_tokens: int, temperature: float | None) -> LLMResponse:
|
|
138
|
+
from google import genai
|
|
139
|
+
|
|
140
|
+
api_key = os.environ.get("GEMINI_API_KEY")
|
|
141
|
+
if not api_key:
|
|
142
|
+
raise LLMError("GEMINI_API_KEY is not set")
|
|
143
|
+
client = genai.Client(api_key=api_key)
|
|
144
|
+
config: dict = {"max_output_tokens": max_tokens}
|
|
145
|
+
if system:
|
|
146
|
+
config["system_instruction"] = system
|
|
147
|
+
if temperature is not None:
|
|
148
|
+
config["temperature"] = temperature
|
|
149
|
+
response = client.models.generate_content(model=model, contents=user, config=config)
|
|
150
|
+
text = response.text
|
|
151
|
+
if text is None:
|
|
152
|
+
raise LLMError(f"Google model returned empty response (blocked?): {response}")
|
|
153
|
+
usage = getattr(response, "usage_metadata", None)
|
|
154
|
+
return LLMResponse(
|
|
155
|
+
text=text,
|
|
156
|
+
input_tokens=int(getattr(usage, "prompt_token_count", 0) or 0),
|
|
157
|
+
output_tokens=int(getattr(usage, "candidates_token_count", 0) or 0),
|
|
158
|
+
model=model,
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _call_openai(
|
|
163
|
+
model: str,
|
|
164
|
+
system: str,
|
|
165
|
+
user: str,
|
|
166
|
+
max_tokens: int,
|
|
167
|
+
temperature: float | None,
|
|
168
|
+
base_url: str | None = None,
|
|
169
|
+
api_key: str | None = None,
|
|
170
|
+
) -> LLMResponse:
|
|
171
|
+
from openai import AuthenticationError, OpenAI, RateLimitError
|
|
172
|
+
|
|
173
|
+
kwargs: dict = {}
|
|
174
|
+
if base_url:
|
|
175
|
+
kwargs["base_url"] = base_url
|
|
176
|
+
if api_key:
|
|
177
|
+
kwargs["api_key"] = api_key
|
|
178
|
+
client = OpenAI(**kwargs)
|
|
179
|
+
|
|
180
|
+
messages = []
|
|
181
|
+
if system:
|
|
182
|
+
messages.append({"role": "system", "content": system})
|
|
183
|
+
messages.append({"role": "user", "content": user})
|
|
184
|
+
|
|
185
|
+
request: dict = dict(model=model, messages=messages, max_completion_tokens=max_tokens)
|
|
186
|
+
if temperature is not None:
|
|
187
|
+
request["temperature"] = temperature
|
|
188
|
+
|
|
189
|
+
try:
|
|
190
|
+
response = client.chat.completions.create(**request)
|
|
191
|
+
content = response.choices[0].message.content
|
|
192
|
+
if not content:
|
|
193
|
+
finish_reason = response.choices[0].finish_reason if response.choices else "unknown"
|
|
194
|
+
raise LLMError(f"OpenAI returned empty content (finish_reason={finish_reason}, model={model})")
|
|
195
|
+
usage = getattr(response, "usage", None)
|
|
196
|
+
return LLMResponse(
|
|
197
|
+
text=content,
|
|
198
|
+
input_tokens=int(getattr(usage, "prompt_tokens", 0) or 0),
|
|
199
|
+
output_tokens=int(getattr(usage, "completion_tokens", 0) or 0),
|
|
200
|
+
model=model,
|
|
201
|
+
)
|
|
202
|
+
except AuthenticationError as e:
|
|
203
|
+
raise LLMError(f"OpenAI authentication failed: {e}") from e
|
|
204
|
+
except RateLimitError as e:
|
|
205
|
+
raise LLMError(f"OpenAI rate limit: {e}") from e
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
import os
|
|
2
|
+
|
|
3
|
+
from .gh_client import GitHubClient
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class ProvisioningError(RuntimeError):
|
|
7
|
+
pass
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class RepoProvisioner:
|
|
11
|
+
"""Handles the physical setup of a GitHub repository for benchmarking."""
|
|
12
|
+
|
|
13
|
+
def __init__(self, gh_client: GitHubClient):
|
|
14
|
+
self.gh_client = gh_client
|
|
15
|
+
self._owns_repo = False
|
|
16
|
+
|
|
17
|
+
@staticmethod
|
|
18
|
+
def _require(result, operation):
|
|
19
|
+
success, error = result
|
|
20
|
+
if not success:
|
|
21
|
+
raise ProvisioningError(f"{operation}: {error}")
|
|
22
|
+
|
|
23
|
+
def provision(
|
|
24
|
+
self,
|
|
25
|
+
workflow_dir: str,
|
|
26
|
+
required_files: dict = None,
|
|
27
|
+
branch: str = None,
|
|
28
|
+
template_repo: str = None,
|
|
29
|
+
secrets: dict = None,
|
|
30
|
+
variables: dict = None,
|
|
31
|
+
):
|
|
32
|
+
"""Create and configure a fresh repository, failing on incomplete setup."""
|
|
33
|
+
if self._owns_repo:
|
|
34
|
+
raise ProvisioningError("A previously created repository still needs cleanup")
|
|
35
|
+
creation = self.gh_client.fork_repo(template_repo) if template_repo else self.gh_client.create_repo(public=True)
|
|
36
|
+
self._require(creation, "Create repository")
|
|
37
|
+
self._owns_repo = True
|
|
38
|
+
if template_repo:
|
|
39
|
+
self.gh_client.wait_until_ready()
|
|
40
|
+
else:
|
|
41
|
+
self._require(
|
|
42
|
+
self.gh_client.put_file(
|
|
43
|
+
"README.md", "# Benchmark Repository\nGenerated by AI Benchmark Suite.", "initial commit", "main"
|
|
44
|
+
),
|
|
45
|
+
"Initialize repository",
|
|
46
|
+
)
|
|
47
|
+
repo_info = self.gh_client.get_repo_info()
|
|
48
|
+
if not repo_info:
|
|
49
|
+
raise ProvisioningError("Created repository is not readable")
|
|
50
|
+
default_branch = repo_info.get("defaultBranchRef", {}).get("name") or "main"
|
|
51
|
+
target_branch = branch or default_branch
|
|
52
|
+
self._require(self.gh_client.enable_actions(), "Enable Actions")
|
|
53
|
+
self._require(self.gh_client.set_fork_pr_approval_policy(), "Set fork approval policy")
|
|
54
|
+
self._require(self.gh_client.enable_issues(), "Enable issues")
|
|
55
|
+
for operation, values in ((self.gh_client.set_secret, secrets), (self.gh_client.set_variable, variables)):
|
|
56
|
+
for name, value in (values or {}).items():
|
|
57
|
+
if value is None or value == "":
|
|
58
|
+
raise ProvisioningError(f"Empty configuration value: {name}")
|
|
59
|
+
self._require(operation(name, value), f"Set configuration {name}")
|
|
60
|
+
contents_dir = os.path.join(workflow_dir, "contents")
|
|
61
|
+
workflow_files = {}
|
|
62
|
+
if os.path.isdir(contents_dir):
|
|
63
|
+
for root, _, filenames in os.walk(contents_dir):
|
|
64
|
+
for filename in filenames:
|
|
65
|
+
local_path = os.path.join(root, filename)
|
|
66
|
+
workflow_files[os.path.relpath(local_path, contents_dir)] = local_path
|
|
67
|
+
elif os.path.isdir(workflow_dir):
|
|
68
|
+
for filename in os.listdir(workflow_dir):
|
|
69
|
+
if filename.endswith((".yml", ".yaml")):
|
|
70
|
+
workflow_files[f".github/workflows/{filename}"] = os.path.join(workflow_dir, filename)
|
|
71
|
+
scenario_files = required_files or {}
|
|
72
|
+
for path in scenario_files:
|
|
73
|
+
if path in workflow_files:
|
|
74
|
+
raise ProvisioningError(f"File defined by both workflow and scenario: {path}")
|
|
75
|
+
additions = {path: self._get_content(content) for path, content in workflow_files.items()}
|
|
76
|
+
self._require(
|
|
77
|
+
self.gh_client.batch_sync(additions, [".github/workflows/"], "provision workflows", default_branch),
|
|
78
|
+
"Sync workflows",
|
|
79
|
+
)
|
|
80
|
+
if target_branch != default_branch and not self.gh_client.get_branch_info(target_branch):
|
|
81
|
+
self._require(self.gh_client.create_branch(target_branch, default_branch), "Create target branch")
|
|
82
|
+
if scenario_files:
|
|
83
|
+
additions = {path: self._get_content(content) for path, content in scenario_files.items()}
|
|
84
|
+
self._require(
|
|
85
|
+
self.gh_client.batch_sync(additions, [], "provision scenario files", target_branch), "Sync scenario"
|
|
86
|
+
)
|
|
87
|
+
elif target_branch != default_branch:
|
|
88
|
+
self._require(
|
|
89
|
+
self.gh_client.put_file(".scanner-probe", "probe\n", "scanner probe commit", target_branch),
|
|
90
|
+
"Create PR comparison commit",
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
def _get_content(self, content_or_path):
|
|
94
|
+
"""Reads file content or returns an inline value."""
|
|
95
|
+
content = content_or_path
|
|
96
|
+
is_binary = False
|
|
97
|
+
if isinstance(content_or_path, str) and os.path.exists(content_or_path):
|
|
98
|
+
with open(content_or_path, "rb") as f:
|
|
99
|
+
content = f.read()
|
|
100
|
+
try:
|
|
101
|
+
content = content.decode("utf-8")
|
|
102
|
+
except UnicodeDecodeError:
|
|
103
|
+
is_binary = True
|
|
104
|
+
|
|
105
|
+
if is_binary:
|
|
106
|
+
return content.decode("latin-1")
|
|
107
|
+
return content
|
|
108
|
+
|
|
109
|
+
def teardown(self):
|
|
110
|
+
"""Clean up a repository created by this provisioner, retaining ownership on failure."""
|
|
111
|
+
if not self._owns_repo:
|
|
112
|
+
return
|
|
113
|
+
self._require(self.gh_client.delete_owned_repo(), "Delete owned repository")
|
|
114
|
+
self._owns_repo = False
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import os
|
|
2
|
+
|
|
3
|
+
from .gh_client import GitHubClient
|
|
4
|
+
from .provisioner import RepoProvisioner
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def setup_attacker_fork(scenario, owner_client):
|
|
8
|
+
"""Create a fresh attacker fork and record ownership before subsequent setup."""
|
|
9
|
+
token = os.environ.get("ATTACKER_GITHUB_TOKEN")
|
|
10
|
+
if not token:
|
|
11
|
+
raise ValueError("Required GitHub actor is unavailable: attacker")
|
|
12
|
+
client = GitHubClient(token=token, actor="attacker", record_event=owner_client.record_event)
|
|
13
|
+
login = client.get_authenticated_user_login()
|
|
14
|
+
client.repo_name = f"{login}/{owner_client.repository.name}"
|
|
15
|
+
RepoProvisioner._require(client.fork_repo(owner_client.repo_name), "Create attacker fork")
|
|
16
|
+
scenario._attacker_fork_client = client
|
|
17
|
+
scenario.runtime_state.update(attacker_login=login, fork_full_name=client.repo_name)
|
|
18
|
+
client.wait_until_ready()
|
|
19
|
+
RepoProvisioner._require(client.create_branch(scenario.branch, client.get_default_branch()), "Create attacker branch")
|
|
20
|
+
additions = {}
|
|
21
|
+
for path, local_path in scenario.get_required_files().items():
|
|
22
|
+
with open(local_path) as handle:
|
|
23
|
+
additions[path] = handle.read()
|
|
24
|
+
RepoProvisioner._require(
|
|
25
|
+
client.batch_sync(additions, [], "provision attacker fixtures", scenario.branch), "Sync attacker fixtures"
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def teardown_attacker_fork(scenario):
|
|
30
|
+
client = getattr(scenario, "_attacker_fork_client", None)
|
|
31
|
+
if client is not None:
|
|
32
|
+
RepoProvisioner._require(client.delete_owned_repo(), "Delete attacker fork")
|
|
33
|
+
scenario._attacker_fork_client = None
|
gitinject/utils/types.py
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
from enum import Enum
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class ScenarioType(str, Enum):
|
|
5
|
+
BENIGN = "benign"
|
|
6
|
+
MALICIOUS = "malicious"
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class GitHubEvent(str, Enum):
|
|
10
|
+
PULL_REQUEST = "pull_request"
|
|
11
|
+
ISSUES = "issues"
|
|
12
|
+
ISSUE_COMMENT = "issue_comment"
|
|
13
|
+
PULL_REQUEST_REVIEW = "pull_request_review"
|
|
14
|
+
PULL_REQUEST_REVIEW_COMMENT = "pull_request_review_comment"
|
|
15
|
+
PUSH = "push"
|
|
16
|
+
WORKFLOW_DISPATCH = "workflow_dispatch"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class WorkflowCategory(str, Enum):
|
|
20
|
+
CODE_MAINTENANCE = "code-maintenance"
|
|
21
|
+
SECURITY_COMPLIANCE = "security-compliance"
|
|
22
|
+
TRIAGE_MANAGEMENT = "triage-management"
|
|
23
|
+
CONTENT_DOCS = "content-docs"
|
|
24
|
+
CODE_REVIEW = "code-review"
|
|
25
|
+
SUPPORT_INTELLIGENCE = "support-intelligence"
|
|
26
|
+
REPORTING_ANALYTICS = "reporting-analytics"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class AIProvider(str, Enum):
|
|
30
|
+
GOOGLE_GEMINI = "google-gemini"
|
|
31
|
+
ANTHROPIC_CLAUDE = "anthropic-claude"
|
|
32
|
+
OPENAI_CODEX = "openai-codex"
|
|
33
|
+
CURSOR = "cursor"
|
|
34
|
+
GITHUB_COPILOT = "github-copilot"
|
|
35
|
+
CLINE = "cline"
|
|
36
|
+
OPEN_CODE = "open-code"
|
|
37
|
+
OPEN_HANDS = "open-hands"
|
|
38
|
+
AMAZON_Q = "amazon-q"
|
|
39
|
+
GOOSE = "goose"
|
|
40
|
+
DEVIN = "devin"
|
|
41
|
+
CISCO_AI = "cisco-ai"
|
|
42
|
+
AIDER = "aider"
|
|
43
|
+
OPENROUTER = "openrouter"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class DefenseLevel(str, Enum):
|
|
47
|
+
BASELINE = "baseline"
|
|
48
|
+
HARDENED = "hardened"
|
|
49
|
+
SANDBOXED = "sandboxed"
|
|
File without changes
|