gitinject 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (164) hide show
  1. gitinject/__init__.py +1 -0
  2. gitinject/__main__.py +5 -0
  3. gitinject/analyzer.py +67 -0
  4. gitinject/attacks/__init__.py +27 -0
  5. gitinject/attacks/autoinject.py +165 -0
  6. gitinject/attacks/base.py +33 -0
  7. gitinject/attacks/static.py +25 -0
  8. gitinject/cli.py +768 -0
  9. gitinject/data/research/scenarios/claude_skills_injection.md +96 -0
  10. gitinject/data/research/scenarios/cline_issue_body_injection.md +93 -0
  11. gitinject/data/research/scenarios/codex_agents_md_injection.md +109 -0
  12. gitinject/data/research/scenarios/dos_request_flood.md +133 -0
  13. gitinject/data/research/scenarios/dropped/ci_log_injection_workflow_poisoning.md +138 -0
  14. gitinject/data/research/scenarios/dropped/claude_md_instructions_injection.md +158 -0
  15. gitinject/data/research/scenarios/dropped/supply_chain_token_pivot.md +101 -0
  16. gitinject/data/research/scenarios/gemini_api_key_exfiltration.md +68 -0
  17. gitinject/data/research/scenarios/gemini_api_key_exfiltration_replication.md +0 -0
  18. gitinject/data/research/scenarios/gemini_md_instructions_injection.md +131 -0
  19. gitinject/data/research/scenarios/nsfw_api_key_block.md +114 -0
  20. gitinject/data/research/scenarios/pr_token_exfiltration_via_git_config.md +117 -0
  21. gitinject/data/research/scenarios/supply_chain_malicious_code.md +135 -0
  22. gitinject/evaluators.py +169 -0
  23. gitinject/evidence.py +72 -0
  24. gitinject/gl_runner.py +163 -0
  25. gitinject/resources.py +30 -0
  26. gitinject/run_context.py +72 -0
  27. gitinject/run_record.py +113 -0
  28. gitinject/runner.py +1024 -0
  29. gitinject/scanner/__init__.py +22 -0
  30. gitinject/scanner/baselines/__init__.py +0 -0
  31. gitinject/scanner/baselines/actionlint_runner.py +44 -0
  32. gitinject/scanner/baselines/zizmor_runner.py +35 -0
  33. gitinject/scanner/diagnostics.py +174 -0
  34. gitinject/scanner/hypothesis_generator.py +164 -0
  35. gitinject/scanner/live_validator.py +284 -0
  36. gitinject/scanner/llm_ranker.py +120 -0
  37. gitinject/scanner/memory.py +239 -0
  38. gitinject/scanner/primitives.py +236 -0
  39. gitinject/scanner/prompt_extractor.py +134 -0
  40. gitinject/scanner/recipe_scenario.py +203 -0
  41. gitinject/scanner/report_generator.py +230 -0
  42. gitinject/scanner/types.py +179 -0
  43. gitinject/scenario_base.py +147 -0
  44. gitinject/scenario_loader.py +77 -0
  45. gitinject/scenarios/__init__.py +0 -0
  46. gitinject/scenarios/benign/documentation_out_of_sync/contents/math_utils.py +3 -0
  47. gitinject/scenarios/benign/documentation_out_of_sync/contents/pydantic/v1/utils.py +21 -0
  48. gitinject/scenarios/benign/documentation_out_of_sync/scenario.py +57 -0
  49. gitinject/scenarios/benign/issue_deduplication_utility/scenario.py +76 -0
  50. gitinject/scenarios/benign/repository_onboarding_qa/contents/README.md +3 -0
  51. gitinject/scenarios/benign/repository_onboarding_qa/scenario.py +52 -0
  52. gitinject/scenarios/benign/secrets_false_positive/contents/tests/test_auth.py +14 -0
  53. gitinject/scenarios/benign/secrets_false_positive/scenario.py +58 -0
  54. gitinject/scenarios/benign/test_coverage_gap/contents/src/requests/auth_extra.py +33 -0
  55. gitinject/scenarios/benign/test_coverage_gap/scenario.py +56 -0
  56. gitinject/scenarios/benign/vulnerability_fix_verification/contents/src/flask/utils.py +15 -0
  57. gitinject/scenarios/benign/vulnerability_fix_verification/scenario.py +57 -0
  58. gitinject/scenarios/benign/vulnerable_code_review/contents/examples/tutorial/flaskr/db_utils.py +18 -0
  59. gitinject/scenarios/benign/vulnerable_code_review/scenario.py +55 -0
  60. gitinject/scenarios/benign/weekly_activity_summary/scenario.py +83 -0
  61. gitinject/scenarios/malicious/agents_md_approval_manipulation/contents/AGENTS.md +25 -0
  62. gitinject/scenarios/malicious/agents_md_approval_manipulation/contents/src/flask/csrf.py +151 -0
  63. gitinject/scenarios/malicious/agents_md_approval_manipulation/contents/tests/test_csrf.py +71 -0
  64. gitinject/scenarios/malicious/agents_md_approval_manipulation/scenario.py +121 -0
  65. gitinject/scenarios/malicious/claude_md_instructions_injection/contents/CLAUDE.md +23 -0
  66. gitinject/scenarios/malicious/claude_md_instructions_injection/contents/scripts/deploy_check.sh +29 -0
  67. gitinject/scenarios/malicious/claude_md_instructions_injection/scenario.py +93 -0
  68. gitinject/scenarios/malicious/cline_issue_body_injection/scenario.py +132 -0
  69. gitinject/scenarios/malicious/codex_agents_md_injection/contents/AGENTS.md +36 -0
  70. gitinject/scenarios/malicious/codex_agents_md_injection/contents/scripts/utils.py +12 -0
  71. gitinject/scenarios/malicious/codex_agents_md_injection/scenario.py +128 -0
  72. gitinject/scenarios/malicious/dos_request_flood/scenario.py +80 -0
  73. gitinject/scenarios/malicious/gemini_md_approval_manipulation/contents/src/flask/csrf.py +151 -0
  74. gitinject/scenarios/malicious/gemini_md_approval_manipulation/contents/tests/test_csrf.py +71 -0
  75. gitinject/scenarios/malicious/gemini_md_approval_manipulation/scenario.py +121 -0
  76. gitinject/scenarios/malicious/gemini_md_instructions_injection/contents/scripts/deploy_check.sh +38 -0
  77. gitinject/scenarios/malicious/gemini_md_instructions_injection/scenario.py +105 -0
  78. gitinject/scenarios/malicious/issue_onboarding_secret_exfiltration/scenario.py +76 -0
  79. gitinject/scenarios/malicious/pr_token_exfiltration_via_git_config/contents/scripts/deploy_check.sh +7 -0
  80. gitinject/scenarios/malicious/pr_token_exfiltration_via_git_config/scenario.py +113 -0
  81. gitinject/scenarios/malicious/supply_chain_malicious_code/contents/CLAUDE.md +25 -0
  82. gitinject/scenarios/malicious/supply_chain_malicious_code/contents/src/flask/csrf.py +151 -0
  83. gitinject/scenarios/malicious/supply_chain_malicious_code/contents/tests/test_csrf.py +71 -0
  84. gitinject/scenarios/malicious/supply_chain_malicious_code/scenario.py +123 -0
  85. gitinject/scenarios/malicious/unauthorized_pr_approval/contents/django/utils/crypto.py +13 -0
  86. gitinject/scenarios/malicious/unauthorized_pr_approval/scenario.py +57 -0
  87. gitinject/simulator.py +89 -0
  88. gitinject/utils/__init__.py +0 -0
  89. gitinject/utils/gh_client.py +628 -0
  90. gitinject/utils/gl_client.py +132 -0
  91. gitinject/utils/gl_provisioner.py +83 -0
  92. gitinject/utils/llm.py +205 -0
  93. gitinject/utils/provisioner.py +114 -0
  94. gitinject/utils/scenario_resources.py +33 -0
  95. gitinject/utils/types.py +49 -0
  96. gitinject/workflows/__init__.py +0 -0
  97. gitinject/workflows/claude-ci-auto-fix/contents/.github/workflows/main.yml +107 -0
  98. gitinject/workflows/claude-ci-auto-fix/metadata.json +10 -0
  99. gitinject/workflows/claude-general/contents/.github/workflows/main.yml +58 -0
  100. gitinject/workflows/claude-general/metadata.json +10 -0
  101. gitinject/workflows/claude-gitlab-mr-review/contents/.gitlab-ci.yml +36 -0
  102. gitinject/workflows/claude-gitlab-mr-review/metadata.json +11 -0
  103. gitinject/workflows/claude-issue-deduplication/contents/.github/workflows/main.yml +66 -0
  104. gitinject/workflows/claude-issue-deduplication/metadata.json +10 -0
  105. gitinject/workflows/claude-issue-triage/contents/.github/workflows/main.yml +34 -0
  106. gitinject/workflows/claude-issue-triage/metadata.json +10 -0
  107. gitinject/workflows/claude-manual-analysis/contents/.github/workflows/main.yml +42 -0
  108. gitinject/workflows/claude-manual-analysis/metadata.json +10 -0
  109. gitinject/workflows/claude-pr-review/contents/.github/workflows/main.yml +77 -0
  110. gitinject/workflows/claude-pr-review/metadata.json +10 -0
  111. gitinject/workflows/claude-pr-review-authors/contents/.github/workflows/main.yml +48 -0
  112. gitinject/workflows/claude-pr-review-authors/metadata.json +10 -0
  113. gitinject/workflows/claude-pr-review-paths/contents/.github/workflows/main.yml +49 -0
  114. gitinject/workflows/claude-pr-review-paths/metadata.json +10 -0
  115. gitinject/workflows/claude-test-analysis/contents/.github/workflows/main.yml +114 -0
  116. gitinject/workflows/claude-test-analysis/metadata.json +10 -0
  117. gitinject/workflows/cline-assistant/contents/.github/workflows/main.yml +87 -0
  118. gitinject/workflows/cline-assistant/contents/git-scripts/analyze-issue.sh +43 -0
  119. gitinject/workflows/cline-assistant/metadata.json +10 -0
  120. gitinject/workflows/codex-pr-review/contents/.github/workflows/main.yml +73 -0
  121. gitinject/workflows/codex-pr-review/metadata.json +10 -0
  122. gitinject/workflows/copilot-ci-doctor/contents/.github/workflows/ci-doctor.yml +1161 -0
  123. gitinject/workflows/copilot-ci-doctor/metadata.json +10 -0
  124. gitinject/workflows/copilot-lean-squad/contents/.github/workflows/lean-squad.yml +1313 -0
  125. gitinject/workflows/copilot-lean-squad/metadata.json +10 -0
  126. gitinject/workflows/copilot-malicious-scan/contents/.github/workflows/daily-malicious-code-scan.yml +899 -0
  127. gitinject/workflows/copilot-malicious-scan/metadata.json +10 -0
  128. gitinject/workflows/copilot-repo-assist/contents/.github/workflows/repo-assist.yml +1503 -0
  129. gitinject/workflows/copilot-repo-assist/metadata.json +10 -0
  130. gitinject/workflows/copilot-wiki-writer/contents/.github/workflows/agentic-wiki-writer.yml +1316 -0
  131. gitinject/workflows/copilot-wiki-writer/metadata.json +10 -0
  132. gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-invoke.yml +122 -0
  133. gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-plan-execute.yml +130 -0
  134. gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-review.yml +118 -0
  135. gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-scheduled-triage.yml +220 -0
  136. gitinject/workflows/gemini-assistant/contents/.github/workflows/gemini-triage.yml +160 -0
  137. gitinject/workflows/gemini-assistant/contents/.github/workflows/main.yml +220 -0
  138. gitinject/workflows/gemini-assistant/metadata.json +10 -0
  139. gitinject/workflows/gemini-assistant-original/AWESOME.md +118 -0
  140. gitinject/workflows/gemini-assistant-original/CONFIGURATION.md +162 -0
  141. gitinject/workflows/gemini-assistant-original/README.md +93 -0
  142. gitinject/workflows/gemini-assistant-original/gemini-assistant/README.md +192 -0
  143. gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-invoke.toml +94 -0
  144. gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-invoke.yml +131 -0
  145. gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-plan-execute.toml +100 -0
  146. gitinject/workflows/gemini-assistant-original/gemini-assistant/gemini-plan-execute.yml +139 -0
  147. gitinject/workflows/gemini-assistant-original/gemini-dispatch/README.md +49 -0
  148. gitinject/workflows/gemini-assistant-original/gemini-dispatch/gemini-dispatch.yml +221 -0
  149. gitinject/workflows/gemini-assistant-original/issue-triage/README.md +190 -0
  150. gitinject/workflows/gemini-assistant-original/issue-triage/gemini-scheduled-triage.toml +96 -0
  151. gitinject/workflows/gemini-assistant-original/issue-triage/gemini-scheduled-triage.yml +223 -0
  152. gitinject/workflows/gemini-assistant-original/issue-triage/gemini-triage.toml +32 -0
  153. gitinject/workflows/gemini-assistant-original/issue-triage/gemini-triage.yml +167 -0
  154. gitinject/workflows/gemini-assistant-original/metadata.json +10 -0
  155. gitinject/workflows/gemini-assistant-original/pr-review/README.md +337 -0
  156. gitinject/workflows/gemini-assistant-original/pr-review/gemini-review.toml +176 -0
  157. gitinject/workflows/gemini-assistant-original/pr-review/gemini-review.yml +119 -0
  158. gitinject/workflows/opencode-pr-review/contents/.github/workflows/main.yml +28 -0
  159. gitinject/workflows/opencode-pr-review/metadata.json +10 -0
  160. gitinject-0.1.0.dist-info/METADATA +128 -0
  161. gitinject-0.1.0.dist-info/RECORD +164 -0
  162. gitinject-0.1.0.dist-info/WHEEL +4 -0
  163. gitinject-0.1.0.dist-info/entry_points.txt +2 -0
  164. gitinject-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,132 @@
1
+ import os
2
+ import time
3
+
4
+ import click
5
+ import gitlab
6
+ import gitlab.exceptions
7
+
8
+
9
+ class GitLabClient:
10
+ """Thin wrapper around python-gitlab for benchmark provisioning and evaluation."""
11
+
12
+ def __init__(self, token: str | None = None, url: str = "https://gitlab.com"):
13
+ token = token or os.environ.get("GITLAB_TOKEN")
14
+ if not token:
15
+ click.echo(
16
+ click.style("Error: GITLAB_TOKEN not set.", fg="red"),
17
+ err=True,
18
+ )
19
+ raise RuntimeError("Missing GitLab authentication")
20
+ self.gl = gitlab.Gitlab(url, private_token=token)
21
+ self.project = None
22
+
23
+ def create_project(
24
+ self,
25
+ name: str,
26
+ namespace_path: str | None = None,
27
+ visibility: str = "public",
28
+ ):
29
+ kwargs: dict = {"name": name, "visibility": visibility, "initialize_with_readme": True}
30
+ if namespace_path:
31
+ results = self.gl.namespaces.list(search=namespace_path)
32
+ if results:
33
+ kwargs["namespace_id"] = results[0].id
34
+ self.project = self.gl.projects.create(kwargs)
35
+ return self.project
36
+
37
+ def delete_project(self) -> None:
38
+ if self.project:
39
+ try:
40
+ self.project.delete()
41
+ except gitlab.exceptions.GitlabDeleteError as e:
42
+ click.echo(click.style(f"Warning: failed to delete project: {e}", fg="yellow"))
43
+ self.project = None
44
+
45
+ def push_files(self, branch: str, files: dict[str, str], commit_message: str = "add files") -> None:
46
+ """Batch-commit multiple files to a branch in a single commit."""
47
+ if not files:
48
+ return
49
+ existing = set()
50
+ try:
51
+ tree = self.project.repository_tree(ref=branch, recursive=True, all=True)
52
+ existing = {item["path"] for item in tree}
53
+ except gitlab.exceptions.GitlabGetError:
54
+ pass
55
+
56
+ actions = []
57
+ for path, content in files.items():
58
+ action = "update" if path in existing else "create"
59
+ actions.append({"action": action, "file_path": path, "content": content})
60
+
61
+ self.project.commits.create(
62
+ {
63
+ "branch": branch,
64
+ "commit_message": commit_message,
65
+ "actions": actions,
66
+ }
67
+ )
68
+
69
+ def create_branch(self, name: str, ref: str = "main") -> None:
70
+ try:
71
+ self.project.branches.create({"branch": name, "ref": ref})
72
+ except gitlab.exceptions.GitlabCreateError:
73
+ pass # Branch already exists
74
+
75
+ def create_merge_request(
76
+ self,
77
+ source_branch: str,
78
+ target_branch: str,
79
+ title: str,
80
+ description: str = "",
81
+ ) -> int:
82
+ mr = self.project.mergerequests.create(
83
+ {
84
+ "source_branch": source_branch,
85
+ "target_branch": target_branch,
86
+ "title": title,
87
+ "description": description,
88
+ }
89
+ )
90
+ return mr.iid
91
+
92
+ def get_mr_notes(self, mr_iid: int) -> list[str]:
93
+ mr = self.project.mergerequests.get(mr_iid)
94
+ notes = mr.notes.list(all=True)
95
+ return [n.body for n in notes if not n.system]
96
+
97
+ def get_mr_pipelines(self, mr_iid: int) -> list:
98
+ mr = self.project.mergerequests.get(mr_iid)
99
+ return mr.pipelines.list()
100
+
101
+ def wait_for_pipeline(self, pipeline_id: int, timeout_s: int = 600, poll_interval: int = 15) -> str:
102
+ deadline = time.time() + timeout_s
103
+ while time.time() < deadline:
104
+ pipeline = self.project.pipelines.get(pipeline_id)
105
+ if pipeline.status not in ("created", "waiting_for_resource", "preparing", "pending", "running", "scheduled"):
106
+ return pipeline.status
107
+ click.echo(f" Pipeline {pipeline_id} status: {pipeline.status}...")
108
+ time.sleep(poll_interval)
109
+ return "timeout"
110
+
111
+ def set_variable(self, key: str, value: str, masked: bool = True, protected: bool = False) -> None:
112
+ try:
113
+ var = self.project.variables.get(key)
114
+ var.value = value
115
+ var.save()
116
+ except gitlab.exceptions.GitlabGetError:
117
+ self.project.variables.create(
118
+ {
119
+ "key": key,
120
+ "value": value,
121
+ "masked": masked,
122
+ "protected": protected,
123
+ }
124
+ )
125
+
126
+ def get_project_url(self) -> str | None:
127
+ return self.project.web_url if self.project else None
128
+
129
+ def get_default_branch(self) -> str:
130
+ if self.project:
131
+ return self.project.default_branch or "main"
132
+ return "main"
@@ -0,0 +1,83 @@
1
+ import os
2
+ import time
3
+
4
+ import click
5
+
6
+ from .gl_client import GitLabClient
7
+
8
+
9
+ class GitLabProvisioner:
10
+ """Handles lifecycle of a GitLab project for benchmarking."""
11
+
12
+ def __init__(self, gl_client: GitLabClient):
13
+ self.gl_client = gl_client
14
+
15
+ def provision(
16
+ self,
17
+ project_name: str,
18
+ workflow_dir: str,
19
+ required_files: dict | None = None,
20
+ branch: str | None = None,
21
+ variables: dict | None = None,
22
+ ) -> None:
23
+ click.echo(f"Creating GitLab project {project_name}...")
24
+ self.gl_client.create_project(project_name)
25
+
26
+ # GitLab needs a moment after initialize_with_readme before branches are writable
27
+ time.sleep(5)
28
+
29
+ default_branch = self.gl_client.get_default_branch()
30
+
31
+ # Collect workflow files from contents/
32
+ workflow_files: dict[str, str] = {}
33
+ contents_dir = os.path.join(workflow_dir, "contents")
34
+ if os.path.isdir(contents_dir):
35
+ for root, _, filenames in os.walk(contents_dir):
36
+ for filename in filenames:
37
+ abs_path = os.path.join(root, filename)
38
+ rel_path = os.path.relpath(abs_path, contents_dir)
39
+ workflow_files[rel_path] = self._read_file(abs_path)
40
+
41
+ if workflow_files:
42
+ click.echo(f"Pushing workflow files to {default_branch}...")
43
+ self.gl_client.push_files(default_branch, workflow_files, "provision: add CI workflow")
44
+
45
+ # Create target branch for scenario files (if different from default)
46
+ target_branch = branch or default_branch
47
+ if target_branch != default_branch:
48
+ click.echo(f"Creating branch {target_branch}...")
49
+ self.gl_client.create_branch(target_branch, ref=default_branch)
50
+
51
+ # Push scenario-specific files to target branch
52
+ if required_files:
53
+ scenario_files: dict[str, str] = {}
54
+ for repo_path, content_or_path in required_files.items():
55
+ if isinstance(content_or_path, str) and os.path.exists(content_or_path):
56
+ scenario_files[repo_path] = self._read_file(content_or_path)
57
+ else:
58
+ scenario_files[repo_path] = str(content_or_path)
59
+ if scenario_files:
60
+ click.echo(f"Pushing scenario files to {target_branch}...")
61
+ self.gl_client.push_files(target_branch, scenario_files, "provision: add scenario files")
62
+
63
+ # Set CI/CD variables
64
+ if variables:
65
+ for key, value in variables.items():
66
+ if value:
67
+ click.echo(f"Setting CI/CD variable '{key}'...")
68
+ self.gl_client.set_variable(key, value, masked=True)
69
+
70
+ click.echo(f"Project ready: {self.gl_client.get_project_url()}")
71
+
72
+ def teardown(self) -> None:
73
+ click.echo("Deleting GitLab project...")
74
+ self.gl_client.delete_project()
75
+ click.echo("Project deleted.")
76
+
77
+ def _read_file(self, path: str) -> str:
78
+ with open(path, "rb") as f:
79
+ raw = f.read()
80
+ try:
81
+ return raw.decode("utf-8")
82
+ except UnicodeDecodeError:
83
+ return raw.decode("latin-1")
gitinject/utils/llm.py ADDED
@@ -0,0 +1,205 @@
1
+ from __future__ import annotations
2
+
3
+ import os
4
+ from contextvars import ContextVar
5
+ from dataclasses import dataclass
6
+
7
+
8
+ class LLMError(Exception):
9
+ pass
10
+
11
+
12
+ @dataclass
13
+ class LLMResponse:
14
+ text: str
15
+ input_tokens: int
16
+ output_tokens: int
17
+ model: str
18
+
19
+
20
+ _usage_log: ContextVar[list[LLMResponse] | None] = ContextVar("usage_log", default=None)
21
+
22
+
23
+ class track_usage:
24
+ """Context manager that captures every LLMResponse produced within its scope.
25
+
26
+ Usage:
27
+ with track_usage() as log:
28
+ call_llm(...)
29
+ call_llm(...)
30
+ # log is a list[LLMResponse]
31
+ """
32
+
33
+ def __enter__(self) -> list[LLMResponse]:
34
+ self._log: list[LLMResponse] = []
35
+ self._token = _usage_log.set(self._log)
36
+ return self._log
37
+
38
+ def __exit__(self, *exc):
39
+ _usage_log.reset(self._token)
40
+
41
+
42
+ def _record(resp: LLMResponse) -> None:
43
+ log = _usage_log.get()
44
+ if log is not None:
45
+ log.append(resp)
46
+
47
+
48
+ def _parse_model(model: str) -> tuple[str, str]:
49
+ """Parse 'provider/model-name' into (provider, model_name).
50
+
51
+ Supported providers: anthropic, google, openai, openrouter.
52
+ Bare model names fall back to heuristic detection.
53
+ """
54
+ if model.startswith("openrouter/"):
55
+ return "openrouter", model[len("openrouter/") :]
56
+ if "/" in model:
57
+ provider, model_name = model.split("/", 1)
58
+ return provider, model_name
59
+ if model.startswith("claude"):
60
+ return "anthropic", model
61
+ if model.startswith("gemini"):
62
+ return "google", model
63
+ return "openai", model
64
+
65
+
66
+ def call_llm(
67
+ model: str,
68
+ system: str,
69
+ user: str,
70
+ max_tokens: int = 2048,
71
+ temperature: float | None = None,
72
+ ) -> LLMResponse:
73
+ """Call an LLM and return an LLMResponse with text and token usage.
74
+
75
+ model format: 'provider/model-name' e.g. 'anthropic/claude-sonnet-4-6',
76
+ 'google/gemini-2.5-flash', 'openai/gpt-4o', 'openrouter/openai/gpt-4o'.
77
+ Bare model names are accepted with heuristic provider detection.
78
+ """
79
+ provider, model_name = _parse_model(model)
80
+
81
+ try:
82
+ if provider == "anthropic":
83
+ resp = _call_anthropic(model_name, system, user, max_tokens, temperature)
84
+ elif provider in ("google", "gemini"):
85
+ resp = _call_google(model_name, system, user, max_tokens, temperature)
86
+ elif provider == "openrouter":
87
+ resp = _call_openai(
88
+ model_name,
89
+ system,
90
+ user,
91
+ max_tokens,
92
+ temperature,
93
+ base_url="https://openrouter.ai/api/v1",
94
+ api_key=os.environ.get("OPENROUTER_API_KEY", ""),
95
+ )
96
+ else:
97
+ resp = _call_openai(model_name, system, user, max_tokens, temperature)
98
+ except LLMError:
99
+ raise
100
+ except Exception as e:
101
+ raise LLMError(str(e)) from e
102
+
103
+ _record(resp)
104
+ return resp
105
+
106
+
107
+ def _call_anthropic(model: str, system: str, user: str, max_tokens: int, temperature: float | None) -> LLMResponse:
108
+ import anthropic
109
+
110
+ try:
111
+ client = anthropic.Anthropic()
112
+ kwargs: dict = dict(
113
+ model=model,
114
+ max_tokens=max_tokens,
115
+ messages=[{"role": "user", "content": user}],
116
+ )
117
+ if system:
118
+ kwargs["system"] = system
119
+ if temperature is not None:
120
+ kwargs["temperature"] = temperature
121
+ response = client.messages.create(**kwargs)
122
+ usage = getattr(response, "usage", None)
123
+ return LLMResponse(
124
+ text=response.content[0].text,
125
+ input_tokens=int(getattr(usage, "input_tokens", 0) or 0),
126
+ output_tokens=int(getattr(usage, "output_tokens", 0) or 0),
127
+ model=model,
128
+ )
129
+ except anthropic.AuthenticationError as e:
130
+ raise LLMError(f"Anthropic authentication failed: {e}") from e
131
+ except anthropic.PermissionDeniedError as e:
132
+ raise LLMError(f"Anthropic access denied (credits exhausted?): {e}") from e
133
+ except anthropic.RateLimitError as e:
134
+ raise LLMError(f"Anthropic rate limit: {e}") from e
135
+
136
+
137
+ def _call_google(model: str, system: str, user: str, max_tokens: int, temperature: float | None) -> LLMResponse:
138
+ from google import genai
139
+
140
+ api_key = os.environ.get("GEMINI_API_KEY")
141
+ if not api_key:
142
+ raise LLMError("GEMINI_API_KEY is not set")
143
+ client = genai.Client(api_key=api_key)
144
+ config: dict = {"max_output_tokens": max_tokens}
145
+ if system:
146
+ config["system_instruction"] = system
147
+ if temperature is not None:
148
+ config["temperature"] = temperature
149
+ response = client.models.generate_content(model=model, contents=user, config=config)
150
+ text = response.text
151
+ if text is None:
152
+ raise LLMError(f"Google model returned empty response (blocked?): {response}")
153
+ usage = getattr(response, "usage_metadata", None)
154
+ return LLMResponse(
155
+ text=text,
156
+ input_tokens=int(getattr(usage, "prompt_token_count", 0) or 0),
157
+ output_tokens=int(getattr(usage, "candidates_token_count", 0) or 0),
158
+ model=model,
159
+ )
160
+
161
+
162
+ def _call_openai(
163
+ model: str,
164
+ system: str,
165
+ user: str,
166
+ max_tokens: int,
167
+ temperature: float | None,
168
+ base_url: str | None = None,
169
+ api_key: str | None = None,
170
+ ) -> LLMResponse:
171
+ from openai import AuthenticationError, OpenAI, RateLimitError
172
+
173
+ kwargs: dict = {}
174
+ if base_url:
175
+ kwargs["base_url"] = base_url
176
+ if api_key:
177
+ kwargs["api_key"] = api_key
178
+ client = OpenAI(**kwargs)
179
+
180
+ messages = []
181
+ if system:
182
+ messages.append({"role": "system", "content": system})
183
+ messages.append({"role": "user", "content": user})
184
+
185
+ request: dict = dict(model=model, messages=messages, max_completion_tokens=max_tokens)
186
+ if temperature is not None:
187
+ request["temperature"] = temperature
188
+
189
+ try:
190
+ response = client.chat.completions.create(**request)
191
+ content = response.choices[0].message.content
192
+ if not content:
193
+ finish_reason = response.choices[0].finish_reason if response.choices else "unknown"
194
+ raise LLMError(f"OpenAI returned empty content (finish_reason={finish_reason}, model={model})")
195
+ usage = getattr(response, "usage", None)
196
+ return LLMResponse(
197
+ text=content,
198
+ input_tokens=int(getattr(usage, "prompt_tokens", 0) or 0),
199
+ output_tokens=int(getattr(usage, "completion_tokens", 0) or 0),
200
+ model=model,
201
+ )
202
+ except AuthenticationError as e:
203
+ raise LLMError(f"OpenAI authentication failed: {e}") from e
204
+ except RateLimitError as e:
205
+ raise LLMError(f"OpenAI rate limit: {e}") from e
@@ -0,0 +1,114 @@
1
+ import os
2
+
3
+ from .gh_client import GitHubClient
4
+
5
+
6
+ class ProvisioningError(RuntimeError):
7
+ pass
8
+
9
+
10
+ class RepoProvisioner:
11
+ """Handles the physical setup of a GitHub repository for benchmarking."""
12
+
13
+ def __init__(self, gh_client: GitHubClient):
14
+ self.gh_client = gh_client
15
+ self._owns_repo = False
16
+
17
+ @staticmethod
18
+ def _require(result, operation):
19
+ success, error = result
20
+ if not success:
21
+ raise ProvisioningError(f"{operation}: {error}")
22
+
23
+ def provision(
24
+ self,
25
+ workflow_dir: str,
26
+ required_files: dict = None,
27
+ branch: str = None,
28
+ template_repo: str = None,
29
+ secrets: dict = None,
30
+ variables: dict = None,
31
+ ):
32
+ """Create and configure a fresh repository, failing on incomplete setup."""
33
+ if self._owns_repo:
34
+ raise ProvisioningError("A previously created repository still needs cleanup")
35
+ creation = self.gh_client.fork_repo(template_repo) if template_repo else self.gh_client.create_repo(public=True)
36
+ self._require(creation, "Create repository")
37
+ self._owns_repo = True
38
+ if template_repo:
39
+ self.gh_client.wait_until_ready()
40
+ else:
41
+ self._require(
42
+ self.gh_client.put_file(
43
+ "README.md", "# Benchmark Repository\nGenerated by AI Benchmark Suite.", "initial commit", "main"
44
+ ),
45
+ "Initialize repository",
46
+ )
47
+ repo_info = self.gh_client.get_repo_info()
48
+ if not repo_info:
49
+ raise ProvisioningError("Created repository is not readable")
50
+ default_branch = repo_info.get("defaultBranchRef", {}).get("name") or "main"
51
+ target_branch = branch or default_branch
52
+ self._require(self.gh_client.enable_actions(), "Enable Actions")
53
+ self._require(self.gh_client.set_fork_pr_approval_policy(), "Set fork approval policy")
54
+ self._require(self.gh_client.enable_issues(), "Enable issues")
55
+ for operation, values in ((self.gh_client.set_secret, secrets), (self.gh_client.set_variable, variables)):
56
+ for name, value in (values or {}).items():
57
+ if value is None or value == "":
58
+ raise ProvisioningError(f"Empty configuration value: {name}")
59
+ self._require(operation(name, value), f"Set configuration {name}")
60
+ contents_dir = os.path.join(workflow_dir, "contents")
61
+ workflow_files = {}
62
+ if os.path.isdir(contents_dir):
63
+ for root, _, filenames in os.walk(contents_dir):
64
+ for filename in filenames:
65
+ local_path = os.path.join(root, filename)
66
+ workflow_files[os.path.relpath(local_path, contents_dir)] = local_path
67
+ elif os.path.isdir(workflow_dir):
68
+ for filename in os.listdir(workflow_dir):
69
+ if filename.endswith((".yml", ".yaml")):
70
+ workflow_files[f".github/workflows/{filename}"] = os.path.join(workflow_dir, filename)
71
+ scenario_files = required_files or {}
72
+ for path in scenario_files:
73
+ if path in workflow_files:
74
+ raise ProvisioningError(f"File defined by both workflow and scenario: {path}")
75
+ additions = {path: self._get_content(content) for path, content in workflow_files.items()}
76
+ self._require(
77
+ self.gh_client.batch_sync(additions, [".github/workflows/"], "provision workflows", default_branch),
78
+ "Sync workflows",
79
+ )
80
+ if target_branch != default_branch and not self.gh_client.get_branch_info(target_branch):
81
+ self._require(self.gh_client.create_branch(target_branch, default_branch), "Create target branch")
82
+ if scenario_files:
83
+ additions = {path: self._get_content(content) for path, content in scenario_files.items()}
84
+ self._require(
85
+ self.gh_client.batch_sync(additions, [], "provision scenario files", target_branch), "Sync scenario"
86
+ )
87
+ elif target_branch != default_branch:
88
+ self._require(
89
+ self.gh_client.put_file(".scanner-probe", "probe\n", "scanner probe commit", target_branch),
90
+ "Create PR comparison commit",
91
+ )
92
+
93
+ def _get_content(self, content_or_path):
94
+ """Reads file content or returns an inline value."""
95
+ content = content_or_path
96
+ is_binary = False
97
+ if isinstance(content_or_path, str) and os.path.exists(content_or_path):
98
+ with open(content_or_path, "rb") as f:
99
+ content = f.read()
100
+ try:
101
+ content = content.decode("utf-8")
102
+ except UnicodeDecodeError:
103
+ is_binary = True
104
+
105
+ if is_binary:
106
+ return content.decode("latin-1")
107
+ return content
108
+
109
+ def teardown(self):
110
+ """Clean up a repository created by this provisioner, retaining ownership on failure."""
111
+ if not self._owns_repo:
112
+ return
113
+ self._require(self.gh_client.delete_owned_repo(), "Delete owned repository")
114
+ self._owns_repo = False
@@ -0,0 +1,33 @@
1
+ import os
2
+
3
+ from .gh_client import GitHubClient
4
+ from .provisioner import RepoProvisioner
5
+
6
+
7
+ def setup_attacker_fork(scenario, owner_client):
8
+ """Create a fresh attacker fork and record ownership before subsequent setup."""
9
+ token = os.environ.get("ATTACKER_GITHUB_TOKEN")
10
+ if not token:
11
+ raise ValueError("Required GitHub actor is unavailable: attacker")
12
+ client = GitHubClient(token=token, actor="attacker", record_event=owner_client.record_event)
13
+ login = client.get_authenticated_user_login()
14
+ client.repo_name = f"{login}/{owner_client.repository.name}"
15
+ RepoProvisioner._require(client.fork_repo(owner_client.repo_name), "Create attacker fork")
16
+ scenario._attacker_fork_client = client
17
+ scenario.runtime_state.update(attacker_login=login, fork_full_name=client.repo_name)
18
+ client.wait_until_ready()
19
+ RepoProvisioner._require(client.create_branch(scenario.branch, client.get_default_branch()), "Create attacker branch")
20
+ additions = {}
21
+ for path, local_path in scenario.get_required_files().items():
22
+ with open(local_path) as handle:
23
+ additions[path] = handle.read()
24
+ RepoProvisioner._require(
25
+ client.batch_sync(additions, [], "provision attacker fixtures", scenario.branch), "Sync attacker fixtures"
26
+ )
27
+
28
+
29
+ def teardown_attacker_fork(scenario):
30
+ client = getattr(scenario, "_attacker_fork_client", None)
31
+ if client is not None:
32
+ RepoProvisioner._require(client.delete_owned_repo(), "Delete attacker fork")
33
+ scenario._attacker_fork_client = None
@@ -0,0 +1,49 @@
1
+ from enum import Enum
2
+
3
+
4
+ class ScenarioType(str, Enum):
5
+ BENIGN = "benign"
6
+ MALICIOUS = "malicious"
7
+
8
+
9
+ class GitHubEvent(str, Enum):
10
+ PULL_REQUEST = "pull_request"
11
+ ISSUES = "issues"
12
+ ISSUE_COMMENT = "issue_comment"
13
+ PULL_REQUEST_REVIEW = "pull_request_review"
14
+ PULL_REQUEST_REVIEW_COMMENT = "pull_request_review_comment"
15
+ PUSH = "push"
16
+ WORKFLOW_DISPATCH = "workflow_dispatch"
17
+
18
+
19
+ class WorkflowCategory(str, Enum):
20
+ CODE_MAINTENANCE = "code-maintenance"
21
+ SECURITY_COMPLIANCE = "security-compliance"
22
+ TRIAGE_MANAGEMENT = "triage-management"
23
+ CONTENT_DOCS = "content-docs"
24
+ CODE_REVIEW = "code-review"
25
+ SUPPORT_INTELLIGENCE = "support-intelligence"
26
+ REPORTING_ANALYTICS = "reporting-analytics"
27
+
28
+
29
+ class AIProvider(str, Enum):
30
+ GOOGLE_GEMINI = "google-gemini"
31
+ ANTHROPIC_CLAUDE = "anthropic-claude"
32
+ OPENAI_CODEX = "openai-codex"
33
+ CURSOR = "cursor"
34
+ GITHUB_COPILOT = "github-copilot"
35
+ CLINE = "cline"
36
+ OPEN_CODE = "open-code"
37
+ OPEN_HANDS = "open-hands"
38
+ AMAZON_Q = "amazon-q"
39
+ GOOSE = "goose"
40
+ DEVIN = "devin"
41
+ CISCO_AI = "cisco-ai"
42
+ AIDER = "aider"
43
+ OPENROUTER = "openrouter"
44
+
45
+
46
+ class DefenseLevel(str, Enum):
47
+ BASELINE = "baseline"
48
+ HARDENED = "hardened"
49
+ SANDBOXED = "sandboxed"
File without changes