benchpress-agent 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
benchpress/__init__.py ADDED
@@ -0,0 +1,36 @@
1
+ """Benchpress — the reliability layer for AI agents with write access.
2
+
3
+ Policy-first, authority-gated, evidence-verified execution for multi-app agents.
4
+ Nothing in this package is keyed to a task id, a seeded name, or a seeded domain:
5
+ every task fact is discovered at runtime from the prompt and from provider reads.
6
+
7
+ import benchpress
8
+
9
+ agent = benchpress.wrap("deepseek-v4-pro", gateway.execute_tool, providers=["hubspot", "stripe"])
10
+ result = await agent.run("Move Acme's renewal notices to ap@acme.example")
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from benchpress.api import DEFAULT_SYSTEM_PROMPT, Benchpress, wrap
16
+ from benchpress.controller import TrialResult, run_trial
17
+ from benchpress.gate import Gate, GateRefusal
18
+ from benchpress.model import ModelConfig
19
+ from benchpress.phases.common import Ablations
20
+ from benchpress.tools import ToolExecutor
21
+
22
+ __version__ = "0.1.0"
23
+
24
+ __all__ = [
25
+ "DEFAULT_SYSTEM_PROMPT",
26
+ "Ablations",
27
+ "Benchpress",
28
+ "Gate",
29
+ "GateRefusal",
30
+ "ModelConfig",
31
+ "ToolExecutor",
32
+ "TrialResult",
33
+ "__version__",
34
+ "run_trial",
35
+ "wrap",
36
+ ]
benchpress/adapter.py ADDED
@@ -0,0 +1,138 @@
1
+ """The ArgaBench candidate contract: `invoke(model_id, system_prompt, user_prompt, tool_schema,
2
+ execute_tool, max_tool_calls, timeout_seconds, *, api_effort, thinking)`.
3
+
4
+ `src/benchpress` never imports the harness, so this returns an `InvocationRecord` with the
5
+ same fields as the harness's `ModelInvocationResult`; the harness-side shim converts it.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import os
11
+ from collections.abc import Mapping, Sequence
12
+ from dataclasses import asdict, dataclass, field, replace
13
+ from pathlib import Path
14
+ from typing import Any, Literal, cast
15
+
16
+ from benchpress.controller import run_trial
17
+ from benchpress.model import ModelConfig
18
+ from benchpress.phases.common import Ablations
19
+ from benchpress.tools import ToolExecutor
20
+
21
+ KNOWN_ROLES: frozenset[str] = frozenset(
22
+ {
23
+ "code_host",
24
+ "email",
25
+ "calendar",
26
+ "file_storage",
27
+ "hubspot_crm",
28
+ "jira_tracker",
29
+ "linear_tracker",
30
+ "professional_network",
31
+ "knowledge_base",
32
+ "salesforce_crm",
33
+ "team_chat",
34
+ "payments",
35
+ }
36
+ )
37
+
38
+
39
+ @dataclass(frozen=True)
40
+ class InvocationRecord:
41
+ requested_model: str
42
+ response_model: str | None
43
+ provider: Literal["anthropic", "openai", "google"]
44
+ final_text: str
45
+ status: str
46
+ stop_reason: str
47
+ system_prompt: str
48
+ user_prompt: str
49
+ events: tuple[dict[str, Any], ...] = ()
50
+ usage: dict[str, Any] = field(default_factory=lambda: dict[str, Any]())
51
+ config: dict[str, Any] = field(default_factory=lambda: dict[str, Any]())
52
+ latency_ms: int = 0
53
+ tool_calls: int = 0
54
+
55
+ def as_dict(self) -> dict[str, Any]:
56
+ return asdict(self)
57
+
58
+
59
+ def providers_from_tool_schema(tool_schema: object) -> list[str]:
60
+ """Provisioned provider names from the `provider_api` tool's enum (roles excluded)."""
61
+ schemas: Sequence[Mapping[str, Any]]
62
+ if isinstance(tool_schema, Mapping):
63
+ schemas = [cast(Mapping[str, Any], tool_schema)]
64
+ else:
65
+ schemas = cast(Sequence[Mapping[str, Any]], tool_schema)
66
+ for schema in schemas:
67
+ if schema.get("name") != "provider_api":
68
+ continue
69
+ parameters = cast(Mapping[str, Any], schema.get("input_schema", schema.get("parameters", {})))
70
+ properties = cast(Mapping[str, Any], parameters.get("properties", {}))
71
+ provider = cast(Mapping[str, Any], properties.get("provider", {}))
72
+ enum = cast(Sequence[object], provider.get("enum", []))
73
+ return [str(item) for item in enum if str(item) not in KNOWN_ROLES]
74
+ return []
75
+
76
+
77
+ async def invoke(
78
+ model_id: str,
79
+ system_prompt: str,
80
+ user_prompt: str,
81
+ tool_schema: object,
82
+ execute_tool: ToolExecutor,
83
+ max_tool_calls: int,
84
+ timeout_seconds: float,
85
+ *,
86
+ api_effort: str = "default",
87
+ thinking: str = "model_default",
88
+ ablations: Ablations | None = None,
89
+ trace_dir: Path | None = None,
90
+ trial_id: str | None = None,
91
+ ) -> InvocationRecord:
92
+ inner_model = model_id.split("/", 1)[1] if model_id.startswith("benchpress/") else model_id
93
+ config = ModelConfig.from_env(model=inner_model)
94
+ if api_effort and api_effort != "default":
95
+ config = replace(config, effort=api_effort)
96
+ resolved_ablations = ablations or Ablations.parse(os.environ.get("BENCHPRESS_ABLATIONS"))
97
+ result = await run_trial(
98
+ system_prompt=system_prompt,
99
+ user_prompt=user_prompt,
100
+ providers=providers_from_tool_schema(tool_schema),
101
+ execute_tool=execute_tool,
102
+ config=config,
103
+ ablations=resolved_ablations,
104
+ trace_dir=trace_dir,
105
+ trial_id=trial_id,
106
+ )
107
+ if result.error == "timeout":
108
+ status, stop_reason = "timed_out", "timeout"
109
+ elif result.error:
110
+ status, stop_reason = "incomplete", result.error[:80]
111
+ else:
112
+ status, stop_reason = "completed", "end_turn"
113
+ provider: Literal["anthropic", "openai", "google"] = "anthropic" if inner_model.startswith("claude") else "openai"
114
+ return InvocationRecord(
115
+ requested_model=model_id,
116
+ response_model=model_id,
117
+ final_text=result.final_text,
118
+ provider=provider,
119
+ status=status,
120
+ stop_reason=stop_reason,
121
+ system_prompt=system_prompt,
122
+ user_prompt=user_prompt,
123
+ events=result.tool_events,
124
+ usage=result.usage.harness_usage(),
125
+ config={
126
+ "model": inner_model,
127
+ "provider": config.provider,
128
+ "effort": config.effort,
129
+ "thinking": {"type": thinking},
130
+ "scaffold": "benchpress",
131
+ "ablations": list(result.ablations),
132
+ "max_tool_calls": max_tool_calls,
133
+ "timeout_seconds": timeout_seconds,
134
+ "benchpress_status": result.status,
135
+ },
136
+ latency_ms=result.latency_ms,
137
+ tool_calls=result.provider_calls + result.docs_calls,
138
+ )
benchpress/api.py ADDED
@@ -0,0 +1,133 @@
1
+ """The public entry point: `wrap(model, executor)` puts the Benchpress loop around any tool layer.
2
+
3
+ agent = benchpress.wrap("deepseek-v4-pro", gateway.execute_tool, providers=["hubspot", "stripe"])
4
+ result = await agent.run("Move Acme's renewal notices to ap@acme.example")
5
+
6
+ The executor is anything with the harness `execute_tool(tool_name, tool_input)` shape, either a
7
+ coroutine function or an object exposing `execute_tool`. Every write it receives has already passed
8
+ the gate; every run ends with an evidence-only status and, when `trace_dir` is set, a receipt.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import asyncio
14
+ from collections.abc import Awaitable, Mapping, Sequence
15
+ from dataclasses import dataclass, field
16
+ from pathlib import Path
17
+ from typing import Any, Protocol, runtime_checkable
18
+
19
+ from benchpress.controller import TrialResult, run_trial
20
+ from benchpress.model import ModelClient, ModelConfig, ModelTransport
21
+ from benchpress.phases.common import Ablations
22
+ from benchpress.playbooks import Playbook
23
+ from benchpress.prompts import system_text
24
+ from benchpress.tools import ToolExecutor
25
+
26
+ DEFAULT_SYSTEM_PROMPT = (
27
+ "You are an operations agent working across the provisioned business systems. Complete the request "
28
+ "through the provider_api tool using ordinary data-plane routes only. Minimize mutations, respect every "
29
+ "explicit prohibition, verify the final state through reads, and return exactly the output requested."
30
+ )
31
+
32
+
33
+ @runtime_checkable
34
+ class HasExecuteTool(Protocol):
35
+ def execute_tool(self, tool_name: str, tool_input: dict[str, Any]) -> Awaitable[object]: ...
36
+
37
+
38
+ type ExecutorLike = ToolExecutor | HasExecuteTool
39
+ type ModelLike = str | ModelConfig | None
40
+
41
+
42
+ @dataclass(frozen=True)
43
+ class Benchpress:
44
+ """A configured loop. Stateless between runs: each `run` gets a fresh context, gate and model client."""
45
+
46
+ executor: ToolExecutor
47
+ providers: tuple[str, ...]
48
+ config: ModelConfig
49
+ system_prompt: str = DEFAULT_SYSTEM_PROMPT
50
+ ablations: Ablations = field(default_factory=Ablations)
51
+ playbooks: Mapping[str, Playbook] | None = None
52
+ transport: ModelTransport | None = None
53
+
54
+ async def run(
55
+ self,
56
+ request: str,
57
+ *,
58
+ trace_dir: str | Path | None = None,
59
+ trial_id: str | None = None,
60
+ ) -> TrialResult:
61
+ if not request.strip():
62
+ raise ValueError("request must be a non-empty string")
63
+ client = None
64
+ if self.transport is not None:
65
+ client = ModelClient(self.config, system_text(self.system_prompt), transport=self.transport)
66
+ return await run_trial(
67
+ system_prompt=self.system_prompt,
68
+ user_prompt=request,
69
+ providers=self.providers,
70
+ execute_tool=self.executor,
71
+ config=self.config,
72
+ ablations=self.ablations,
73
+ trace_dir=Path(trace_dir) if trace_dir is not None else None,
74
+ trial_id=trial_id,
75
+ model_client=client,
76
+ playbooks=self.playbooks,
77
+ )
78
+
79
+ def run_sync(
80
+ self,
81
+ request: str,
82
+ *,
83
+ trace_dir: str | Path | None = None,
84
+ trial_id: str | None = None,
85
+ ) -> TrialResult:
86
+ """`run` for synchronous callers. Raises if called from inside a running event loop."""
87
+ return asyncio.run(self.run(request, trace_dir=trace_dir, trial_id=trial_id))
88
+
89
+
90
+ def _config(model: ModelLike) -> ModelConfig:
91
+ if isinstance(model, ModelConfig):
92
+ return model
93
+ return ModelConfig.from_env(model=model)
94
+
95
+
96
+ def _executor(executor: ExecutorLike) -> ToolExecutor:
97
+ if isinstance(executor, HasExecuteTool):
98
+ return executor.execute_tool
99
+ if callable(executor):
100
+ return executor
101
+ raise TypeError("executor must be an async callable (tool_name, tool_input) or expose execute_tool")
102
+
103
+
104
+ def wrap(
105
+ model: ModelLike,
106
+ executor: ExecutorLike,
107
+ *,
108
+ providers: Sequence[str],
109
+ system_prompt: str = DEFAULT_SYSTEM_PROMPT,
110
+ ablations: Ablations | None = None,
111
+ playbooks: Mapping[str, Playbook] | None = None,
112
+ transport: ModelTransport | None = None,
113
+ ) -> Benchpress:
114
+ """Wrap a tool layer in the Benchpress loop.
115
+
116
+ `model` is a model id (credentials from the environment: `DEEPSEEK_API_KEY`/`BENCHPRESS_API_KEY`
117
+ for OpenAI-compatible endpoints, `ANTHROPIC_API_KEY` for `claude-*`), a full `ModelConfig`, or
118
+ None for `BENCHPRESS_MODEL`. `providers` names the systems the executor can reach; built-in
119
+ playbooks cover slack, gmail, hubspot and stripe, and `playbooks` supplies your own. `transport`
120
+ swaps the model wire protocol (bring your own LLM client, or a scripted one in tests).
121
+ """
122
+ names = tuple(name.strip() for name in providers if name.strip())
123
+ if not names:
124
+ raise ValueError("providers must name at least one system the executor can reach")
125
+ return Benchpress(
126
+ executor=_executor(executor),
127
+ providers=names,
128
+ config=_config(model),
129
+ system_prompt=system_prompt,
130
+ ablations=ablations or Ablations(),
131
+ playbooks=playbooks,
132
+ transport=transport,
133
+ )
benchpress/cli.py ADDED
@@ -0,0 +1,112 @@
1
+ """`benchpress` command line: run a request against real apps or local twins, print receipts."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import asyncio
7
+ import importlib
8
+ import json
9
+ import os
10
+ import sys
11
+ from collections.abc import Sequence
12
+ from datetime import UTC, datetime
13
+ from pathlib import Path
14
+ from typing import Any, Protocol, cast
15
+
16
+ from benchpress.api import DEFAULT_SYSTEM_PROMPT
17
+ from benchpress.controller import run_trial
18
+ from benchpress.model import ModelConfig
19
+ from benchpress.phases.common import Ablations
20
+ from benchpress.receipt_html import write_receipt_html
21
+ from benchpress.report import receipt_summary
22
+
23
+
24
+ class GatewayLike(Protocol):
25
+ async def execute_tool(self, tool_name: str, tool_input: dict[str, Any]) -> object: ...
26
+
27
+ async def aclose(self) -> None: ...
28
+
29
+
30
+ class GatewayFactory(Protocol):
31
+ def __call__(self, providers: Sequence[str]) -> GatewayLike: ...
32
+
33
+
34
+ def _load_env() -> None:
35
+ try:
36
+ from dotenv import load_dotenv
37
+ except ImportError: # pragma: no cover - dotenv is a dev dependency
38
+ return
39
+ load_dotenv(Path.cwd() / ".env")
40
+
41
+
42
+ async def _run(args: argparse.Namespace) -> int:
43
+ prompt = Path(args.prompt_file).read_text() if args.prompt_file else str(args.prompt or "")
44
+ if not prompt.strip():
45
+ print("a --prompt or --prompt-file is required", file=sys.stderr)
46
+ return 2
47
+ providers = [name.strip() for name in str(args.providers).split(",") if name.strip()]
48
+ try:
49
+ realapp = importlib.import_module("benchpress.realapp")
50
+ except ImportError as exc:
51
+ print(f"real-app gateway unavailable: {exc}", file=sys.stderr)
52
+ return 2
53
+ factory = cast(GatewayFactory, getattr(realapp, "gateway_from_env")) # noqa: B009 - module resolved at runtime
54
+ gateway = factory(providers)
55
+ trace_dir = Path(args.trace_dir or f"runs/local/{datetime.now(UTC).strftime('%Y%m%dT%H%M%SZ')}")
56
+ config = ModelConfig.from_env(model=args.model)
57
+ result = await run_trial(
58
+ system_prompt=DEFAULT_SYSTEM_PROMPT,
59
+ user_prompt=prompt,
60
+ providers=providers,
61
+ execute_tool=gateway.execute_tool,
62
+ config=config,
63
+ ablations=Ablations.parse(args.ablations),
64
+ trace_dir=trace_dir,
65
+ )
66
+ await gateway.aclose()
67
+ receipt = json.loads((trace_dir / "receipt.json").read_text())
68
+ print(receipt_summary(receipt))
69
+ print(f"\nfinal:\n{result.final_text}")
70
+ print(f"\nreceipt: {trace_dir / 'receipt.json'}")
71
+ return 0
72
+
73
+
74
+ def main(argv: list[str] | None = None) -> int:
75
+ _load_env()
76
+ parser = argparse.ArgumentParser(
77
+ prog="benchpress", description="the reliability layer for agents with write access"
78
+ )
79
+ sub = parser.add_subparsers(dest="command", required=True)
80
+ run = sub.add_parser("run", help="run one request end to end")
81
+ run.add_argument("--prompt")
82
+ run.add_argument("--prompt-file")
83
+ run.add_argument("--providers", default="slack,gmail,hubspot,stripe")
84
+ run.add_argument("--trace-dir")
85
+ run.add_argument("--model", default=os.environ.get("BENCHPRESS_MODEL"))
86
+ run.add_argument("--ablations", default=os.environ.get("BENCHPRESS_ABLATIONS"))
87
+ receipt = sub.add_parser("receipt", help="print a receipt summary")
88
+ receipt.add_argument("path")
89
+ receipt.add_argument(
90
+ "--html",
91
+ nargs="?",
92
+ const="",
93
+ metavar="OUT",
94
+ help="also render the self-contained receipt page; defaults to receipt.html next to the JSON",
95
+ )
96
+ args = parser.parse_args(argv)
97
+ if args.command == "run":
98
+ return asyncio.run(_run(args))
99
+ if args.command == "receipt":
100
+ path = Path(args.path)
101
+ if path.is_dir():
102
+ path = path / "receipt.json"
103
+ print(receipt_summary(json.loads(path.read_text())))
104
+ if args.html is not None:
105
+ out = Path(args.html) if args.html else path.with_suffix(".html")
106
+ print(f"\nreceipt page: {write_receipt_html(path, out)}")
107
+ return 0
108
+ return 1
109
+
110
+
111
+ if __name__ == "__main__":
112
+ sys.exit(main())