agent-harness-adk 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_harness_adk-0.1.0/PKG-INFO +456 -0
- agent_harness_adk-0.1.0/README.md +424 -0
- agent_harness_adk-0.1.0/pyproject.toml +83 -0
- agent_harness_adk-0.1.0/pyproject.toml.orig +63 -0
- agent_harness_adk-0.1.0/src/agent_harness/__init__.py +243 -0
- agent_harness_adk-0.1.0/src/agent_harness/agent.py +797 -0
- agent_harness_adk-0.1.0/src/agent_harness/cli.py +256 -0
- agent_harness_adk-0.1.0/src/agent_harness/context.py +190 -0
- agent_harness_adk-0.1.0/src/agent_harness/errors.py +84 -0
- agent_harness_adk-0.1.0/src/agent_harness/harness.py +110 -0
- agent_harness_adk-0.1.0/src/agent_harness/mcp/__init__.py +5 -0
- agent_harness_adk-0.1.0/src/agent_harness/mcp/client.py +330 -0
- agent_harness_adk-0.1.0/src/agent_harness/memory/__init__.py +36 -0
- agent_harness_adk-0.1.0/src/agent_harness/memory/base.py +173 -0
- agent_harness_adk-0.1.0/src/agent_harness/memory/manager.py +325 -0
- agent_harness_adk-0.1.0/src/agent_harness/memory/semantic.py +163 -0
- agent_harness_adk-0.1.0/src/agent_harness/orchestrator.py +512 -0
- agent_harness_adk-0.1.0/src/agent_harness/prompts.py +215 -0
- agent_harness_adk-0.1.0/src/agent_harness/providers/__init__.py +94 -0
- agent_harness_adk-0.1.0/src/agent_harness/providers/anthropic.py +261 -0
- agent_harness_adk-0.1.0/src/agent_harness/providers/base.py +304 -0
- agent_harness_adk-0.1.0/src/agent_harness/providers/fake.py +98 -0
- agent_harness_adk-0.1.0/src/agent_harness/providers/gemini.py +246 -0
- agent_harness_adk-0.1.0/src/agent_harness/providers/openai.py +251 -0
- agent_harness_adk-0.1.0/src/agent_harness/py.typed +0 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/__init__.py +49 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/budget.py +127 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/cache.py +97 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/checkpoints.py +117 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/guardrails.py +128 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/hooks.py +111 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/journal.py +89 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/permissions.py +118 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/router.py +89 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/scheduler.py +105 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/session.py +135 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/tracing.py +158 -0
- agent_harness_adk-0.1.0/src/agent_harness/runtime/workspace.py +307 -0
- agent_harness_adk-0.1.0/src/agent_harness/skills.py +199 -0
- agent_harness_adk-0.1.0/src/agent_harness/subagent.py +403 -0
- agent_harness_adk-0.1.0/src/agent_harness/toolkits/__init__.py +15 -0
- agent_harness_adk-0.1.0/src/agent_harness/toolkits/basics.py +124 -0
- agent_harness_adk-0.1.0/src/agent_harness/toolkits/web.py +103 -0
- agent_harness_adk-0.1.0/src/agent_harness/tools.py +373 -0
- agent_harness_adk-0.1.0/src/agent_harness/types.py +272 -0
|
@@ -0,0 +1,456 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: agent-harness-adk
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A fast, lightweight harness for building production AI agents: agents, sub-agents, skills, prompts, tools, MCP, memory and the runtime rails underneath them.
|
|
5
|
+
Keywords: agents,llm,ai,mcp,orchestration,anthropic,openai,gemini
|
|
6
|
+
Author: MuhammadHusnainAli
|
|
7
|
+
Author-email: MuhammadHusnainAli <muhammad.husnain.ali.738@gmail.com>
|
|
8
|
+
License: MIT
|
|
9
|
+
Classifier: Development Status :: 4 - Beta
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
17
|
+
Classifier: Topic :: Software Development :: Libraries :: Application Frameworks
|
|
18
|
+
Classifier: Typing :: Typed
|
|
19
|
+
Requires-Dist: pydantic>=2.7
|
|
20
|
+
Requires-Dist: httpx>=0.27
|
|
21
|
+
Requires-Dist: pyyaml>=6.0
|
|
22
|
+
Requires-Dist: rich>=13.0 ; extra == 'cli'
|
|
23
|
+
Requires-Dist: pytest>=8.0 ; extra == 'dev'
|
|
24
|
+
Requires-Dist: pytest-asyncio>=0.23 ; extra == 'dev'
|
|
25
|
+
Requires-Dist: ruff>=0.6 ; extra == 'dev'
|
|
26
|
+
Requires-Python: >=3.10
|
|
27
|
+
Project-URL: Homepage, https://github.com/MuhammadHusnainAli/agent-harness-adk
|
|
28
|
+
Project-URL: Issues, https://github.com/MuhammadHusnainAli/agent-harness-adk/issues
|
|
29
|
+
Provides-Extra: cli
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Description-Content-Type: text/markdown
|
|
32
|
+
|
|
33
|
+
# agent-harness-adk
|
|
34
|
+
|
|
35
|
+
[](https://github.com/MuhammadHusnainAli/agent-harness-adk/actions/workflows/ci.yml)
|
|
36
|
+
[](https://pypi.org/project/agent-harness-adk/)
|
|
37
|
+
[](https://pypi.org/project/agent-harness-adk/)
|
|
38
|
+
[](LICENSE)
|
|
39
|
+
|
|
40
|
+
A fast, lightweight harness for building production AI agents in Python.
|
|
41
|
+
|
|
42
|
+
Agents, sub-agents, skills, prompts, tools, MCP servers, memory — and the runtime
|
|
43
|
+
rails underneath them: permissions, budgets, hooks, guardrails, tracing,
|
|
44
|
+
checkpoints and isolated workspaces. Three model providers, one loop, no
|
|
45
|
+
framework lock-in.
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
pip install agent-harness-adk # or: uv add agent-harness-adk
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
import agent_harness # installed as agent-harness-adk, imported as agent_harness
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Python 3.10 – 3.14. Three dependencies (`pydantic`, `httpx`, `pyyaml`), ~100 ms
|
|
56
|
+
to import, and no vendor SDKs — the provider adapters speak HTTP directly so
|
|
57
|
+
Anthropic, OpenAI and Gemini all travel the same retry, cost and tracing path.
|
|
58
|
+
|
|
59
|
+
---
|
|
60
|
+
|
|
61
|
+
## 60 seconds
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
from agent_harness import Agent, tool
|
|
65
|
+
|
|
66
|
+
@tool
|
|
67
|
+
def order_status(order_id: str) -> str:
|
|
68
|
+
"""Look up the status of a customer order.
|
|
69
|
+
|
|
70
|
+
Args:
|
|
71
|
+
order_id: the order number, digits only.
|
|
72
|
+
"""
|
|
73
|
+
return db.lookup(order_id)
|
|
74
|
+
|
|
75
|
+
agent = Agent(
|
|
76
|
+
"support",
|
|
77
|
+
"Answer customer questions about orders. Look the order up before answering.",
|
|
78
|
+
tools=[order_status],
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
result = agent.run_sync("Where is order 4182?")
|
|
82
|
+
print(result.output, result.cost_usd, result.steps)
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
The decorator reads your signature and docstring and builds the JSON Schema the
|
|
86
|
+
model needs. Arguments coming back from the model are validated before your
|
|
87
|
+
function is called. `await agent.run(...)` is the real implementation;
|
|
88
|
+
`run_sync` is the wrapper for scripts and notebooks.
|
|
89
|
+
|
|
90
|
+
---
|
|
91
|
+
|
|
92
|
+
## The shape of the system
|
|
93
|
+
|
|
94
|
+
```
|
|
95
|
+
┌──────────────────────────────────────────────────────────┐
|
|
96
|
+
│ Orchestrator — plan · staff · run · consolidate · review │
|
|
97
|
+
└───────────────┬──────────────────────────────────────────┘
|
|
98
|
+
│ staffing decision: reuse or create?
|
|
99
|
+
┌──────────────┴───────────────┐
|
|
100
|
+
┌──────▼──────┐ ┌───────▼────────┐
|
|
101
|
+
│ The bench │ │ The factory │
|
|
102
|
+
│ pre-defined │ │ a new spec │
|
|
103
|
+
│ sub-agents │ │ written at run │
|
|
104
|
+
└──────┬──────┘ └───────┬────────┘
|
|
105
|
+
└──────────────┬───────────────┘
|
|
106
|
+
┌──────▼───────┐
|
|
107
|
+
│ Agent loop │ think → act → observe → repeat
|
|
108
|
+
└──────┬───────┘
|
|
109
|
+
┌─────────────────────┼─────────────────────────┐
|
|
110
|
+
│ context assembler │ tools · skills · MCP │ memory: user · session
|
|
111
|
+
│ context compactor │ workspace · providers │ orchestrator · sub-agent
|
|
112
|
+
└─────────────────────┴─────────────────────────┘
|
|
113
|
+
rails: permissions · budget · hooks · guardrails · tracing · journal ·
|
|
114
|
+
cache · checkpoints · sessions · scheduler · router
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
---
|
|
118
|
+
|
|
119
|
+
## Tools
|
|
120
|
+
|
|
121
|
+
```python
|
|
122
|
+
from agent_harness import tool, ToolContext
|
|
123
|
+
|
|
124
|
+
@tool(permission="ask", cacheable=True, tags=["billing"])
|
|
125
|
+
async def issue_refund(order_id: str, amount: float, ctx: ToolContext) -> str:
|
|
126
|
+
"""Refund a customer. Costs real money.
|
|
127
|
+
|
|
128
|
+
Args:
|
|
129
|
+
order_id: the order to refund.
|
|
130
|
+
amount: how much, in EUR.
|
|
131
|
+
"""
|
|
132
|
+
ctx.log("refunding", order=order_id)
|
|
133
|
+
return await billing.refund(order_id, amount)
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
- Sync or async, it makes no difference.
|
|
137
|
+
- A parameter named `ctx` (or annotated `ToolContext`) is injected and hidden
|
|
138
|
+
from the model.
|
|
139
|
+
- A pydantic model as a parameter type is validated and passed through as a
|
|
140
|
+
model, not a dict.
|
|
141
|
+
- `permission` can tighten the policy for one tool. It can never loosen it.
|
|
142
|
+
- Tools run in parallel when the model asks for several at once.
|
|
143
|
+
|
|
144
|
+
Built-ins: `agent_harness.toolkits` has `now`, `calculate`, `make_corpus_search`,
|
|
145
|
+
`make_fetch_tool` (domain allowlist, private-address refusal, HTML stripping) and
|
|
146
|
+
`make_http_tool`. A workspace brings `fs_read`, `fs_write`, `fs_list`,
|
|
147
|
+
`fs_delete` and — only when you ask for it — `shell`.
|
|
148
|
+
|
|
149
|
+
## Skills
|
|
150
|
+
|
|
151
|
+
A skill is packaged know-how: a folder with `SKILL.md` and, optionally, its own
|
|
152
|
+
tools and reference files.
|
|
153
|
+
|
|
154
|
+
```
|
|
155
|
+
skills/refunds/SKILL.md
|
|
156
|
+
---
|
|
157
|
+
name: refunds
|
|
158
|
+
description: How we process a refund, including the approval thresholds.
|
|
159
|
+
---
|
|
160
|
+
1. Check the order is inside the 30-day window...
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
```python
|
|
164
|
+
agent = Agent("support", "Answer support questions.", skills="./skills")
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
Only each skill's **name and description** go into the system prompt. The body
|
|
168
|
+
is loaded on demand through the `load_skill` tool, so twenty skills cost twenty
|
|
169
|
+
lines of context instead of twenty documents. A `tools.py` in the skill folder is
|
|
170
|
+
imported and its tools come along with it.
|
|
171
|
+
|
|
172
|
+
## Prompts
|
|
173
|
+
|
|
174
|
+
```python
|
|
175
|
+
from agent_harness import Prompt, PromptLibrary
|
|
176
|
+
|
|
177
|
+
triage = Prompt("triage", "Sort {ticket} into {buckets}.", version="2")
|
|
178
|
+
triage.render(ticket="T-1", buckets="p1/p2/p3")
|
|
179
|
+
|
|
180
|
+
library = PromptLibrary.from_dir("./prompts") # .md files with YAML frontmatter
|
|
181
|
+
library.render("triage", ticket="T-1")
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
Versioned, reviewable, `.partial()`-able, composable with `+`. Jinja is used
|
|
185
|
+
only when a template contains a `{% %}` statement and `jinja2` is installed.
|
|
186
|
+
|
|
187
|
+
## Memory — four scopes
|
|
188
|
+
|
|
189
|
+
| Scope | Stored | Loaded | Lifetime |
|
|
190
|
+
|---|---|---|---|
|
|
191
|
+
| **user** (`user.md`) | preferences, standards, settled decisions | in full, every message | permanent, rewritten at session close |
|
|
192
|
+
| **session** | the whole conversation plus its artefacts | in full | this session |
|
|
193
|
+
| **orchestrator** | plans, staffing decisions, spend, findings | **a digest only** | this job, then distilled into user memory |
|
|
194
|
+
| **sub-agent** | only the resources its task produced | nothing carried in | the task |
|
|
195
|
+
|
|
196
|
+
```python
|
|
197
|
+
agent = Agent("assistant", memory=True) # the default
|
|
198
|
+
await agent.run("I bill my customers in EUR")
|
|
199
|
+
await agent.run("What currency do I use?", messages=[]) # clean run, still knows
|
|
200
|
+
|
|
201
|
+
print(await agent.close_session()) # session close → user.md rewritten
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
The agent gets `remember` and `recall` tools. Recall is semantic: embeddings
|
|
205
|
+
come from whatever you configure, and the default is a deterministic offline
|
|
206
|
+
hashing embedder so semantic recall works with no extra dependency and no
|
|
207
|
+
network. Swap it for the real thing when you want to:
|
|
208
|
+
|
|
209
|
+
```python
|
|
210
|
+
from agent_harness import MemoryManager, ProviderEmbedder, OpenAIProvider, FileStore
|
|
211
|
+
|
|
212
|
+
memory = MemoryManager(FileStore(".harness/memory"),
|
|
213
|
+
embedder=ProviderEmbedder(OpenAIProvider()))
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
## Sub-agents: the bench and the factory
|
|
217
|
+
|
|
218
|
+
Before staffing a task, the orchestrator asks one question: **is there already a
|
|
219
|
+
sub-agent that covers this?**
|
|
220
|
+
|
|
221
|
+
```python
|
|
222
|
+
from agent_harness import Agent, SubAgentSpec
|
|
223
|
+
|
|
224
|
+
manager = Agent(
|
|
225
|
+
"manager",
|
|
226
|
+
"Delegate the lookups, then consolidate what comes back.",
|
|
227
|
+
tools=[lookup],
|
|
228
|
+
subagents=[
|
|
229
|
+
SubAgentSpec(name="revenue_reader", description="Finds revenue figures.",
|
|
230
|
+
instructions="Look up the figure and report it with its source.",
|
|
231
|
+
tools=["lookup"], tier="fast"),
|
|
232
|
+
SubAgentSpec(name="cost_reader", description="Finds cost figures.",
|
|
233
|
+
tools=["lookup"], tier="fast"),
|
|
234
|
+
],
|
|
235
|
+
)
|
|
236
|
+
result = await manager.run("How did Q3 go?")
|
|
237
|
+
for child in result.children:
|
|
238
|
+
print(child.agent, child.steps, child.cost_usd)
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
A `delegate` tool appears automatically. Sub-agents **start clean** — no parent
|
|
242
|
+
transcript, no parent memory — and hand back a result, not a conversation.
|
|
243
|
+
Delegation does not cascade by default, and a spec's `tools` list is a hard
|
|
244
|
+
allowlist. Ask for several delegations in one turn and they run in parallel
|
|
245
|
+
under the concurrency cap.
|
|
246
|
+
|
|
247
|
+
Nothing on the bench fits? The factory writes a new specialist during the run —
|
|
248
|
+
name, instructions, tool allowlist, model tier, step ceiling and workspace
|
|
249
|
+
isolation — and that specialist exists only for this job.
|
|
250
|
+
|
|
251
|
+
```python
|
|
252
|
+
from agent_harness import Bench
|
|
253
|
+
Bench.standard().names
|
|
254
|
+
# ['compliance_checker', 'data_analyst', 'document_extractor', 'drafting',
|
|
255
|
+
# 'planner', 'report_writer', 'research', 'validator']
|
|
256
|
+
```
|
|
257
|
+
|
|
258
|
+
## The orchestrator
|
|
259
|
+
|
|
260
|
+
```python
|
|
261
|
+
from agent_harness import Orchestrator, Budget
|
|
262
|
+
|
|
263
|
+
boss = Orchestrator("boss", max_concurrency=4, review=True, max_rework=1,
|
|
264
|
+
budget=Budget(max_usd=2.00))
|
|
265
|
+
result = await boss.run("Summarise how Q3 went, with the numbers cited.")
|
|
266
|
+
```
|
|
267
|
+
|
|
268
|
+
1. **Plan** — acceptance tests are written *before* any work starts, then the
|
|
269
|
+
task graph, then a cost estimate.
|
|
270
|
+
2. **Staff** — reuse from the bench, else build with the factory.
|
|
271
|
+
3. **Run** — dependency-ordered waves, parallel inside each wave, per-task
|
|
272
|
+
retries, dependent tasks receive only what they depend on.
|
|
273
|
+
4. **Consolidate** — merge, de-duplicate, rank, attribute.
|
|
274
|
+
5. **Review** — an independent critic checks the deliverable against the
|
|
275
|
+
definition of done; a rejection becomes new tasks and one rework round.
|
|
276
|
+
|
|
277
|
+
`result.data["plan"]` and `result.data["review"]` carry the full record.
|
|
278
|
+
|
|
279
|
+
## MCP
|
|
280
|
+
|
|
281
|
+
```python
|
|
282
|
+
from agent_harness import Agent, MCPManager, MCPServer
|
|
283
|
+
|
|
284
|
+
servers = [
|
|
285
|
+
MCPServer(name="files", command="npx",
|
|
286
|
+
args=["-y", "@modelcontextprotocol/server-filesystem", "/data"]),
|
|
287
|
+
MCPServer(name="api", url="https://mcp.internal/rpc",
|
|
288
|
+
headers={"authorization": "Bearer ..."}),
|
|
289
|
+
]
|
|
290
|
+
|
|
291
|
+
async with MCPManager(servers) as mcp:
|
|
292
|
+
agent = Agent("analyst", "Answer from the files.", tools=mcp.tools())
|
|
293
|
+
print((await agent.run("What is in /data/report.md?")).output)
|
|
294
|
+
```
|
|
295
|
+
|
|
296
|
+
Both transports (stdio and streamable HTTP), tools, resources and prompts. A
|
|
297
|
+
server that will not connect is reported in `mcp.errors`, not raised into your
|
|
298
|
+
run. `allowed_tools` trims what a server may expose.
|
|
299
|
+
|
|
300
|
+
## The rails
|
|
301
|
+
|
|
302
|
+
```python
|
|
303
|
+
from agent_harness import (Harness, Budget, PolicyGate, HookEngine, Guardrails,
|
|
304
|
+
console_exporter)
|
|
305
|
+
|
|
306
|
+
harness = Harness.local(".harness") # sessions, memory, traces, checkpoints
|
|
307
|
+
harness.policy = PolicyGate("allow", ask=["issue_refund"], deny=["shell"],
|
|
308
|
+
approver=my_approver)
|
|
309
|
+
harness.guardrails = Guardrails(strict=True)
|
|
310
|
+
harness.tracer.add_exporter(console_exporter())
|
|
311
|
+
harness.reset_budget(Budget(max_usd=0.50, max_steps=8, max_subagents=4))
|
|
312
|
+
|
|
313
|
+
hooks = HookEngine()
|
|
314
|
+
|
|
315
|
+
@hooks.on("pre_tool")
|
|
316
|
+
def cap_refunds(ctx):
|
|
317
|
+
if ctx.data["tool"] == "issue_refund" and ctx.data["args"]["amount"] > 100:
|
|
318
|
+
ctx.block("refunds over 100 EUR need a manager")
|
|
319
|
+
|
|
320
|
+
agent = Agent("refunds", harness=harness, hooks=hooks, tools=[issue_refund])
|
|
321
|
+
print(harness.report()) # spend by agent and task, cache hit rate, concurrency
|
|
322
|
+
```
|
|
323
|
+
|
|
324
|
+
| Rail | What it does |
|
|
325
|
+
|---|---|
|
|
326
|
+
| `PolicyGate` | allow / ask / deny per action, glob rules, conditional on arguments, approver callback |
|
|
327
|
+
| `BudgetGuard` | spend, token, step, tool-call and sub-agent ceilings; child guards roll up to the parent |
|
|
328
|
+
| `HookEngine` | 12 events; `pre_tool` can block or rewrite arguments, `post_tool` can rewrite the result |
|
|
329
|
+
| `Guardrails` | secret redaction, private-key blocking, injection warnings, size caps — on tool output *and* final answers |
|
|
330
|
+
| `Tracer` | one span per run, step, model call, tool and sub-agent; console and JSONL exporters |
|
|
331
|
+
| `RunJournal` | what each agent was asked and what it returned, append-only |
|
|
332
|
+
| `ResultCache` | identical task + identical input served from cache, memory and disk tiers |
|
|
333
|
+
| `Checkpointer` | step-level snapshots; resume or replay from any prior step |
|
|
334
|
+
| `SessionStore` | resume, fork or branch a run; a long job survives a restart |
|
|
335
|
+
| `WorkspaceBroker` | a jailed directory per sub-agent (or a shared one for handovers), local or Docker |
|
|
336
|
+
| `ConcurrencyScheduler` | semaphore, queue, backpressure, peak tracking |
|
|
337
|
+
| `ModelRouter` | per-task model and effort tier instead of one model for everything |
|
|
338
|
+
|
|
339
|
+
Path safety is enforced, not clamped: a workspace tool given `../../etc/passwd`
|
|
340
|
+
refuses rather than resolving it. `shell` is absent unless the workspace was
|
|
341
|
+
created with `allow_shell=True`, and even then it asks for approval.
|
|
342
|
+
|
|
343
|
+
## Providers
|
|
344
|
+
|
|
345
|
+
```python
|
|
346
|
+
Agent("a", model="claude-opus-5") # → Anthropic
|
|
347
|
+
Agent("b", model="gpt-4.1") # → OpenAI
|
|
348
|
+
Agent("c", model="gemini-2.5-pro") # → Gemini
|
|
349
|
+
Agent("d", provider=OpenAIProvider(base_url="http://localhost:11434/v1"))
|
|
350
|
+
```
|
|
351
|
+
|
|
352
|
+
The provider is inferred from the model id. Keys come from `ANTHROPIC_API_KEY`,
|
|
353
|
+
`OPENAI_API_KEY`, `GEMINI_API_KEY`. Anything that speaks the OpenAI wire format
|
|
354
|
+
(Azure, Groq, Together, Ollama, vLLM) works through `OpenAIProvider(base_url=...)`,
|
|
355
|
+
and `register_provider("name", MyProvider)` adds your own.
|
|
356
|
+
|
|
357
|
+
Adapters normalise everything the loop depends on: tool calls, tool results,
|
|
358
|
+
thinking blocks, cache tokens, stop reasons and refusals. Cost is computed per
|
|
359
|
+
call from a built-in price table (`register_model` to extend it), so
|
|
360
|
+
`result.cost_usd` is real money, not an estimate.
|
|
361
|
+
|
|
362
|
+
## Streaming and structured output
|
|
363
|
+
|
|
364
|
+
```python
|
|
365
|
+
async for event in agent.stream("Summarise the incident"):
|
|
366
|
+
if event.type == "text":
|
|
367
|
+
print(event.text, end="", flush=True)
|
|
368
|
+
elif event.type == "tool_result":
|
|
369
|
+
print(f"\n· {event.data['tool']}")
|
|
370
|
+
elif event.type == "run_end":
|
|
371
|
+
result = event.data["result"]
|
|
372
|
+
```
|
|
373
|
+
|
|
374
|
+
```python
|
|
375
|
+
from pydantic import BaseModel
|
|
376
|
+
|
|
377
|
+
class Ticket(BaseModel):
|
|
378
|
+
id: str
|
|
379
|
+
priority: int
|
|
380
|
+
summary: str
|
|
381
|
+
|
|
382
|
+
agent = Agent("triage", output_type=Ticket)
|
|
383
|
+
result = await agent.run("Customer cannot log in since the deploy")
|
|
384
|
+
result.data.priority # a validated Ticket, retried if the model got it wrong
|
|
385
|
+
```
|
|
386
|
+
|
|
387
|
+
## Testing your agents
|
|
388
|
+
|
|
389
|
+
```python
|
|
390
|
+
from agent_harness import Agent, FakeProvider, Harness, tool_call
|
|
391
|
+
|
|
392
|
+
provider = FakeProvider([tool_call("order_status", order_id="4182"),
|
|
393
|
+
"It ships Thursday."])
|
|
394
|
+
agent = Agent("support", provider=provider, harness=Harness.testing(provider),
|
|
395
|
+
tools=[order_status])
|
|
396
|
+
|
|
397
|
+
result = await agent.run("Where is order 4182?")
|
|
398
|
+
assert result.output == "It ships Thursday."
|
|
399
|
+
assert provider.requests[0].system.startswith("You are support")
|
|
400
|
+
```
|
|
401
|
+
|
|
402
|
+
No network, no keys, no recorded cassettes. Script strings, tool calls, whole
|
|
403
|
+
messages, exceptions, or a callable that inspects the request and answers
|
|
404
|
+
accordingly. The harness's own suite is 150 tests and runs in half a second.
|
|
405
|
+
|
|
406
|
+
## CLI
|
|
407
|
+
|
|
408
|
+
```bash
|
|
409
|
+
agent-harness run "summarise this incident" --tools --stream --state .harness
|
|
410
|
+
agent-harness chat --skills ./skills --state .harness --approve
|
|
411
|
+
agent-harness models
|
|
412
|
+
agent-harness sessions --state .harness
|
|
413
|
+
agent-harness journal --state .harness
|
|
414
|
+
agent-harness mcp npx -y @modelcontextprotocol/server-filesystem /data
|
|
415
|
+
```
|
|
416
|
+
|
|
417
|
+
## Design notes
|
|
418
|
+
|
|
419
|
+
- **Async core, sync wrapper.** Parallel sub-agents, MCP and the concurrency cap
|
|
420
|
+
all need it. `run_sync` covers scripts.
|
|
421
|
+
- **Compaction never orphans a tool call.** Fat tool results are hollowed out
|
|
422
|
+
first, and the summarise-the-head fallback moves its cut forward until no
|
|
423
|
+
`tool_result` is left without its `tool_use`. Naive trimming corrupts a
|
|
424
|
+
conversation; this does not.
|
|
425
|
+
- **Least privilege by default.** Sub-agents get an explicit tool allowlist,
|
|
426
|
+
delegation does not cascade, `shell` is opt-in, and a tool's own permission can
|
|
427
|
+
only tighten the policy.
|
|
428
|
+
- **Everything is optional.** An `Agent` with no memory, no skills and no
|
|
429
|
+
sub-agents is a tight `while` loop around one model call.
|
|
430
|
+
|
|
431
|
+
## Contributing
|
|
432
|
+
|
|
433
|
+
```bash
|
|
434
|
+
git clone https://github.com/MuhammadHusnainAli/agent-harness-adk
|
|
435
|
+
cd agent-harness-adk
|
|
436
|
+
uv sync --extra dev
|
|
437
|
+
uv run pytest -q
|
|
438
|
+
uv run ruff check src tests examples
|
|
439
|
+
```
|
|
440
|
+
|
|
441
|
+
Every push to `main` runs the suite on Python 3.10, 3.11, 3.12, 3.13 and 3.14,
|
|
442
|
+
lints, builds the wheel and smoke-tests it. Releases are cut by pushing a tag —
|
|
443
|
+
see [RELEASING.md](RELEASING.md).
|
|
444
|
+
|
|
445
|
+
## Status
|
|
446
|
+
|
|
447
|
+
0.1.0 — the first release. The public API above is what we intend to keep.
|
|
448
|
+
Changes are recorded in [CHANGELOG.md](CHANGELOG.md).
|
|
449
|
+
|
|
450
|
+
Not in this release: a vector-database backend (the built-in index is exact
|
|
451
|
+
brute force, fine to ~50k records), OCR and document parsing, and provider-side
|
|
452
|
+
batch APIs.
|
|
453
|
+
|
|
454
|
+
## Licence
|
|
455
|
+
|
|
456
|
+
MIT — see [LICENSE](LICENSE).
|